cadence/cadence.py
pedro-bento ed8c0145ff
All checks were successful
CI / lint-and-test (push) Successful in 11s
Version 0.1.0 (#1)
Reviewed-on: #1
Co-authored-by: pedro-bento <mail@pbento.pt>
Co-committed-by: pedro-bento <mail@pbento.pt>
2026-09-30 18:33:59 +01:00

1522 lines
51 KiB
Python
Executable file

#!/usr/bin/env -S uv run --python 3.13 --script
# /// script
# requires-python = ">=3.13, <3.14"
# dependencies = [
# "faster-whisper>=1.2.1",
# "mlx-whisper>=0.4.3; platform_system == 'Darwin' and platform_machine == 'arm64'",
# "nvidia-cublas-cu12; platform_system == 'Linux'",
# "nvidia-cudnn-cu12; platform_system == 'Linux'",
# "torch>=2.8.0",
# "torchaudio>=2.8.0",
# "tqdm>=4.70.1",
# "whisperx>=3.8.6",
# ]
# ///
"""cadence: High-retention auto-editor for raw video recordings.
Features:
- Supports multi-file input concatenated in chronological sequence.
- Selectable audio track extraction for multi-track OBS recordings.
- CTC forced alignment (WhisperX/Wav2Vec2) for phoneme-level cut accuracy.
- Removes silences, regex-matched elongated sounds, and English filler words.
- Eliminates stuttered word repetitions without destroying intentional pauses.
- Frame-locked J-Cuts and L-Cuts to mask visible jump cuts without cumulative A/V drift.
- Generates a native .kdenlive project file with synchronized audio/video cuts.
- Generates synced YouTube .srt captions and an LLM-ready transcript (prompt + timeline) for downstream summaries.
- Built-in pacing presets (relaxed, balanced, punchy) to eliminate the need for shell wrappers.
"""
from __future__ import annotations
import argparse
import hashlib
import json
import logging
import math
import os
import platform
import re
import shutil
import subprocess
import sys
import tempfile
import uuid
import xml.etree.ElementTree as ET
from collections.abc import Callable
from dataclasses import asdict, dataclass
from pathlib import Path
from typing import Literal
logger = logging.getLogger("cadence")
__version__ = "0.1.0"
# Regex to catch elongated sounds like 'aaaa', 'uhhh', 'ummm', 'mmmm', 'eeeh', etc.
HESITATION_REGEX = re.compile(
r"^(a{2,}|e{2,}|u{2,}|m{2,}|h{2,}|o{2,}|"
r"uh+m*|um+|eh+|er+|erm+|ah+|ha+|hm+|mm+|mhm+|huh+)$",
re.IGNORECASE,
)
# English hesitation sounds (note: single 'a' is excluded as it is a standard article)
HESITATION_FILLERS = {
"ehm", "ehmm", "ehmmm", "uhm", "uhmm", "mh", "mhm", "mmm", "mmmm", "mmh",
"hmm", "eh", "ehh", "ah", "ahh", "uh", "uhh", "eee", "ee", "umm", "um",
"er", "erm", "hm", "uh-huh", "huh",
}
# Less aggressive discourse markers / mental pause words
DISCOURSE_FILLERS = {
"sort of", "kind of", "you know", "i mean", "to be honest", "like i say", "like i said",
}
# Built-in pacing profiles to eliminate shell script wrappers
PACING_PRESETS = {
"relaxed": {
"max_silence": 0.85,
"pad": 0.15,
"min_keep": 0.10,
"jl_frames": 0,
},
"balanced": {
"max_silence": 0.50,
"pad": 0.10,
"min_keep": 0.08,
"jl_frames": 2,
},
"punchy": {
"max_silence": 0.35,
"pad": 0.08,
"min_keep": 0.08,
"jl_frames": 4,
},
}
MLX_MODEL_MAP = {
"tiny": "mlx-community/whisper-tiny-mlx",
"base": "mlx-community/whisper-base-mlx",
"small": "mlx-community/whisper-small-mlx",
"medium": "mlx-community/whisper-medium-mlx",
"large-v3": "mlx-community/whisper-large-v3-mlx",
"large-v3-turbo": "mlx-community/whisper-large-v3-turbo",
}
VERBATIM_PROMPT = (
"Verbatim transcription with all hesitations, stutters, and verbal fillers like um, uh, er."
)
@dataclass(frozen=True, slots=True)
class Word:
text: str
start: float
end: float
@dataclass(frozen=True, slots=True)
class Segment:
start: float
end: float
action: Literal["keep", "drop"]
@dataclass(slots=True)
class VideoTrackData:
video: Path
duration: float
fps_num: int
fps_den: int
width: int
height: int
audio_track_idx: int
audio_stream_count: int
words: list[Word]
timeline: list[Segment]
@property
def fps(self) -> float:
return self.fps_num / self.fps_den
def check_dependencies() -> None:
for tool in ("ffmpeg", "ffprobe"):
if shutil.which(tool) is None:
sys.exit(f"[!] Error: Required dependency '{tool}' was not found in PATH.")
def run_ff(cmd: list[str], what: str, *, stream: bool = False) -> None:
if stream and logger.isEnabledFor(logging.INFO):
tail: list[str] = []
proc = subprocess.Popen(
cmd, stderr=subprocess.PIPE, text=True, errors="replace", bufsize=1
)
assert proc.stderr is not None
for line in proc.stderr:
sys.stderr.write(line)
tail.append(line)
if len(tail) > 40:
tail = tail[-40:]
proc.wait()
rc, last = proc.returncode, "".join(tail[-20:])
else:
completed = subprocess.run(
cmd, stderr=subprocess.PIPE, text=True, errors="replace"
)
rc = completed.returncode
last = "\n".join(completed.stderr.splitlines()[-20:])
if rc != 0:
raise SystemExit(f"[!] {what} failed (exit {rc}):\n{last}")
def check_ff_output(cmd: list[str], what: str) -> str:
completed = subprocess.run(
cmd, capture_output=True, text=True, errors="replace"
)
if completed.returncode != 0:
tail = "\n".join(completed.stderr.splitlines()[-20:])
raise SystemExit(f"[!] {what} failed (exit {completed.returncode}):\n{tail}")
return completed.stdout
def probe_duration(path: Path) -> float:
out = check_ff_output([
"ffprobe", "-v", "error", "-show_entries", "format=duration",
"-of", "default=noprint_wrappers=1:nokey=1", str(path),
], f"ffprobe duration for {path.name}")
try:
return max(0.0, float(out.strip()))
except ValueError:
raise SystemExit(f"[!] Unable to parse duration for {path.name}: '{out.strip()}'")
def probe_video_format(path: Path) -> tuple[int, int, int, int]:
out = check_ff_output([
"ffprobe", "-v", "error", "-select_streams", "v:0",
"-show_entries", "stream=r_frame_rate,avg_frame_rate,width,height",
"-of", "json", str(path),
], f"ffprobe video format for {path.name}")
try:
data = json.loads(out)
streams = data.get("streams", [])
if not streams:
raise ValueError(f"No video streams found in {path.name}")
stream = streams[0]
rate_str = stream.get("r_frame_rate")
def _valid_rate(value: object) -> str | None:
if not isinstance(value, str):
return None
value = value.strip()
if not value or value in ("N/A", "0/0", "0"):
return None
return value
rate_str = _valid_rate(rate_str)
if rate_str is None:
rate_str = _valid_rate(stream.get("avg_frame_rate"))
if rate_str is None:
rate_str = "30/1"
num_str, _, den_str = rate_str.partition("/")
try:
num = int(num_str) if num_str else 30
den = int(den_str) if den_str else 1
except ValueError:
num, den = 30, 1
if den == 0 or num == 0:
num, den = 30, 1
width = int(stream["width"])
height = int(stream["height"])
return num, den, width, height
except (KeyError, IndexError, ValueError) as e:
raise SystemExit(f"[!] Could not determine video format for {path.name}: {e}")
def probe_audio_stream_count(path: Path) -> int:
out = check_ff_output([
"ffprobe", "-v", "error", "-select_streams", "a",
"-show_entries", "stream=index",
"-of", "json", str(path),
], f"ffprobe audio streams for {path.name}")
try:
data = json.loads(out)
return len(data.get("streams", []))
except json.JSONDecodeError as e:
raise SystemExit(f"[!] Could not parse audio streams for {path.name}: {e}")
def extract_audio(video: Path, wav_16k: Path, track_index: int = 0) -> None:
logger.info(f"[*] Extracting 16kHz audio ({video.name}) from stream a:{track_index}...")
run_ff([
"ffmpeg", "-y", "-loglevel", "error", "-i", str(video),
"-map", f"0:a:{track_index}", "-ac", "1", "-ar", "16000",
"-c:a", "pcm_s16le", str(wav_16k),
], "audio extraction")
if not wav_16k.exists() or wav_16k.stat().st_size == 0:
raise SystemExit(
f"[!] Audio extraction produced an empty file. Verify stream a:{track_index} in {video.name}."
)
def detect_backend() -> str:
if platform.system() == "Darwin" and platform.machine() == "arm64":
try:
import mlx_whisper # noqa: F401
return "mlx"
except ImportError:
pass
return "faster-whisper"
def preload_nvidia_libs() -> None:
if platform.system() != "Linux":
return
try:
import ctypes
import glob
import nvidia.cublas.lib
import nvidia.cudnn.lib
for module in (nvidia.cublas.lib, nvidia.cudnn.lib):
for d in getattr(module, "__path__", []):
for lib in glob.glob(os.path.join(d, "lib*.so*")):
try:
ctypes.CDLL(lib, mode=ctypes.RTLD_GLOBAL)
except OSError:
pass
except Exception as e:
logger.debug(f"Note: Preload of NVIDIA dynamic libraries skipped/failed: {e}")
def _align_with_whisperx(
raw_segments: list[dict],
wav_16k: Path,
language: str,
align_cache: dict | None = None,
) -> list[Word]:
"""Applies CTC Forced Alignment using WhisperX / Wav2Vec2."""
try:
import torch
import whisperx
except ImportError:
logger.warning("[!] 'whisperx' or 'torch' not installed. Skipping forced alignment.")
return []
try:
align_device = (
"cuda"
if torch.cuda.is_available()
else ("mps" if getattr(torch.backends, "mps", None) and torch.backends.mps.is_available() else "cpu")
)
logger.info(f"[*] Running CTC forced alignment ({align_device})...")
if isinstance(align_cache, dict) and language in align_cache:
align_model, align_meta = align_cache[language]
else:
align_model, align_meta = whisperx.load_align_model(
language_code=language, device=align_device
)
if isinstance(align_cache, dict):
align_cache[language] = (align_model, align_meta)
audio_data = whisperx.load_audio(str(wav_16k))
aligned_result = whisperx.align(
raw_segments,
align_model,
align_meta,
audio_data,
align_device,
return_char_alignments=False,
)
aligned_words: list[Word] = []
for w in aligned_result.get("word_segments", []):
if "start" in w and "end" in w and w.get("word"):
text = str(w["word"]).strip()
s = float(w["start"])
e = float(w["end"])
if text and e > s:
aligned_words.append(Word(text=text, start=s, end=e))
logger.info(f"[*] CTC alignment succeeded: {len(aligned_words)} words aligned.")
return aligned_words
except Exception as e:
logger.warning(f"[*] CTC forced alignment failed ({e}). Falling back to Whisper timestamps.")
return []
def _transcribe_mlx(
wav: Path, model_size: str, language: str = "en"
) -> tuple[list[Word], list[dict]]:
import mlx_whisper
repo = MLX_MODEL_MAP.get(model_size, model_size)
result = mlx_whisper.transcribe(
str(wav),
path_or_hf_repo=repo,
language=language,
word_timestamps=True,
condition_on_previous_text=False,
initial_prompt=VERBATIM_PROMPT,
verbose=False,
)
words: list[Word] = []
raw_segments: list[dict] = []
for seg in result.get("segments", []):
raw_segments.append({
"start": float(seg["start"]),
"end": float(seg["end"]),
"text": str(seg.get("text", "")).strip(),
})
for w in seg.get("words", []) or []:
text = (w.get("word") or w.get("text") or "").strip()
if text:
words.append(Word(text=text, start=float(w["start"]), end=float(w["end"])))
return words, raw_segments
def _load_faster_whisper_model(model_size: str):
preload_nvidia_libs()
from faster_whisper import WhisperModel
target_device = "cuda" if platform.system() == "Linux" else "cpu"
target_compute = "float16" if target_device == "cuda" else "int8"
try:
return WhisperModel(model_size, device=target_device, compute_type=target_compute)
except Exception as e:
logger.info(f"[*] {target_device.upper()} init failed ({e}), falling back to CPU.")
return WhisperModel(model_size, device="cpu", compute_type="int8")
def _transcribe_faster_whisper(
wav: Path,
model_size: str,
language: str = "en",
model=None,
) -> tuple[list[Word], list[dict]]:
from tqdm import tqdm
if model is None:
model = _load_faster_whisper_model(model_size)
segments, meta = model.transcribe(
str(wav),
language=language,
word_timestamps=True,
vad_filter=False,
beam_size=5,
condition_on_previous_text=False,
initial_prompt=VERBATIM_PROMPT,
)
words: list[Word] = []
raw_segments: list[dict] = []
pbar = tqdm(
total=round(meta.duration, 2),
unit="s",
disable=not logger.isEnabledFor(logging.INFO),
desc="Transcribing",
)
last = 0.0
for seg in segments:
raw_segments.append({
"start": float(seg.start),
"end": float(seg.end),
"text": seg.text.strip(),
})
if seg.words:
for w in seg.words:
text = w.word.strip()
if text:
words.append(Word(text=text, start=float(w.start), end=float(w.end)))
pbar.update(max(0.0, seg.end - last))
last = seg.end
pbar.close()
return words, raw_segments
def _cache_key(model: str, backend: str, language: str) -> str:
return hashlib.sha256(
json.dumps([backend, language, model]).encode()
).hexdigest()[:12]
def _cache_path(video: Path, track_idx: int, model: str, backend: str, language: str) -> Path:
return video.with_name(
f"{video.name}.trk{track_idx}.{_cache_key(model, backend, language)}.whisper.json"
)
def _read_cache(path: Path) -> list[Word] | None:
try:
data = json.loads(path.read_text(encoding="utf-8"))
words = data.get("words") if isinstance(data, dict) else None
if not isinstance(words, list):
return None
return [Word(**w) for w in words]
except (OSError, ValueError, TypeError, KeyError):
return None
def _write_cache(path: Path, words: list[Word]) -> None:
if not words:
return
payload = {"words": [asdict(w) for w in words]}
try:
tmp = path.with_suffix(path.suffix + ".tmp")
tmp.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8")
os.replace(tmp, path)
except OSError as e:
logger.warning(f"[!] Could not write cache {path.name}: {e}")
def _write_text_atomic(path: Path, text: str) -> None:
tmp = path.with_suffix(path.suffix + ".tmp")
tmp.write_text(text, encoding="utf-8")
os.replace(tmp, path)
def load_or_transcribe(
video: Path,
wav_16k: Path,
model: str,
language: str,
track_idx: int,
no_cache: bool,
backend: str,
stt_model=None,
align_cache: dict | None = None,
extract: Callable[[], None] | None = None,
) -> list[Word]:
cache_path = _cache_path(video, track_idx, model, backend, language)
if not no_cache:
cached = _read_cache(cache_path)
if cached is not None:
logger.info(f"[*] Reusing cached transcription: {cache_path.name}")
return cached
if extract is not None:
extract()
logger.info(f"[*] Transcribing ({backend}) | Model: {model} | File: {video.name}")
if backend == "mlx":
words, raw_segments = _transcribe_mlx(wav_16k, model, language=language)
else:
words, raw_segments = _transcribe_faster_whisper(
wav_16k, model, language=language, model=stt_model
)
if raw_segments:
aligned_words = _align_with_whisperx(
raw_segments=raw_segments,
wav_16k=wav_16k,
language=language,
align_cache=align_cache,
)
if aligned_words:
words = aligned_words
_write_cache(cache_path, words)
return words
def normalize_token(s: str) -> str:
return s.strip(" .,?!\"'…-—–:;()[]{}*~`").lower()
def build_cuts(
words: list[Word],
duration: float,
single_fillers: set[str],
phrase_fillers: set[str],
max_silence: float,
pad: float,
) -> list[tuple[float, float, str]]:
if duration <= 0:
return []
cuts: list[tuple[float, float, str]] = []
n = len(words)
drop_indices: set[int] = set()
for phrase in phrase_fillers:
tokens = [normalize_token(t) for t in phrase.split() if normalize_token(t)]
plen = len(tokens)
if plen == 0 or plen > n:
continue
for i in range(n - plen + 1):
window = [normalize_token(words[i + k].text) for k in range(plen)]
if window == tokens:
for k in range(plen):
drop_indices.add(i + k)
for i in range(n - 1):
w1, w2 = words[i], words[i + 1]
t1, t2 = normalize_token(w1.text), normalize_token(w2.text)
if t1 and t1 == t2 and (w2.start - w1.end) < 0.8:
drop_indices.add(i)
for i in range(n - 3):
pair1 = (normalize_token(words[i].text), normalize_token(words[i + 1].text))
pair2 = (normalize_token(words[i + 2].text), normalize_token(words[i + 3].text))
if (
pair1[0] and pair1[1]
and pair1 == pair2
and (words[i + 2].start - words[i + 1].end) < 0.8
):
drop_indices.add(i)
drop_indices.add(i + 1)
for i, w in enumerate(words):
tok = normalize_token(w.text)
if not tok:
drop_indices.add(i)
continue
if tok in single_fillers or HESITATION_REGEX.match(tok):
drop_indices.add(i)
prev_end = 0.0
for i, w in enumerate(words):
silence_gap = w.start - prev_end
if silence_gap > max_silence:
s = 0.0 if prev_end == 0.0 else prev_end + pad
e = max(s, w.start - pad)
if e - s > 0.04:
cuts.append((s, e, "silence"))
if i in drop_indices:
left_bound = words[i - 1].end if (i > 0 and (i - 1) not in drop_indices) else 0.0
right_bound = (
words[i + 1].start if (i < n - 1 and (i + 1) not in drop_indices) else duration
)
cut_s = max(left_bound, w.start - pad / 2)
cut_e = min(right_bound, w.end + pad / 2)
if cut_e > cut_s:
cuts.append((cut_s, cut_e, "filler"))
prev_end = max(prev_end, w.end)
if duration - prev_end > max_silence:
s = prev_end + pad
if duration - s > 0.04:
cuts.append((min(s, duration), duration, "silence"))
if not cuts:
return []
cuts.sort(key=lambda x: x[0])
merged: list[tuple[float, float, str]] = [cuts[0]]
for s, e, r in cuts[1:]:
ps, pe, pr = merged[-1]
if s <= pe + 0.01:
merged[-1] = (ps, max(pe, e), pr if pr == r else "mixed")
else:
merged.append((s, e, r))
return merged
def build_timeline(
cuts: list[tuple[float, float, str]], duration: float, min_keep_dur: float = 0.08
) -> list[Segment]:
timeline: list[Segment] = []
cursor = 0.0
for s, e, _ in cuts:
s = min(max(s, 0.0), duration)
e = min(max(e, 0.0), duration)
if s > cursor:
timeline.append(Segment(start=cursor, end=s, action="keep"))
if e > s:
timeline.append(Segment(start=s, end=e, action="drop"))
cursor = max(cursor, e)
if cursor < duration:
timeline.append(Segment(start=cursor, end=duration, action="keep"))
filtered: list[Segment] = []
for seg in timeline:
if seg.action == "keep" and (seg.end - seg.start) < min_keep_dur:
filtered.append(Segment(start=seg.start, end=seg.end, action="drop"))
else:
filtered.append(seg)
if not filtered:
return [Segment(0.0, duration, "keep")]
consolidated: list[Segment] = [filtered[0]]
for s in filtered[1:]:
last = consolidated[-1]
if s.action == last.action:
consolidated[-1] = Segment(start=last.start, end=s.end, action=last.action)
else:
consolidated.append(s)
return consolidated
def frames_to_tc(frames: int, fps: float) -> str:
"""Format frame index securely to Kdenlive/MLT strictly compliant HH:MM:SS.mmm format."""
if frames <= 0:
return "00:00:00.000"
secs = frames / fps
h = int(secs / 3600)
m = int((secs % 3600) / 60)
s = int(secs % 60)
ms = int(round((secs - int(secs)) * 1000))
if ms >= 1000:
s += 1
ms -= 1000
if s >= 60:
m += 1
s -= 60
if m >= 60:
h += 1
m -= 60
return f"{h:02d}:{m:02d}:{s:02d}.{ms:03d}"
def compute_jl_cut_shifts(
keep_segments: list[Segment],
fps: float,
jl_mode: Literal["j", "l", "off"],
jl_frames: int,
) -> list[int]:
count = len(keep_segments)
shifts = [0] * count
if jl_mode == "off" or jl_frames <= 0 or count < 2:
return shifts
for i in range(count - 1):
seg = keep_segments[i]
next_seg = keep_segments[i + 1]
cur_in = int(round(seg.start * fps))
cur_out = int(round(seg.end * fps))
cur_dur = max(1, cur_out - cur_in)
next_in = int(round(next_seg.start * fps))
next_out = int(round(next_seg.end * fps))
next_dur = max(1, next_out - next_in)
handle = max(0, next_in - cur_out - 1)
if jl_mode == "j":
max_allowed = min(jl_frames, handle, cur_dur // 4, next_dur // 4)
shifts[i] = max_allowed
elif jl_mode == "l":
max_allowed = min(jl_frames, handle, cur_dur // 4, next_dur // 4)
shifts[i] = -max_allowed
shifts[count - 1] = 0
return shifts
def _generate_multi_kdenlive_project(
track_data_list: list[VideoTrackData],
output: Path,
jl_mode: Literal["j", "l", "off"] = "j",
jl_frames: int = 2,
) -> None:
base = track_data_list[0]
fps_num, fps_den = base.fps_num, base.fps_den
width, height = base.width, base.height
fps = base.fps
gcd = math.gcd(width, height)
dar_num = width // gcd
dar_den = height // gcd
total_timeline_frames = 0
total_kept_segments = 0
for td in track_data_list:
for seg in td.timeline:
if seg.action == "keep":
start_f = int(round(seg.start * fps))
end_f = int(round(seg.end * fps))
dur_f = max(1, end_f - start_f)
total_timeline_frames += dur_f
total_kept_segments += 1
last_frame_index = max(0, total_timeline_frames - 1)
logger.info(
f"[*] Generating Kdenlive Project: {total_kept_segments} cuts | "
f"{total_timeline_frames} frames ({total_timeline_frames / fps:.2f}s) across {len(track_data_list)} video(s)..."
)
mlt = ET.Element("mlt", {
"LC_NUMERIC": "C",
"version": "7.22.0",
"producer": "main_bin",
"root": str(base.video.parent.resolve()),
})
ET.SubElement(mlt, "profile", {
"description": "automatic",
"width": str(width),
"height": str(height),
"progressive": "1",
"sample_aspect_num": "1",
"sample_aspect_den": "1",
"display_aspect_num": str(dar_num),
"display_aspect_den": str(dar_den),
"frame_rate_num": str(fps_num),
"frame_rate_den": str(fps_den),
"colorspace": "709",
})
def add_prop(parent: ET.Element, name: str, val: str) -> None:
p = ET.SubElement(parent, "property", {"name": name})
p.text = val
producer0 = ET.SubElement(mlt, "producer", {
"id": "producer0",
})
add_prop(producer0, "length", frames_to_tc(total_timeline_frames, fps))
add_prop(producer0, "eof", "continue")
add_prop(producer0, "resource", "black")
add_prop(producer0, "mlt_service", "color")
add_prop(producer0, "kdenlive:playlistid", "black_track")
add_prop(producer0, "mlt_image_format", "rgba")
add_prop(producer0, "aspect_ratio", "1")
use_system = bool(track_data_list) and all(
td.audio_stream_count >= 2 for td in track_data_list
)
for idx, td in enumerate(track_data_list):
src_path = str(td.video.resolve())
kdenlive_bin_id = str(idx + 1)
astream_idx = str(td.audio_track_idx)
# Track 1 (Voice) Producer Chain
chain_a = ET.SubElement(mlt, "chain", {"id": f"chain0_{idx}"})
add_prop(chain_a, "resource", src_path)
add_prop(chain_a, "mlt_service", "avformat-novalidate")
add_prop(chain_a, "vstream", "-1")
add_prop(chain_a, "astream", astream_idx)
add_prop(chain_a, "audio_index", astream_idx)
add_prop(chain_a, "video_index", "-1")
add_prop(chain_a, "set.test_audio", "0")
add_prop(chain_a, "set.test_video", "1")
add_prop(chain_a, "kdenlive:id", kdenlive_bin_id)
# Track 2 (System) Producer Chain
if use_system:
sys_stream = "1" if td.audio_track_idx == 0 else "0"
chain_a2 = ET.SubElement(mlt, "chain", {"id": f"chain_a2_{idx}"})
add_prop(chain_a2, "resource", src_path)
add_prop(chain_a2, "mlt_service", "avformat-novalidate")
add_prop(chain_a2, "vstream", "-1")
add_prop(chain_a2, "astream", sys_stream)
add_prop(chain_a2, "audio_index", sys_stream)
add_prop(chain_a2, "video_index", "-1")
add_prop(chain_a2, "set.test_audio", "0")
add_prop(chain_a2, "set.test_video", "1")
add_prop(chain_a2, "kdenlive:id", kdenlive_bin_id)
# Video Producer Chain
chain_v = ET.SubElement(mlt, "chain", {"id": f"chain1_{idx}"})
add_prop(chain_v, "resource", src_path)
add_prop(chain_v, "mlt_service", "avformat-novalidate")
add_prop(chain_v, "vstream", "0")
add_prop(chain_v, "astream", "-1")
add_prop(chain_v, "audio_index", "-1")
add_prop(chain_v, "video_index", "0")
add_prop(chain_v, "set.test_audio", "1")
add_prop(chain_v, "set.test_video", "0")
add_prop(chain_v, "kdenlive:id", kdenlive_bin_id)
# 1. Audio Track 1 (Voice)
playlist0 = ET.SubElement(mlt, "playlist", {"id": "playlist0"})
add_prop(playlist0, "kdenlive:audio_track", "1")
playlist1 = ET.SubElement(mlt, "playlist", {"id": "playlist1"})
add_prop(playlist1, "kdenlive:audio_track", "1")
tractor0 = ET.SubElement(mlt, "tractor", {
"id": "tractor0",
"in": "00:00:00.000",
"out": frames_to_tc(last_frame_index, fps),
})
add_prop(tractor0, "kdenlive:audio_track", "1")
add_prop(tractor0, "kdenlive:timeline_active", "1")
add_prop(tractor0, "kdenlive:track_name", "Voice")
ET.SubElement(tractor0, "track", {"hide": "video", "producer": "playlist0"})
ET.SubElement(tractor0, "track", {"hide": "video", "producer": "playlist1"})
# 2. Audio Track 2 (System Audio)
if use_system:
playlist4 = ET.SubElement(mlt, "playlist", {"id": "playlist4"})
add_prop(playlist4, "kdenlive:audio_track", "1")
playlist5 = ET.SubElement(mlt, "playlist", {"id": "playlist5"})
add_prop(playlist5, "kdenlive:audio_track", "1")
tractor_a2 = ET.SubElement(mlt, "tractor", {
"id": "tractor_a2",
"in": "00:00:00.000",
"out": frames_to_tc(last_frame_index, fps),
})
add_prop(tractor_a2, "kdenlive:audio_track", "1")
add_prop(tractor_a2, "kdenlive:timeline_active", "1")
add_prop(tractor_a2, "kdenlive:track_name", "System")
ET.SubElement(tractor_a2, "track", {"hide": "video", "producer": "playlist4"})
ET.SubElement(tractor_a2, "track", {"hide": "video", "producer": "playlist5"})
# 3. Video Track 1
playlist2 = ET.SubElement(mlt, "playlist", {"id": "playlist2"})
ET.SubElement(mlt, "playlist", {"id": "playlist3"})
tractor1 = ET.SubElement(mlt, "tractor", {
"id": "tractor1",
"in": "00:00:00.000",
"out": frames_to_tc(last_frame_index, fps),
})
add_prop(tractor1, "kdenlive:timeline_active", "1")
ET.SubElement(tractor1, "track", {"hide": "audio", "producer": "playlist2"})
ET.SubElement(tractor1, "track", {"hide": "audio", "producer": "playlist3"})
for idx, td in enumerate(track_data_list):
src_path = str(td.video.resolve())
kdenlive_bin_id = str(idx + 1)
chain_bin = ET.SubElement(mlt, "chain", {"id": f"chain2_{idx}"})
add_prop(chain_bin, "resource", src_path)
add_prop(chain_bin, "mlt_service", "avformat-novalidate")
add_prop(chain_bin, "audio_index", str(td.audio_track_idx))
add_prop(chain_bin, "video_index", "0")
add_prop(chain_bin, "vstream", "0")
add_prop(chain_bin, "astream", str(td.audio_track_idx))
add_prop(chain_bin, "kdenlive:id", kdenlive_bin_id)
groups = []
current_tl_audio_frame = 0
current_tl_video_frame = 0
for idx, td in enumerate(track_data_list):
kdenlive_bin_id = str(idx + 1)
keep_segments = [s for s in td.timeline if s.action == "keep"]
shifts = compute_jl_cut_shifts(keep_segments, fps, jl_mode=jl_mode, jl_frames=jl_frames)
for k, seg in enumerate(keep_segments):
in_frame = int(round(seg.start * fps))
out_frame_target = int(round(seg.end * fps))
seg_dur_frames = max(1, out_frame_target - in_frame)
a_out_frame = in_frame + seg_dur_frames - 1
# Audio 1 Entry (Voice)
a_entry = ET.SubElement(playlist0, "entry", {
"producer": f"chain0_{idx}",
"in": frames_to_tc(in_frame, fps),
"out": frames_to_tc(a_out_frame, fps),
})
add_prop(a_entry, "kdenlive:id", kdenlive_bin_id)
# Audio 2 Entry (System Audio - cut in lockstep with Voice)
if use_system:
a2_entry = ET.SubElement(playlist4, "entry", {
"producer": f"chain_a2_{idx}",
"in": frames_to_tc(in_frame, fps),
"out": frames_to_tc(a_out_frame, fps),
})
add_prop(a2_entry, "kdenlive:id", kdenlive_bin_id)
shift_prev = shifts[k - 1] if k > 0 else 0
shift_cur = shifts[k]
v_in_frame = in_frame + shift_prev
v_out_frame = a_out_frame + shift_cur
# Video Entry
v_entry = ET.SubElement(playlist2, "entry", {
"producer": f"chain1_{idx}",
"in": frames_to_tc(v_in_frame, fps),
"out": frames_to_tc(v_out_frame, fps),
})
add_prop(v_entry, "kdenlive:id", kdenlive_bin_id)
# Group timeline elements (0: Voice, [1: System], Video last)
children: list[dict] = [
{"data": f"0:{current_tl_audio_frame}", "leaf": "clip", "type": "Leaf"},
]
if use_system:
children.append(
{"data": f"1:{current_tl_audio_frame}", "leaf": "clip", "type": "Leaf"}
)
video_group_idx = 2 if use_system else 1
children.append(
{"data": f"{video_group_idx}:{current_tl_video_frame}", "leaf": "clip", "type": "Leaf"}
)
groups.append({
"children": children,
"type": "Normal",
})
current_tl_audio_frame += seg_dur_frames
current_tl_video_frame += (v_out_frame - v_in_frame + 1)
seq_uuid = str(uuid.uuid4())
seq_uuid_str = f"{{{seq_uuid}}}"
sequence = ET.SubElement(mlt, "tractor", {
"id": seq_uuid_str,
"in": "00:00:00.000",
"out": "00:00:00.000",
})
add_prop(sequence, "kdenlive:uuid", seq_uuid_str)
add_prop(sequence, "kdenlive:clipname", "Sequence 1")
add_prop(sequence, "kdenlive:sequenceproperties.groups", json.dumps(groups, separators=(',', ':'), indent=4))
# Timeline stack order: Black track -> Audio 1 -> Audio 2 -> Video 1
ET.SubElement(sequence, "track", {"producer": "producer0"})
ET.SubElement(sequence, "track", {"producer": "tractor0"})
if use_system:
ET.SubElement(sequence, "track", {"producer": "tractor_a2"})
ET.SubElement(sequence, "track", {"producer": "tractor1"})
main_bin = ET.SubElement(mlt, "playlist", {"id": "main_bin"})
add_prop(main_bin, "kdenlive:docproperties.uuid", seq_uuid_str)
add_prop(main_bin, "kdenlive:docproperties.version", "1.1")
add_prop(main_bin, "xml_retain", "1")
ET.SubElement(main_bin, "entry", {
"producer": seq_uuid_str,
"in": "00:00:00.000",
"out": frames_to_tc(last_frame_index, fps),
})
for idx, td in enumerate(track_data_list):
src_out_frame = int(round(td.duration * fps)) - 1
ET.SubElement(main_bin, "entry", {
"producer": f"chain2_{idx}",
"in": "00:00:00.000",
"out": frames_to_tc(src_out_frame, fps),
})
proj_tractor = ET.SubElement(mlt, "tractor", {
"id": "tractor2",
"in": "00:00:00.000",
"out": frames_to_tc(last_frame_index, fps),
})
add_prop(proj_tractor, "kdenlive:projectTractor", "1")
ET.SubElement(proj_tractor, "track", {
"producer": seq_uuid_str,
"in": "00:00:00.000",
"out": frames_to_tc(last_frame_index, fps),
})
tree = ET.ElementTree(mlt)
if hasattr(ET, "indent"):
ET.indent(tree, space=" ", level=0)
kdenlive_path = output.with_suffix(".kdenlive")
tmp_path = kdenlive_path.with_suffix(".kdenlive.tmp")
tree.write(tmp_path, encoding="utf-8", xml_declaration=True)
os.replace(tmp_path, kdenlive_path)
logger.info(f"[OK] Kdenlive Project saved -> {kdenlive_path.name}")
def map_words_to_edited_timeline(
track_data_list: list[VideoTrackData],
) -> list[tuple[float, float, str]]:
mapped: list[tuple[float, float, str]] = []
accumulated_offset = 0.0
for td in track_data_list:
fps = td.fps
kept_segs = [s for s in td.timeline if s.action == "keep"]
seg_positions: list[tuple[Segment, float]] = []
accumulated_offset_frames = 0
for seg in kept_segs:
seg_tl_start = accumulated_offset_frames / fps
seg_positions.append((seg, seg_tl_start))
dur_f = max(1, int(round(seg.end * fps)) - int(round(seg.start * fps)))
accumulated_offset_frames += dur_f
for w in td.words:
w_text = w.text.strip()
if not w_text:
continue
for seg, seg_tl_start in seg_positions:
overlap_start = max(w.start, seg.start)
overlap_end = min(w.end, seg.end)
if overlap_end > overlap_start:
overlap_dur = overlap_end - overlap_start
word_dur = max(0.01, w.end - w.start)
if overlap_dur / word_dur >= 0.35 or overlap_dur >= 0.08:
ns = seg_tl_start + (overlap_start - seg.start)
ne = seg_tl_start + (overlap_end - seg.start)
if ne > ns:
mapped.append((
accumulated_offset + ns,
accumulated_offset + ne,
w_text,
))
break
accumulated_offset += accumulated_offset_frames / fps
return mapped
LLM_PROMPT_LINES: tuple[str, ...] = (
"SYSTEM PROMPT / INSTRUCTIONS:",
"You are an elite YouTube packaging strategist and retention specialist.",
"Your single goal: Maximize Click-Through Rate (CTR), Average Percentage Viewed (APV/Retention), and Subscriber Conversion.",
"",
"Analyze the provided transcript timeline (which has already been cut for pacing with dead silences and hesitations removed) and generate high-performance video assets.",
"",
"---",
"",
"### DELIVERABLES REQUIRED",
"",
"#### 1. PACKAGING CONCEPTS (A/B/C Test Pairs for Maximum CTR)",
"Provide exactly 3 distinct packaging angles. Each angle MUST be an interconnected Title + Thumbnail Concept pair.",
"- Titles must be under 55 characters, front-load high stakes, avoid disappointment, and open compelling curiosity gaps.",
"- Thumbnail concepts must specify clear visual framing, high-contrast focal points, clean backgrounds, emotional expressions, and an optional 1-3 word punchy text overlay.",
"",
"Angle A (High-Stakes Challenge / Conflict / Curiosity):",
"- Title:",
"- Thumbnail Visual:",
"- Text Overlay:",
"",
"Angle B (Transformation / Extreme Result / Value Delivery):",
"- Title:",
"- Thumbnail Visual:",
"- Text Overlay:",
"",
"Angle C (Contrarian / Unconventional / 'I Was Wrong' Angle):",
"- Title:",
"- Thumbnail Visual:",
"- Text Overlay:",
"",
"#### 2. HOOK-FIRST DESCRIPTION SUMMARY",
"- Write exactly one punchy, high-retention paragraph (3-4 sentences max).",
"- Sentence 1: The core hook stating the primary objective or premise.",
"- Sentence 2: The tension, struggle, or key pivot point encountered.",
"- Sentence 3: The outcome/payoff with primary search keywords naturally embedded.",
"",
"#### 3. RETENTION-DRIVEN VIDEO CHAPTERS",
"- Format: `00:00 - Chapter Title`",
"- The first chapter MUST start at `00:00`.",
"- Space chapters every 90 seconds to 4 minutes based on key milestone transitions.",
"- Use curiosity-driven milestone titles (e.g. 'The fatal mistake', 'Refactoring everything', 'The final run').",
"",
"---",
"[EDITED VIDEO TRANSCRIPT TIMELINE STARTS BELOW]",
"",
)
def generate_combined_transcripts(
track_data_list: list[VideoTrackData],
output_path: Path,
*,
max_line_chars: int = 42,
max_line_words: int = 7,
max_duration_sec: float = 3.2,
min_duration_sec: float = 0.8,
max_gap_sec: float = 0.5,
) -> None:
def format_ts(ts: float, is_srt: bool = True) -> str:
ts = max(0.0, ts)
ms = int(round(ts * 1000)) % 1000
s = int(ts)
h, m, sec = s // 3600, (s % 3600) // 60, s % 60
if is_srt:
return f"{h:02d}:{m:02d}:{sec:02d},{ms:03d}"
return f"{h:02d}:{m:02d}:{sec:02d}"
mapped_words = map_words_to_edited_timeline(track_data_list)
if not mapped_words:
logger.warning("[!] No words remaining for transcript/caption generation.")
return
subs: list[tuple[float, float, str]] = []
chunk: list[str] = []
chunk_start: float | None = None
prev_end: float = 0.0
terminal_punct = {".", "?", "!"}
clause_punct = {",", ";", ":", "—", "-"}
for ns, ne, text in mapped_words:
if chunk_start is None:
chunk_start = ns
time_gap = ns - prev_end
current_dur = ne - chunk_start
prospective_chars = sum(len(w) for w in chunk) + len(chunk) + len(text)
prospective_words = len(chunk) + 1
last_char = chunk[-1][-1] if chunk and chunk[-1] else ""
has_clause_break = last_char in (terminal_punct | clause_punct)
is_too_long = (
prospective_chars > max_line_chars
or prospective_words > max_line_words
or current_dur > max_duration_sec
)
is_paused = bool(chunk) and (time_gap > max_gap_sec)
if chunk and (is_too_long or is_paused or (has_clause_break and current_dur >= min_duration_sec)):
sub_end = min(ns, max(prev_end, chunk_start + min_duration_sec))
subs.append((chunk_start, sub_end, " ".join(chunk)))
chunk = [text]
chunk_start = ns
else:
chunk.append(text)
prev_end = ne
if text and text[-1] in terminal_punct and (ne - chunk_start) >= min_duration_sec:
subs.append((chunk_start, ne, " ".join(chunk)))
chunk = []
chunk_start = None
if chunk and chunk_start is not None:
subs.append((chunk_start, max(prev_end, chunk_start + min_duration_sec), " ".join(chunk)))
sanitized_subs: list[tuple[float, float, str]] = []
for i, (s, e, text) in enumerate(subs):
if sanitized_subs and s < sanitized_subs[-1][1]:
s = sanitized_subs[-1][1] + 0.01
e = max(s + 0.1, e)
sanitized_subs.append((s, e, text))
srt_path = output_path.with_suffix(".srt")
lines: list[str] = []
for i, (s, e, text) in enumerate(sanitized_subs, 1):
lines.extend([str(i), f"{format_ts(s)} --> {format_ts(e)}", text, ""])
_write_text_atomic(srt_path, "\n".join(lines))
logger.info(f"[*] YouTube Captions generated -> {srt_path.name}")
llm_transcript_path = output_path.with_name(f"{output_path.stem}_llm_transcript.txt")
llm_lines: list[str] = list(LLM_PROMPT_LINES)
current_para: list[str] = []
para_start = mapped_words[0][0]
for ns, ne, text in mapped_words:
if not current_para:
para_start = ns
current_para.append(text)
time_elapsed = ne - para_start
is_sentence_end = bool(text and text[-1] in terminal_punct)
if (time_elapsed >= 45.0 and is_sentence_end) or time_elapsed >= 75.0:
llm_lines.append(f"[{format_ts(para_start, False)}] {' '.join(current_para)}")
current_para = []
if current_para:
llm_lines.append(f"[{format_ts(para_start, False)}] {' '.join(current_para)}")
_write_text_atomic(llm_transcript_path, "\n\n".join(llm_lines))
logger.info(f"[*] LLM transcript generated -> {llm_transcript_path.name}")
def _positive_int(value: str) -> int:
try:
ivalue = int(value)
except ValueError:
raise argparse.ArgumentTypeError(f"invalid int value: '{value}'")
if ivalue < 1:
raise argparse.ArgumentTypeError(f"must be >= 1 (got {ivalue})")
return ivalue
def _non_negative_int(value: str) -> int:
try:
ivalue = int(value)
except ValueError:
raise argparse.ArgumentTypeError(f"invalid int value: '{value}'")
if ivalue < 0:
raise argparse.ArgumentTypeError(f"must be >= 0 (got {ivalue})")
return ivalue
def _non_negative_float(value: str) -> float:
try:
fvalue = float(value)
except ValueError:
raise argparse.ArgumentTypeError(f"invalid float value: '{value}'")
if fvalue < 0:
raise argparse.ArgumentTypeError(f"must be >= 0 (got {fvalue})")
return fvalue
def parse_args() -> argparse.Namespace:
p = argparse.ArgumentParser(
description="cadence: High-retention audio-visual auto-editor for raw recordings."
)
p.add_argument(
"--version",
action="version",
version=f"%(prog)s {__version__}",
)
p.add_argument(
"inputs",
nargs="+",
type=Path,
help="One or more video files in sequence order.",
)
p.add_argument(
"--preset",
choices=["relaxed", "balanced", "punchy"],
default=None,
help="Apply built-in editing pacing settings (overrides defaults).",
)
p.add_argument(
"-o", "--output",
type=Path,
default=None,
help="Target output path (defaults to first video path with .kdenlive suffix).",
)
p.add_argument(
"--audio-track",
type=_positive_int,
default=1,
help="1-based OBS audio track index to analyze and map (default: 1).",
)
p.add_argument(
"--jl-mode",
choices=["j", "l", "off"],
default="j",
help="Split cut mode: 'j' (audio leads video), 'l' (video leads audio), 'off' (default: j).",
)
p.add_argument(
"--jl-frames",
type=_non_negative_int,
default=None,
help="Frame lead/lag offset for J/L split cuts to mask jump cuts (default: 2).",
)
p.add_argument(
"--captions",
action="store_true",
help="Generate synced .srt subtitles and LLM packaging transcript file.",
)
p.add_argument(
"--max-silence",
type=_non_negative_float,
default=None,
help="Max allowable silence before cutting (default: 0.50s).",
)
p.add_argument(
"--pad",
type=_non_negative_float,
default=None,
help="Breath padding around cut boundaries (default: 0.10s).",
)
p.add_argument(
"--min-keep",
type=_non_negative_float,
default=None,
help="Minimum duration for a kept cut to prevent audio popping (default: 0.08s).",
)
p.add_argument(
"--model",
default="large-v3",
help="Whisper model size/repo (default: large-v3).",
)
p.add_argument(
"--language",
default="en",
help="Spoken audio language code (default: en).",
)
p.add_argument(
"--extra-fillers",
type=str,
default="",
help="Comma-separated list of additional single filler words to remove.",
)
p.add_argument(
"--no-cache",
action="store_true",
help="Force re-transcription by ignoring existing .whisper.json caches.",
)
p.add_argument(
"-q", "--quiet",
action="store_true",
help="Silence non-essential console status output.",
)
return p.parse_args()
def main() -> None:
args = parse_args()
# Inject predefined settings based on chosen preset without clobbering user overrides
if args.preset:
preset_cfg = PACING_PRESETS[args.preset]
if args.max_silence is None:
args.max_silence = preset_cfg["max_silence"]
if args.pad is None:
args.pad = preset_cfg["pad"]
if args.min_keep is None:
args.min_keep = preset_cfg["min_keep"]
if args.jl_frames is None:
args.jl_frames = preset_cfg["jl_frames"]
if args.max_silence is None:
args.max_silence = 0.50
if args.pad is None:
args.pad = 0.10
if args.min_keep is None:
args.min_keep = 0.08
if args.jl_frames is None:
args.jl_frames = 2
logging.basicConfig(
level=logging.WARNING,
format="%(message)s",
)
logger.setLevel(logging.WARNING if args.quiet else logging.INFO)
if args.preset:
logger.info(f"[*] Applying '{args.preset.upper()}' preset: Max Silence={args.max_silence}s, Pad={args.pad}s")
check_dependencies()
input_videos: list[Path] = [f.resolve() for f in args.inputs]
for v in input_videos:
if not v.exists() or not v.is_file():
sys.exit(f"[!] Input file not found or invalid: {v}")
if args.output:
output = args.output.resolve()
else:
if len(input_videos) == 1:
output = input_videos[0].with_suffix(".kdenlive")
else:
output = input_videos[0].with_name(f"{input_videos[0].stem}_concat.kdenlive")
single_fillers = set(HESITATION_FILLERS)
if args.extra_fillers:
for f in args.extra_fillers.split(","):
norm = normalize_token(f)
if norm:
single_fillers.add(norm)
phrase_fillers = set(DISCOURSE_FILLERS)
backend = detect_backend()
stt_model = _load_faster_whisper_model(args.model) if backend != "mlx" else None
align_cache: dict = {}
track_data_list: list[VideoTrackData] = []
user_audio_track_idx = max(0, args.audio_track - 1)
ref_fps: float | None = None
ref_w = ref_h = 0
for idx, video in enumerate(input_videos, 1):
duration = probe_duration(video)
fps_num, fps_den, width, height = probe_video_format(video)
fps = fps_num / fps_den
if ref_fps is None:
ref_fps, ref_w, ref_h = fps, width, height
elif (fps, width, height) != (ref_fps, ref_w, ref_h):
sys.exit(
f"[!] {video.name} is {fps:g}fps {width}x{height}; all inputs must match "
f"the first ({ref_fps:g}fps {ref_w}x{ref_h})."
)
available_audio_streams = probe_audio_stream_count(video)
if available_audio_streams == 0:
sys.exit(f"[!] No audio streams found in {video.name}.")
if user_audio_track_idx >= available_audio_streams:
logger.warning(
f"[!] Warning: Audio track {args.audio_track} requested, but {video.name} "
f"only has {available_audio_streams} audio stream(s). Falling back to stream 0."
)
resolved_track_idx = 0
else:
resolved_track_idx = user_audio_track_idx
logger.info(
f"\n[*] [{idx}/{len(input_videos)}] {video.name} | "
f"{duration:.2f}s | {fps:.2f} fps | {width}x{height} | Audio stream: a:{resolved_track_idx}"
)
with tempfile.TemporaryDirectory(prefix="cadence_") as td:
wav_16k = Path(td) / "audio_16k.wav"
def do_extract() -> None:
extract_audio(video, wav_16k, track_index=resolved_track_idx)
words = load_or_transcribe(
video=video,
wav_16k=wav_16k,
model=args.model,
language=args.language,
track_idx=resolved_track_idx,
no_cache=args.no_cache,
backend=backend,
stt_model=stt_model,
align_cache=align_cache,
extract=do_extract,
)
if not words:
sys.exit(
f"[!] No speech detected in {video.name}. "
f"Check --language, --audio-track, or the recording."
)
cuts = build_cuts(
words=words,
duration=duration,
single_fillers=single_fillers,
phrase_fillers=phrase_fillers,
max_silence=args.max_silence,
pad=args.pad,
)
timeline = build_timeline(cuts, duration, min_keep_dur=args.min_keep)
kept = sum(s.end - s.start for s in timeline if s.action == "keep")
if duration > 0 and kept < 0.05 * duration:
sys.exit(
f"[!] Only {kept:.2f}s of {duration:.2f}s kept for {video.name}; "
f"refusing to write a near-empty project. "
f"Retranscribe or adjust --max-silence/--min-keep."
)
track_data_list.append(VideoTrackData(
video=video,
duration=duration,
fps_num=fps_num,
fps_den=fps_den,
width=width,
height=height,
audio_track_idx=resolved_track_idx,
audio_stream_count=available_audio_streams,
words=words,
timeline=timeline,
))
output.parent.mkdir(parents=True, exist_ok=True)
_generate_multi_kdenlive_project(
track_data_list,
output=output,
jl_mode=args.jl_mode,
jl_frames=args.jl_frames,
)
if args.captions:
generate_combined_transcripts(track_data_list, output)
logger.info(f"\n[OK] Processing complete. Master project saved to:\n {output.with_suffix('.kdenlive')}")
if __name__ == "__main__":
main()