#!/usr/bin/env -S uv run --python 3.13 --script # /// script # requires-python = ">=3.13, <3.14" # dependencies = [ # "faster-whisper>=1.2.1", # "mlx-whisper>=0.4.3; platform_system == 'Darwin' and platform_machine == 'arm64'", # "nvidia-cublas-cu12; platform_system == 'Linux'", # "nvidia-cudnn-cu12; platform_system == 'Linux'", # "torch>=2.8.0", # "torchaudio>=2.8.0", # "tqdm>=4.70.1", # "whisperx>=3.8.6", # ] # /// """cadence: High-retention auto-editor for raw video recordings. Features: - Supports multi-file input concatenated in chronological sequence. - Selectable audio track extraction for multi-track OBS recordings. - CTC forced alignment (WhisperX/Wav2Vec2) for phoneme-level cut accuracy. - Removes silences, regex-matched elongated sounds, and English filler words. - Eliminates stuttered word repetitions without destroying intentional pauses. - Frame-locked J-Cuts and L-Cuts to mask visible jump cuts without cumulative A/V drift. - Generates a native .kdenlive project file with synchronized audio/video cuts. - Generates synced YouTube .srt captions and an LLM-ready transcript (prompt + timeline) for downstream summaries. - Built-in pacing presets (relaxed, balanced, punchy) to eliminate the need for shell wrappers. """ from __future__ import annotations import argparse import hashlib import json import logging import math import os import platform import re import shutil import subprocess import sys import tempfile import uuid import xml.etree.ElementTree as ET from collections.abc import Callable from dataclasses import asdict, dataclass from pathlib import Path from typing import Literal logger = logging.getLogger("cadence") __version__ = "0.1.0" # Regex to catch elongated sounds like 'aaaa', 'uhhh', 'ummm', 'mmmm', 'eeeh', etc. HESITATION_REGEX = re.compile( r"^(a{2,}|e{2,}|u{2,}|m{2,}|h{2,}|o{2,}|" r"uh+m*|um+|eh+|er+|erm+|ah+|ha+|hm+|mm+|mhm+|huh+)$", re.IGNORECASE, ) # English hesitation sounds (note: single 'a' is excluded as it is a standard article) HESITATION_FILLERS = { "ehm", "ehmm", "ehmmm", "uhm", "uhmm", "mh", "mhm", "mmm", "mmmm", "mmh", "hmm", "eh", "ehh", "ah", "ahh", "uh", "uhh", "eee", "ee", "umm", "um", "er", "erm", "hm", "uh-huh", "huh", } # Less aggressive discourse markers / mental pause words DISCOURSE_FILLERS = { "sort of", "kind of", "you know", "i mean", "to be honest", "like i say", "like i said", } # Built-in pacing profiles to eliminate shell script wrappers PACING_PRESETS = { "relaxed": { "max_silence": 0.85, "pad": 0.15, "min_keep": 0.10, "jl_frames": 0, }, "balanced": { "max_silence": 0.50, "pad": 0.10, "min_keep": 0.08, "jl_frames": 2, }, "punchy": { "max_silence": 0.35, "pad": 0.08, "min_keep": 0.08, "jl_frames": 4, }, } MLX_MODEL_MAP = { "tiny": "mlx-community/whisper-tiny-mlx", "base": "mlx-community/whisper-base-mlx", "small": "mlx-community/whisper-small-mlx", "medium": "mlx-community/whisper-medium-mlx", "large-v3": "mlx-community/whisper-large-v3-mlx", "large-v3-turbo": "mlx-community/whisper-large-v3-turbo", } VERBATIM_PROMPT = ( "Verbatim transcription with all hesitations, stutters, and verbal fillers like um, uh, er." ) @dataclass(frozen=True, slots=True) class Word: text: str start: float end: float @dataclass(frozen=True, slots=True) class Segment: start: float end: float action: Literal["keep", "drop"] @dataclass(slots=True) class VideoTrackData: video: Path duration: float fps_num: int fps_den: int width: int height: int audio_track_idx: int audio_stream_count: int words: list[Word] timeline: list[Segment] @property def fps(self) -> float: return self.fps_num / self.fps_den def check_dependencies() -> None: for tool in ("ffmpeg", "ffprobe"): if shutil.which(tool) is None: sys.exit(f"[!] Error: Required dependency '{tool}' was not found in PATH.") def run_ff(cmd: list[str], what: str, *, stream: bool = False) -> None: if stream and logger.isEnabledFor(logging.INFO): tail: list[str] = [] proc = subprocess.Popen( cmd, stderr=subprocess.PIPE, text=True, errors="replace", bufsize=1 ) assert proc.stderr is not None for line in proc.stderr: sys.stderr.write(line) tail.append(line) if len(tail) > 40: tail = tail[-40:] proc.wait() rc, last = proc.returncode, "".join(tail[-20:]) else: completed = subprocess.run( cmd, stderr=subprocess.PIPE, text=True, errors="replace" ) rc = completed.returncode last = "\n".join(completed.stderr.splitlines()[-20:]) if rc != 0: raise SystemExit(f"[!] {what} failed (exit {rc}):\n{last}") def check_ff_output(cmd: list[str], what: str) -> str: completed = subprocess.run( cmd, capture_output=True, text=True, errors="replace" ) if completed.returncode != 0: tail = "\n".join(completed.stderr.splitlines()[-20:]) raise SystemExit(f"[!] {what} failed (exit {completed.returncode}):\n{tail}") return completed.stdout def probe_duration(path: Path) -> float: out = check_ff_output([ "ffprobe", "-v", "error", "-show_entries", "format=duration", "-of", "default=noprint_wrappers=1:nokey=1", str(path), ], f"ffprobe duration for {path.name}") try: return max(0.0, float(out.strip())) except ValueError: raise SystemExit(f"[!] Unable to parse duration for {path.name}: '{out.strip()}'") def probe_video_format(path: Path) -> tuple[int, int, int, int]: out = check_ff_output([ "ffprobe", "-v", "error", "-select_streams", "v:0", "-show_entries", "stream=r_frame_rate,avg_frame_rate,width,height", "-of", "json", str(path), ], f"ffprobe video format for {path.name}") try: data = json.loads(out) streams = data.get("streams", []) if not streams: raise ValueError(f"No video streams found in {path.name}") stream = streams[0] rate_str = stream.get("r_frame_rate") def _valid_rate(value: object) -> str | None: if not isinstance(value, str): return None value = value.strip() if not value or value in ("N/A", "0/0", "0"): return None return value rate_str = _valid_rate(rate_str) if rate_str is None: rate_str = _valid_rate(stream.get("avg_frame_rate")) if rate_str is None: rate_str = "30/1" num_str, _, den_str = rate_str.partition("/") try: num = int(num_str) if num_str else 30 den = int(den_str) if den_str else 1 except ValueError: num, den = 30, 1 if den == 0 or num == 0: num, den = 30, 1 width = int(stream["width"]) height = int(stream["height"]) return num, den, width, height except (KeyError, IndexError, ValueError) as e: raise SystemExit(f"[!] Could not determine video format for {path.name}: {e}") def probe_audio_stream_count(path: Path) -> int: out = check_ff_output([ "ffprobe", "-v", "error", "-select_streams", "a", "-show_entries", "stream=index", "-of", "json", str(path), ], f"ffprobe audio streams for {path.name}") try: data = json.loads(out) return len(data.get("streams", [])) except json.JSONDecodeError as e: raise SystemExit(f"[!] Could not parse audio streams for {path.name}: {e}") def extract_audio(video: Path, wav_16k: Path, track_index: int = 0) -> None: logger.info(f"[*] Extracting 16kHz audio ({video.name}) from stream a:{track_index}...") run_ff([ "ffmpeg", "-y", "-loglevel", "error", "-i", str(video), "-map", f"0:a:{track_index}", "-ac", "1", "-ar", "16000", "-c:a", "pcm_s16le", str(wav_16k), ], "audio extraction") if not wav_16k.exists() or wav_16k.stat().st_size == 0: raise SystemExit( f"[!] Audio extraction produced an empty file. Verify stream a:{track_index} in {video.name}." ) def detect_backend() -> str: if platform.system() == "Darwin" and platform.machine() == "arm64": try: import mlx_whisper # noqa: F401 return "mlx" except ImportError: pass return "faster-whisper" def preload_nvidia_libs() -> None: if platform.system() != "Linux": return try: import ctypes import glob import nvidia.cublas.lib import nvidia.cudnn.lib for module in (nvidia.cublas.lib, nvidia.cudnn.lib): for d in getattr(module, "__path__", []): for lib in glob.glob(os.path.join(d, "lib*.so*")): try: ctypes.CDLL(lib, mode=ctypes.RTLD_GLOBAL) except OSError: pass except Exception as e: logger.debug(f"Note: Preload of NVIDIA dynamic libraries skipped/failed: {e}") def _align_with_whisperx( raw_segments: list[dict], wav_16k: Path, language: str, align_cache: dict | None = None, ) -> list[Word]: """Applies CTC Forced Alignment using WhisperX / Wav2Vec2.""" try: import torch import whisperx except ImportError: logger.warning("[!] 'whisperx' or 'torch' not installed. Skipping forced alignment.") return [] try: align_device = ( "cuda" if torch.cuda.is_available() else ("mps" if getattr(torch.backends, "mps", None) and torch.backends.mps.is_available() else "cpu") ) logger.info(f"[*] Running CTC forced alignment ({align_device})...") if isinstance(align_cache, dict) and language in align_cache: align_model, align_meta = align_cache[language] else: align_model, align_meta = whisperx.load_align_model( language_code=language, device=align_device ) if isinstance(align_cache, dict): align_cache[language] = (align_model, align_meta) audio_data = whisperx.load_audio(str(wav_16k)) aligned_result = whisperx.align( raw_segments, align_model, align_meta, audio_data, align_device, return_char_alignments=False, ) aligned_words: list[Word] = [] for w in aligned_result.get("word_segments", []): if "start" in w and "end" in w and w.get("word"): text = str(w["word"]).strip() s = float(w["start"]) e = float(w["end"]) if text and e > s: aligned_words.append(Word(text=text, start=s, end=e)) logger.info(f"[*] CTC alignment succeeded: {len(aligned_words)} words aligned.") return aligned_words except Exception as e: logger.warning(f"[*] CTC forced alignment failed ({e}). Falling back to Whisper timestamps.") return [] def _transcribe_mlx( wav: Path, model_size: str, language: str = "en" ) -> tuple[list[Word], list[dict]]: import mlx_whisper repo = MLX_MODEL_MAP.get(model_size, model_size) result = mlx_whisper.transcribe( str(wav), path_or_hf_repo=repo, language=language, word_timestamps=True, condition_on_previous_text=False, initial_prompt=VERBATIM_PROMPT, verbose=False, ) words: list[Word] = [] raw_segments: list[dict] = [] for seg in result.get("segments", []): raw_segments.append({ "start": float(seg["start"]), "end": float(seg["end"]), "text": str(seg.get("text", "")).strip(), }) for w in seg.get("words", []) or []: text = (w.get("word") or w.get("text") or "").strip() if text: words.append(Word(text=text, start=float(w["start"]), end=float(w["end"]))) return words, raw_segments def _load_faster_whisper_model(model_size: str): preload_nvidia_libs() from faster_whisper import WhisperModel target_device = "cuda" if platform.system() == "Linux" else "cpu" target_compute = "float16" if target_device == "cuda" else "int8" try: return WhisperModel(model_size, device=target_device, compute_type=target_compute) except Exception as e: logger.info(f"[*] {target_device.upper()} init failed ({e}), falling back to CPU.") return WhisperModel(model_size, device="cpu", compute_type="int8") def _transcribe_faster_whisper( wav: Path, model_size: str, language: str = "en", model=None, ) -> tuple[list[Word], list[dict]]: from tqdm import tqdm if model is None: model = _load_faster_whisper_model(model_size) segments, meta = model.transcribe( str(wav), language=language, word_timestamps=True, vad_filter=False, beam_size=5, condition_on_previous_text=False, initial_prompt=VERBATIM_PROMPT, ) words: list[Word] = [] raw_segments: list[dict] = [] pbar = tqdm( total=round(meta.duration, 2), unit="s", disable=not logger.isEnabledFor(logging.INFO), desc="Transcribing", ) last = 0.0 for seg in segments: raw_segments.append({ "start": float(seg.start), "end": float(seg.end), "text": seg.text.strip(), }) if seg.words: for w in seg.words: text = w.word.strip() if text: words.append(Word(text=text, start=float(w.start), end=float(w.end))) pbar.update(max(0.0, seg.end - last)) last = seg.end pbar.close() return words, raw_segments def _cache_key(model: str, backend: str, language: str) -> str: return hashlib.sha256( json.dumps([backend, language, model]).encode() ).hexdigest()[:12] def _cache_path(video: Path, track_idx: int, model: str, backend: str, language: str) -> Path: return video.with_name( f"{video.name}.trk{track_idx}.{_cache_key(model, backend, language)}.whisper.json" ) def _read_cache(path: Path) -> list[Word] | None: try: data = json.loads(path.read_text(encoding="utf-8")) words = data.get("words") if isinstance(data, dict) else None if not isinstance(words, list): return None return [Word(**w) for w in words] except (OSError, ValueError, TypeError, KeyError): return None def _write_cache(path: Path, words: list[Word]) -> None: if not words: return payload = {"words": [asdict(w) for w in words]} try: tmp = path.with_suffix(path.suffix + ".tmp") tmp.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8") os.replace(tmp, path) except OSError as e: logger.warning(f"[!] Could not write cache {path.name}: {e}") def _write_text_atomic(path: Path, text: str) -> None: tmp = path.with_suffix(path.suffix + ".tmp") tmp.write_text(text, encoding="utf-8") os.replace(tmp, path) def load_or_transcribe( video: Path, wav_16k: Path, model: str, language: str, track_idx: int, no_cache: bool, backend: str, stt_model=None, align_cache: dict | None = None, extract: Callable[[], None] | None = None, ) -> list[Word]: cache_path = _cache_path(video, track_idx, model, backend, language) if not no_cache: cached = _read_cache(cache_path) if cached is not None: logger.info(f"[*] Reusing cached transcription: {cache_path.name}") return cached if extract is not None: extract() logger.info(f"[*] Transcribing ({backend}) | Model: {model} | File: {video.name}") if backend == "mlx": words, raw_segments = _transcribe_mlx(wav_16k, model, language=language) else: words, raw_segments = _transcribe_faster_whisper( wav_16k, model, language=language, model=stt_model ) if raw_segments: aligned_words = _align_with_whisperx( raw_segments=raw_segments, wav_16k=wav_16k, language=language, align_cache=align_cache, ) if aligned_words: words = aligned_words _write_cache(cache_path, words) return words def normalize_token(s: str) -> str: return s.strip(" .,?!\"'…-—–:;()[]{}*~`").lower() def build_cuts( words: list[Word], duration: float, single_fillers: set[str], phrase_fillers: set[str], max_silence: float, pad: float, ) -> list[tuple[float, float, str]]: if duration <= 0: return [] cuts: list[tuple[float, float, str]] = [] n = len(words) drop_indices: set[int] = set() for phrase in phrase_fillers: tokens = [normalize_token(t) for t in phrase.split() if normalize_token(t)] plen = len(tokens) if plen == 0 or plen > n: continue for i in range(n - plen + 1): window = [normalize_token(words[i + k].text) for k in range(plen)] if window == tokens: for k in range(plen): drop_indices.add(i + k) for i in range(n - 1): w1, w2 = words[i], words[i + 1] t1, t2 = normalize_token(w1.text), normalize_token(w2.text) if t1 and t1 == t2 and (w2.start - w1.end) < 0.8: drop_indices.add(i) for i in range(n - 3): pair1 = (normalize_token(words[i].text), normalize_token(words[i + 1].text)) pair2 = (normalize_token(words[i + 2].text), normalize_token(words[i + 3].text)) if ( pair1[0] and pair1[1] and pair1 == pair2 and (words[i + 2].start - words[i + 1].end) < 0.8 ): drop_indices.add(i) drop_indices.add(i + 1) for i, w in enumerate(words): tok = normalize_token(w.text) if not tok: drop_indices.add(i) continue if tok in single_fillers or HESITATION_REGEX.match(tok): drop_indices.add(i) prev_end = 0.0 for i, w in enumerate(words): silence_gap = w.start - prev_end if silence_gap > max_silence: s = 0.0 if prev_end == 0.0 else prev_end + pad e = max(s, w.start - pad) if e - s > 0.04: cuts.append((s, e, "silence")) if i in drop_indices: left_bound = words[i - 1].end if (i > 0 and (i - 1) not in drop_indices) else 0.0 right_bound = ( words[i + 1].start if (i < n - 1 and (i + 1) not in drop_indices) else duration ) cut_s = max(left_bound, w.start - pad / 2) cut_e = min(right_bound, w.end + pad / 2) if cut_e > cut_s: cuts.append((cut_s, cut_e, "filler")) prev_end = max(prev_end, w.end) if duration - prev_end > max_silence: s = prev_end + pad if duration - s > 0.04: cuts.append((min(s, duration), duration, "silence")) if not cuts: return [] cuts.sort(key=lambda x: x[0]) merged: list[tuple[float, float, str]] = [cuts[0]] for s, e, r in cuts[1:]: ps, pe, pr = merged[-1] if s <= pe + 0.01: merged[-1] = (ps, max(pe, e), pr if pr == r else "mixed") else: merged.append((s, e, r)) return merged def build_timeline( cuts: list[tuple[float, float, str]], duration: float, min_keep_dur: float = 0.08 ) -> list[Segment]: timeline: list[Segment] = [] cursor = 0.0 for s, e, _ in cuts: s = min(max(s, 0.0), duration) e = min(max(e, 0.0), duration) if s > cursor: timeline.append(Segment(start=cursor, end=s, action="keep")) if e > s: timeline.append(Segment(start=s, end=e, action="drop")) cursor = max(cursor, e) if cursor < duration: timeline.append(Segment(start=cursor, end=duration, action="keep")) filtered: list[Segment] = [] for seg in timeline: if seg.action == "keep" and (seg.end - seg.start) < min_keep_dur: filtered.append(Segment(start=seg.start, end=seg.end, action="drop")) else: filtered.append(seg) if not filtered: return [Segment(0.0, duration, "keep")] consolidated: list[Segment] = [filtered[0]] for s in filtered[1:]: last = consolidated[-1] if s.action == last.action: consolidated[-1] = Segment(start=last.start, end=s.end, action=last.action) else: consolidated.append(s) return consolidated def frames_to_tc(frames: int, fps: float) -> str: """Format frame index securely to Kdenlive/MLT strictly compliant HH:MM:SS.mmm format.""" if frames <= 0: return "00:00:00.000" secs = frames / fps h = int(secs / 3600) m = int((secs % 3600) / 60) s = int(secs % 60) ms = int(round((secs - int(secs)) * 1000)) if ms >= 1000: s += 1 ms -= 1000 if s >= 60: m += 1 s -= 60 if m >= 60: h += 1 m -= 60 return f"{h:02d}:{m:02d}:{s:02d}.{ms:03d}" def compute_jl_cut_shifts( keep_segments: list[Segment], fps: float, jl_mode: Literal["j", "l", "off"], jl_frames: int, ) -> list[int]: count = len(keep_segments) shifts = [0] * count if jl_mode == "off" or jl_frames <= 0 or count < 2: return shifts for i in range(count - 1): seg = keep_segments[i] next_seg = keep_segments[i + 1] cur_in = int(round(seg.start * fps)) cur_out = int(round(seg.end * fps)) cur_dur = max(1, cur_out - cur_in) next_in = int(round(next_seg.start * fps)) next_out = int(round(next_seg.end * fps)) next_dur = max(1, next_out - next_in) handle = max(0, next_in - cur_out - 1) if jl_mode == "j": max_allowed = min(jl_frames, handle, cur_dur // 4, next_dur // 4) shifts[i] = max_allowed elif jl_mode == "l": max_allowed = min(jl_frames, handle, cur_dur // 4, next_dur // 4) shifts[i] = -max_allowed shifts[count - 1] = 0 return shifts def _generate_multi_kdenlive_project( track_data_list: list[VideoTrackData], output: Path, jl_mode: Literal["j", "l", "off"] = "j", jl_frames: int = 2, ) -> None: base = track_data_list[0] fps_num, fps_den = base.fps_num, base.fps_den width, height = base.width, base.height fps = base.fps gcd = math.gcd(width, height) dar_num = width // gcd dar_den = height // gcd total_timeline_frames = 0 total_kept_segments = 0 for td in track_data_list: for seg in td.timeline: if seg.action == "keep": start_f = int(round(seg.start * fps)) end_f = int(round(seg.end * fps)) dur_f = max(1, end_f - start_f) total_timeline_frames += dur_f total_kept_segments += 1 last_frame_index = max(0, total_timeline_frames - 1) logger.info( f"[*] Generating Kdenlive Project: {total_kept_segments} cuts | " f"{total_timeline_frames} frames ({total_timeline_frames / fps:.2f}s) across {len(track_data_list)} video(s)..." ) mlt = ET.Element("mlt", { "LC_NUMERIC": "C", "version": "7.22.0", "producer": "main_bin", "root": str(base.video.parent.resolve()), }) ET.SubElement(mlt, "profile", { "description": "automatic", "width": str(width), "height": str(height), "progressive": "1", "sample_aspect_num": "1", "sample_aspect_den": "1", "display_aspect_num": str(dar_num), "display_aspect_den": str(dar_den), "frame_rate_num": str(fps_num), "frame_rate_den": str(fps_den), "colorspace": "709", }) def add_prop(parent: ET.Element, name: str, val: str) -> None: p = ET.SubElement(parent, "property", {"name": name}) p.text = val producer0 = ET.SubElement(mlt, "producer", { "id": "producer0", }) add_prop(producer0, "length", frames_to_tc(total_timeline_frames, fps)) add_prop(producer0, "eof", "continue") add_prop(producer0, "resource", "black") add_prop(producer0, "mlt_service", "color") add_prop(producer0, "kdenlive:playlistid", "black_track") add_prop(producer0, "mlt_image_format", "rgba") add_prop(producer0, "aspect_ratio", "1") use_system = bool(track_data_list) and all( td.audio_stream_count >= 2 for td in track_data_list ) for idx, td in enumerate(track_data_list): src_path = str(td.video.resolve()) kdenlive_bin_id = str(idx + 1) astream_idx = str(td.audio_track_idx) # Track 1 (Voice) Producer Chain chain_a = ET.SubElement(mlt, "chain", {"id": f"chain0_{idx}"}) add_prop(chain_a, "resource", src_path) add_prop(chain_a, "mlt_service", "avformat-novalidate") add_prop(chain_a, "vstream", "-1") add_prop(chain_a, "astream", astream_idx) add_prop(chain_a, "audio_index", astream_idx) add_prop(chain_a, "video_index", "-1") add_prop(chain_a, "set.test_audio", "0") add_prop(chain_a, "set.test_video", "1") add_prop(chain_a, "kdenlive:id", kdenlive_bin_id) # Track 2 (System) Producer Chain if use_system: sys_stream = "1" if td.audio_track_idx == 0 else "0" chain_a2 = ET.SubElement(mlt, "chain", {"id": f"chain_a2_{idx}"}) add_prop(chain_a2, "resource", src_path) add_prop(chain_a2, "mlt_service", "avformat-novalidate") add_prop(chain_a2, "vstream", "-1") add_prop(chain_a2, "astream", sys_stream) add_prop(chain_a2, "audio_index", sys_stream) add_prop(chain_a2, "video_index", "-1") add_prop(chain_a2, "set.test_audio", "0") add_prop(chain_a2, "set.test_video", "1") add_prop(chain_a2, "kdenlive:id", kdenlive_bin_id) # Video Producer Chain chain_v = ET.SubElement(mlt, "chain", {"id": f"chain1_{idx}"}) add_prop(chain_v, "resource", src_path) add_prop(chain_v, "mlt_service", "avformat-novalidate") add_prop(chain_v, "vstream", "0") add_prop(chain_v, "astream", "-1") add_prop(chain_v, "audio_index", "-1") add_prop(chain_v, "video_index", "0") add_prop(chain_v, "set.test_audio", "1") add_prop(chain_v, "set.test_video", "0") add_prop(chain_v, "kdenlive:id", kdenlive_bin_id) # 1. Audio Track 1 (Voice) playlist0 = ET.SubElement(mlt, "playlist", {"id": "playlist0"}) add_prop(playlist0, "kdenlive:audio_track", "1") playlist1 = ET.SubElement(mlt, "playlist", {"id": "playlist1"}) add_prop(playlist1, "kdenlive:audio_track", "1") tractor0 = ET.SubElement(mlt, "tractor", { "id": "tractor0", "in": "00:00:00.000", "out": frames_to_tc(last_frame_index, fps), }) add_prop(tractor0, "kdenlive:audio_track", "1") add_prop(tractor0, "kdenlive:timeline_active", "1") add_prop(tractor0, "kdenlive:track_name", "Voice") ET.SubElement(tractor0, "track", {"hide": "video", "producer": "playlist0"}) ET.SubElement(tractor0, "track", {"hide": "video", "producer": "playlist1"}) # 2. Audio Track 2 (System Audio) if use_system: playlist4 = ET.SubElement(mlt, "playlist", {"id": "playlist4"}) add_prop(playlist4, "kdenlive:audio_track", "1") playlist5 = ET.SubElement(mlt, "playlist", {"id": "playlist5"}) add_prop(playlist5, "kdenlive:audio_track", "1") tractor_a2 = ET.SubElement(mlt, "tractor", { "id": "tractor_a2", "in": "00:00:00.000", "out": frames_to_tc(last_frame_index, fps), }) add_prop(tractor_a2, "kdenlive:audio_track", "1") add_prop(tractor_a2, "kdenlive:timeline_active", "1") add_prop(tractor_a2, "kdenlive:track_name", "System") ET.SubElement(tractor_a2, "track", {"hide": "video", "producer": "playlist4"}) ET.SubElement(tractor_a2, "track", {"hide": "video", "producer": "playlist5"}) # 3. Video Track 1 playlist2 = ET.SubElement(mlt, "playlist", {"id": "playlist2"}) ET.SubElement(mlt, "playlist", {"id": "playlist3"}) tractor1 = ET.SubElement(mlt, "tractor", { "id": "tractor1", "in": "00:00:00.000", "out": frames_to_tc(last_frame_index, fps), }) add_prop(tractor1, "kdenlive:timeline_active", "1") ET.SubElement(tractor1, "track", {"hide": "audio", "producer": "playlist2"}) ET.SubElement(tractor1, "track", {"hide": "audio", "producer": "playlist3"}) for idx, td in enumerate(track_data_list): src_path = str(td.video.resolve()) kdenlive_bin_id = str(idx + 1) chain_bin = ET.SubElement(mlt, "chain", {"id": f"chain2_{idx}"}) add_prop(chain_bin, "resource", src_path) add_prop(chain_bin, "mlt_service", "avformat-novalidate") add_prop(chain_bin, "audio_index", str(td.audio_track_idx)) add_prop(chain_bin, "video_index", "0") add_prop(chain_bin, "vstream", "0") add_prop(chain_bin, "astream", str(td.audio_track_idx)) add_prop(chain_bin, "kdenlive:id", kdenlive_bin_id) groups = [] current_tl_audio_frame = 0 current_tl_video_frame = 0 for idx, td in enumerate(track_data_list): kdenlive_bin_id = str(idx + 1) keep_segments = [s for s in td.timeline if s.action == "keep"] shifts = compute_jl_cut_shifts(keep_segments, fps, jl_mode=jl_mode, jl_frames=jl_frames) for k, seg in enumerate(keep_segments): in_frame = int(round(seg.start * fps)) out_frame_target = int(round(seg.end * fps)) seg_dur_frames = max(1, out_frame_target - in_frame) a_out_frame = in_frame + seg_dur_frames - 1 # Audio 1 Entry (Voice) a_entry = ET.SubElement(playlist0, "entry", { "producer": f"chain0_{idx}", "in": frames_to_tc(in_frame, fps), "out": frames_to_tc(a_out_frame, fps), }) add_prop(a_entry, "kdenlive:id", kdenlive_bin_id) # Audio 2 Entry (System Audio - cut in lockstep with Voice) if use_system: a2_entry = ET.SubElement(playlist4, "entry", { "producer": f"chain_a2_{idx}", "in": frames_to_tc(in_frame, fps), "out": frames_to_tc(a_out_frame, fps), }) add_prop(a2_entry, "kdenlive:id", kdenlive_bin_id) shift_prev = shifts[k - 1] if k > 0 else 0 shift_cur = shifts[k] v_in_frame = in_frame + shift_prev v_out_frame = a_out_frame + shift_cur # Video Entry v_entry = ET.SubElement(playlist2, "entry", { "producer": f"chain1_{idx}", "in": frames_to_tc(v_in_frame, fps), "out": frames_to_tc(v_out_frame, fps), }) add_prop(v_entry, "kdenlive:id", kdenlive_bin_id) # Group timeline elements (0: Voice, [1: System], Video last) children: list[dict] = [ {"data": f"0:{current_tl_audio_frame}", "leaf": "clip", "type": "Leaf"}, ] if use_system: children.append( {"data": f"1:{current_tl_audio_frame}", "leaf": "clip", "type": "Leaf"} ) video_group_idx = 2 if use_system else 1 children.append( {"data": f"{video_group_idx}:{current_tl_video_frame}", "leaf": "clip", "type": "Leaf"} ) groups.append({ "children": children, "type": "Normal", }) current_tl_audio_frame += seg_dur_frames current_tl_video_frame += (v_out_frame - v_in_frame + 1) seq_uuid = str(uuid.uuid4()) seq_uuid_str = f"{{{seq_uuid}}}" sequence = ET.SubElement(mlt, "tractor", { "id": seq_uuid_str, "in": "00:00:00.000", "out": "00:00:00.000", }) add_prop(sequence, "kdenlive:uuid", seq_uuid_str) add_prop(sequence, "kdenlive:clipname", "Sequence 1") add_prop(sequence, "kdenlive:sequenceproperties.groups", json.dumps(groups, separators=(',', ':'), indent=4)) # Timeline stack order: Black track -> Audio 1 -> Audio 2 -> Video 1 ET.SubElement(sequence, "track", {"producer": "producer0"}) ET.SubElement(sequence, "track", {"producer": "tractor0"}) if use_system: ET.SubElement(sequence, "track", {"producer": "tractor_a2"}) ET.SubElement(sequence, "track", {"producer": "tractor1"}) main_bin = ET.SubElement(mlt, "playlist", {"id": "main_bin"}) add_prop(main_bin, "kdenlive:docproperties.uuid", seq_uuid_str) add_prop(main_bin, "kdenlive:docproperties.version", "1.1") add_prop(main_bin, "xml_retain", "1") ET.SubElement(main_bin, "entry", { "producer": seq_uuid_str, "in": "00:00:00.000", "out": frames_to_tc(last_frame_index, fps), }) for idx, td in enumerate(track_data_list): src_out_frame = int(round(td.duration * fps)) - 1 ET.SubElement(main_bin, "entry", { "producer": f"chain2_{idx}", "in": "00:00:00.000", "out": frames_to_tc(src_out_frame, fps), }) proj_tractor = ET.SubElement(mlt, "tractor", { "id": "tractor2", "in": "00:00:00.000", "out": frames_to_tc(last_frame_index, fps), }) add_prop(proj_tractor, "kdenlive:projectTractor", "1") ET.SubElement(proj_tractor, "track", { "producer": seq_uuid_str, "in": "00:00:00.000", "out": frames_to_tc(last_frame_index, fps), }) tree = ET.ElementTree(mlt) if hasattr(ET, "indent"): ET.indent(tree, space=" ", level=0) kdenlive_path = output.with_suffix(".kdenlive") tmp_path = kdenlive_path.with_suffix(".kdenlive.tmp") tree.write(tmp_path, encoding="utf-8", xml_declaration=True) os.replace(tmp_path, kdenlive_path) logger.info(f"[OK] Kdenlive Project saved -> {kdenlive_path.name}") def map_words_to_edited_timeline( track_data_list: list[VideoTrackData], ) -> list[tuple[float, float, str]]: mapped: list[tuple[float, float, str]] = [] accumulated_offset = 0.0 for td in track_data_list: fps = td.fps kept_segs = [s for s in td.timeline if s.action == "keep"] seg_positions: list[tuple[Segment, float]] = [] accumulated_offset_frames = 0 for seg in kept_segs: seg_tl_start = accumulated_offset_frames / fps seg_positions.append((seg, seg_tl_start)) dur_f = max(1, int(round(seg.end * fps)) - int(round(seg.start * fps))) accumulated_offset_frames += dur_f for w in td.words: w_text = w.text.strip() if not w_text: continue for seg, seg_tl_start in seg_positions: overlap_start = max(w.start, seg.start) overlap_end = min(w.end, seg.end) if overlap_end > overlap_start: overlap_dur = overlap_end - overlap_start word_dur = max(0.01, w.end - w.start) if overlap_dur / word_dur >= 0.35 or overlap_dur >= 0.08: ns = seg_tl_start + (overlap_start - seg.start) ne = seg_tl_start + (overlap_end - seg.start) if ne > ns: mapped.append(( accumulated_offset + ns, accumulated_offset + ne, w_text, )) break accumulated_offset += accumulated_offset_frames / fps return mapped LLM_PROMPT_LINES: tuple[str, ...] = ( "SYSTEM PROMPT / INSTRUCTIONS:", "You are an elite YouTube packaging strategist and retention specialist.", "Your single goal: Maximize Click-Through Rate (CTR), Average Percentage Viewed (APV/Retention), and Subscriber Conversion.", "", "Analyze the provided transcript timeline (which has already been cut for pacing with dead silences and hesitations removed) and generate high-performance video assets.", "", "---", "", "### DELIVERABLES REQUIRED", "", "#### 1. PACKAGING CONCEPTS (A/B/C Test Pairs for Maximum CTR)", "Provide exactly 3 distinct packaging angles. Each angle MUST be an interconnected Title + Thumbnail Concept pair.", "- Titles must be under 55 characters, front-load high stakes, avoid disappointment, and open compelling curiosity gaps.", "- Thumbnail concepts must specify clear visual framing, high-contrast focal points, clean backgrounds, emotional expressions, and an optional 1-3 word punchy text overlay.", "", "Angle A (High-Stakes Challenge / Conflict / Curiosity):", "- Title:", "- Thumbnail Visual:", "- Text Overlay:", "", "Angle B (Transformation / Extreme Result / Value Delivery):", "- Title:", "- Thumbnail Visual:", "- Text Overlay:", "", "Angle C (Contrarian / Unconventional / 'I Was Wrong' Angle):", "- Title:", "- Thumbnail Visual:", "- Text Overlay:", "", "#### 2. HOOK-FIRST DESCRIPTION SUMMARY", "- Write exactly one punchy, high-retention paragraph (3-4 sentences max).", "- Sentence 1: The core hook stating the primary objective or premise.", "- Sentence 2: The tension, struggle, or key pivot point encountered.", "- Sentence 3: The outcome/payoff with primary search keywords naturally embedded.", "", "#### 3. RETENTION-DRIVEN VIDEO CHAPTERS", "- Format: `00:00 - Chapter Title`", "- The first chapter MUST start at `00:00`.", "- Space chapters every 90 seconds to 4 minutes based on key milestone transitions.", "- Use curiosity-driven milestone titles (e.g. 'The fatal mistake', 'Refactoring everything', 'The final run').", "", "---", "[EDITED VIDEO TRANSCRIPT TIMELINE STARTS BELOW]", "", ) def generate_combined_transcripts( track_data_list: list[VideoTrackData], output_path: Path, *, max_line_chars: int = 42, max_line_words: int = 7, max_duration_sec: float = 3.2, min_duration_sec: float = 0.8, max_gap_sec: float = 0.5, ) -> None: def format_ts(ts: float, is_srt: bool = True) -> str: ts = max(0.0, ts) ms = int(round(ts * 1000)) % 1000 s = int(ts) h, m, sec = s // 3600, (s % 3600) // 60, s % 60 if is_srt: return f"{h:02d}:{m:02d}:{sec:02d},{ms:03d}" return f"{h:02d}:{m:02d}:{sec:02d}" mapped_words = map_words_to_edited_timeline(track_data_list) if not mapped_words: logger.warning("[!] No words remaining for transcript/caption generation.") return subs: list[tuple[float, float, str]] = [] chunk: list[str] = [] chunk_start: float | None = None prev_end: float = 0.0 terminal_punct = {".", "?", "!"} clause_punct = {",", ";", ":", "—", "-"} for ns, ne, text in mapped_words: if chunk_start is None: chunk_start = ns time_gap = ns - prev_end current_dur = ne - chunk_start prospective_chars = sum(len(w) for w in chunk) + len(chunk) + len(text) prospective_words = len(chunk) + 1 last_char = chunk[-1][-1] if chunk and chunk[-1] else "" has_clause_break = last_char in (terminal_punct | clause_punct) is_too_long = ( prospective_chars > max_line_chars or prospective_words > max_line_words or current_dur > max_duration_sec ) is_paused = bool(chunk) and (time_gap > max_gap_sec) if chunk and (is_too_long or is_paused or (has_clause_break and current_dur >= min_duration_sec)): sub_end = min(ns, max(prev_end, chunk_start + min_duration_sec)) subs.append((chunk_start, sub_end, " ".join(chunk))) chunk = [text] chunk_start = ns else: chunk.append(text) prev_end = ne if text and text[-1] in terminal_punct and (ne - chunk_start) >= min_duration_sec: subs.append((chunk_start, ne, " ".join(chunk))) chunk = [] chunk_start = None if chunk and chunk_start is not None: subs.append((chunk_start, max(prev_end, chunk_start + min_duration_sec), " ".join(chunk))) sanitized_subs: list[tuple[float, float, str]] = [] for i, (s, e, text) in enumerate(subs): if sanitized_subs and s < sanitized_subs[-1][1]: s = sanitized_subs[-1][1] + 0.01 e = max(s + 0.1, e) sanitized_subs.append((s, e, text)) srt_path = output_path.with_suffix(".srt") lines: list[str] = [] for i, (s, e, text) in enumerate(sanitized_subs, 1): lines.extend([str(i), f"{format_ts(s)} --> {format_ts(e)}", text, ""]) _write_text_atomic(srt_path, "\n".join(lines)) logger.info(f"[*] YouTube Captions generated -> {srt_path.name}") llm_transcript_path = output_path.with_name(f"{output_path.stem}_llm_transcript.txt") llm_lines: list[str] = list(LLM_PROMPT_LINES) current_para: list[str] = [] para_start = mapped_words[0][0] for ns, ne, text in mapped_words: if not current_para: para_start = ns current_para.append(text) time_elapsed = ne - para_start is_sentence_end = bool(text and text[-1] in terminal_punct) if (time_elapsed >= 45.0 and is_sentence_end) or time_elapsed >= 75.0: llm_lines.append(f"[{format_ts(para_start, False)}] {' '.join(current_para)}") current_para = [] if current_para: llm_lines.append(f"[{format_ts(para_start, False)}] {' '.join(current_para)}") _write_text_atomic(llm_transcript_path, "\n\n".join(llm_lines)) logger.info(f"[*] LLM transcript generated -> {llm_transcript_path.name}") def _positive_int(value: str) -> int: try: ivalue = int(value) except ValueError: raise argparse.ArgumentTypeError(f"invalid int value: '{value}'") if ivalue < 1: raise argparse.ArgumentTypeError(f"must be >= 1 (got {ivalue})") return ivalue def _non_negative_int(value: str) -> int: try: ivalue = int(value) except ValueError: raise argparse.ArgumentTypeError(f"invalid int value: '{value}'") if ivalue < 0: raise argparse.ArgumentTypeError(f"must be >= 0 (got {ivalue})") return ivalue def _non_negative_float(value: str) -> float: try: fvalue = float(value) except ValueError: raise argparse.ArgumentTypeError(f"invalid float value: '{value}'") if fvalue < 0: raise argparse.ArgumentTypeError(f"must be >= 0 (got {fvalue})") return fvalue def parse_args() -> argparse.Namespace: p = argparse.ArgumentParser( description="cadence: High-retention audio-visual auto-editor for raw recordings." ) p.add_argument( "--version", action="version", version=f"%(prog)s {__version__}", ) p.add_argument( "inputs", nargs="+", type=Path, help="One or more video files in sequence order.", ) p.add_argument( "--preset", choices=["relaxed", "balanced", "punchy"], default=None, help="Apply built-in editing pacing settings (overrides defaults).", ) p.add_argument( "-o", "--output", type=Path, default=None, help="Target output path (defaults to first video path with .kdenlive suffix).", ) p.add_argument( "--audio-track", type=_positive_int, default=1, help="1-based OBS audio track index to analyze and map (default: 1).", ) p.add_argument( "--jl-mode", choices=["j", "l", "off"], default="j", help="Split cut mode: 'j' (audio leads video), 'l' (video leads audio), 'off' (default: j).", ) p.add_argument( "--jl-frames", type=_non_negative_int, default=None, help="Frame lead/lag offset for J/L split cuts to mask jump cuts (default: 2).", ) p.add_argument( "--captions", action="store_true", help="Generate synced .srt subtitles and LLM packaging transcript file.", ) p.add_argument( "--max-silence", type=_non_negative_float, default=None, help="Max allowable silence before cutting (default: 0.50s).", ) p.add_argument( "--pad", type=_non_negative_float, default=None, help="Breath padding around cut boundaries (default: 0.10s).", ) p.add_argument( "--min-keep", type=_non_negative_float, default=None, help="Minimum duration for a kept cut to prevent audio popping (default: 0.08s).", ) p.add_argument( "--model", default="large-v3", help="Whisper model size/repo (default: large-v3).", ) p.add_argument( "--language", default="en", help="Spoken audio language code (default: en).", ) p.add_argument( "--extra-fillers", type=str, default="", help="Comma-separated list of additional single filler words to remove.", ) p.add_argument( "--no-cache", action="store_true", help="Force re-transcription by ignoring existing .whisper.json caches.", ) p.add_argument( "-q", "--quiet", action="store_true", help="Silence non-essential console status output.", ) return p.parse_args() def main() -> None: args = parse_args() # Inject predefined settings based on chosen preset without clobbering user overrides if args.preset: preset_cfg = PACING_PRESETS[args.preset] if args.max_silence is None: args.max_silence = preset_cfg["max_silence"] if args.pad is None: args.pad = preset_cfg["pad"] if args.min_keep is None: args.min_keep = preset_cfg["min_keep"] if args.jl_frames is None: args.jl_frames = preset_cfg["jl_frames"] if args.max_silence is None: args.max_silence = 0.50 if args.pad is None: args.pad = 0.10 if args.min_keep is None: args.min_keep = 0.08 if args.jl_frames is None: args.jl_frames = 2 logging.basicConfig( level=logging.WARNING, format="%(message)s", ) logger.setLevel(logging.WARNING if args.quiet else logging.INFO) if args.preset: logger.info(f"[*] Applying '{args.preset.upper()}' preset: Max Silence={args.max_silence}s, Pad={args.pad}s") check_dependencies() input_videos: list[Path] = [f.resolve() for f in args.inputs] for v in input_videos: if not v.exists() or not v.is_file(): sys.exit(f"[!] Input file not found or invalid: {v}") if args.output: output = args.output.resolve() else: if len(input_videos) == 1: output = input_videos[0].with_suffix(".kdenlive") else: output = input_videos[0].with_name(f"{input_videos[0].stem}_concat.kdenlive") single_fillers = set(HESITATION_FILLERS) if args.extra_fillers: for f in args.extra_fillers.split(","): norm = normalize_token(f) if norm: single_fillers.add(norm) phrase_fillers = set(DISCOURSE_FILLERS) backend = detect_backend() stt_model = _load_faster_whisper_model(args.model) if backend != "mlx" else None align_cache: dict = {} track_data_list: list[VideoTrackData] = [] user_audio_track_idx = max(0, args.audio_track - 1) ref_fps: float | None = None ref_w = ref_h = 0 for idx, video in enumerate(input_videos, 1): duration = probe_duration(video) fps_num, fps_den, width, height = probe_video_format(video) fps = fps_num / fps_den if ref_fps is None: ref_fps, ref_w, ref_h = fps, width, height elif (fps, width, height) != (ref_fps, ref_w, ref_h): sys.exit( f"[!] {video.name} is {fps:g}fps {width}x{height}; all inputs must match " f"the first ({ref_fps:g}fps {ref_w}x{ref_h})." ) available_audio_streams = probe_audio_stream_count(video) if available_audio_streams == 0: sys.exit(f"[!] No audio streams found in {video.name}.") if user_audio_track_idx >= available_audio_streams: logger.warning( f"[!] Warning: Audio track {args.audio_track} requested, but {video.name} " f"only has {available_audio_streams} audio stream(s). Falling back to stream 0." ) resolved_track_idx = 0 else: resolved_track_idx = user_audio_track_idx logger.info( f"\n[*] [{idx}/{len(input_videos)}] {video.name} | " f"{duration:.2f}s | {fps:.2f} fps | {width}x{height} | Audio stream: a:{resolved_track_idx}" ) with tempfile.TemporaryDirectory(prefix="cadence_") as td: wav_16k = Path(td) / "audio_16k.wav" def do_extract() -> None: extract_audio(video, wav_16k, track_index=resolved_track_idx) words = load_or_transcribe( video=video, wav_16k=wav_16k, model=args.model, language=args.language, track_idx=resolved_track_idx, no_cache=args.no_cache, backend=backend, stt_model=stt_model, align_cache=align_cache, extract=do_extract, ) if not words: sys.exit( f"[!] No speech detected in {video.name}. " f"Check --language, --audio-track, or the recording." ) cuts = build_cuts( words=words, duration=duration, single_fillers=single_fillers, phrase_fillers=phrase_fillers, max_silence=args.max_silence, pad=args.pad, ) timeline = build_timeline(cuts, duration, min_keep_dur=args.min_keep) kept = sum(s.end - s.start for s in timeline if s.action == "keep") if duration > 0 and kept < 0.05 * duration: sys.exit( f"[!] Only {kept:.2f}s of {duration:.2f}s kept for {video.name}; " f"refusing to write a near-empty project. " f"Retranscribe or adjust --max-silence/--min-keep." ) track_data_list.append(VideoTrackData( video=video, duration=duration, fps_num=fps_num, fps_den=fps_den, width=width, height=height, audio_track_idx=resolved_track_idx, audio_stream_count=available_audio_streams, words=words, timeline=timeline, )) output.parent.mkdir(parents=True, exist_ok=True) _generate_multi_kdenlive_project( track_data_list, output=output, jl_mode=args.jl_mode, jl_frames=args.jl_frames, ) if args.captions: generate_combined_transcripts(track_data_list, output) logger.info(f"\n[OK] Processing complete. Master project saved to:\n {output.with_suffix('.kdenlive')}") if __name__ == "__main__": main()