diff --git a/.forgejo/workflows/ci.yml b/.forgejo/workflows/ci.yml new file mode 100644 index 0000000..40f85c4 --- /dev/null +++ b/.forgejo/workflows/ci.yml @@ -0,0 +1,93 @@ +name: CI + +# Forgejo Actions workflow. Forgejo resolves workflows from +# .forgejo/workflows/ first (then .gitea/workflows/, then .github/workflows/). +# +# `runs-on: docker` matches the label exposed by the standard self-hosted +# Forgejo runner. Change it if your runner was registered with another label. + +on: + push: + pull_request: + +jobs: + lint-and-test: + runs-on: docker + steps: + - name: Checkout + # Fully qualified so a changed DEFAULT_ACTIONS_URL can't break the run. + uses: https://data.forgejo.org/actions/checkout@v6 + + - name: Install uv + uses: https://data.forgejo.org/astral-sh/setup-uv@v10 + with: + python-version: "3.13" + + - name: Lint (ruff) + run: uvx ruff check . + + - name: Test (pytest) + # The suite only needs pytest: cadence.py imports nothing third-party + # at module level, so no torch/whisper/ffmpeg is required here. + run: uv run --with pytest pytest -q + + version-bump: + # Only meaningful on pull requests, where there is a base branch to diff + # against. On `push` there is nothing to compare, so the job is skipped. + if: github.event_name == 'pull_request' + runs-on: docker + steps: + - name: Checkout + uses: https://data.forgejo.org/actions/checkout@v6 + with: + # Full history so the base branch is reachable for the comparison. + fetch-depth: 0 + + - name: Require a version bump + shell: bash + run: | + set -euo pipefail + + # `github.base_ref` is the GitHub-compatible alias and is understood + # by Forgejo; `forgejo.base_ref` is the canonical spelling. + BASE="${{ github.base_ref }}" + if [ -z "$BASE" ]; then + echo "No base ref (not a pull request); skipping version check." + exit 0 + fi + + git fetch --no-tags origin "$BASE" + + if [ ! -f cadence.py ]; then + echo "::error::cadence.py is missing from the change." + exit 1 + fi + + NEW=$(sed -n 's/^__version__[[:space:]]*=[[:space:]]*"\([^"]*\)".*/\1/p' cadence.py | head -n1) + if [ -z "$NEW" ]; then + echo "::error::No __version__ found in cadence.py" + exit 1 + fi + + # The script may be introduced by this very PR: if it does not exist + # on the base branch there is no previous version to bump from. + if ! git cat-file -e "FETCH_HEAD:cadence.py" 2>/dev/null; then + echo "cadence.py is new to '$BASE' (no base version); version check passes." + exit 0 + fi + + OLD=$(git show "FETCH_HEAD:cadence.py" | sed -n 's/^__version__[[:space:]]*=[[:space:]]*"\([^"]*\)".*/\1/p' | head -n1) + echo "base __version__: ${OLD:-} head __version__: $NEW" + if [ -z "$OLD" ]; then + echo "Could not read a base version; skipping." + exit 0 + fi + + # Require NEW to be strictly greater than OLD (version sort). + GREATER=$(printf '%s\n%s\n' "$OLD" "$NEW" | sort -V | tail -n1) + if [ "$NEW" = "$GREATER" ] && [ "$OLD" != "$NEW" ]; then + echo "Version bumped: $OLD -> $NEW" + else + echo "::error::cadence.py __version__ must be bumped above $OLD (currently $NEW)." + exit 1 + fi diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..996a1aa --- /dev/null +++ b/.gitignore @@ -0,0 +1,18 @@ +# Python +__pycache__/ +*.py[cod] +.pytest_cache/ +.ruff_cache/ +.venv/ +venv/ + +# Files generated by cadence +*.kdenlive +*.srt +*_llm_transcript.txt +*.whisper.json +*.tmp + +# OS / editor cruft +.DS_Store +Thumbs.db diff --git a/README.md b/README.md index 145ab0c..b9cb4eb 100644 --- a/README.md +++ b/README.md @@ -1,2 +1,77 @@ # cadence +High-retention auto-editor for raw video recordings. It transcribes your audio, cuts silences and filler words, and writes a native `.kdenlive` project ready for final polish — plus `.srt` captions and an LLM-friendly transcript. + +## Quick start + +### Linux + +```sh +# uv — installs to ~/.local/bin +curl -LsSf https://astral.sh/uv/install.sh | sh + +# FFmpeg — pick your distro +sudo apt install ffmpeg # Debian / Ubuntu +sudo dnf swap ffmpeg-free ffmpeg --allowerasing # Fedora +sudo pacman -S ffmpeg # Arch + +# Run +uv run cadence.py recording.mp4 --preset balanced --captions +``` + +> - **uv** installs to `~/.local/bin`; make sure that directory is on your `PATH` (the installer normally adds it — otherwise `export PATH="$HOME/.local/bin:$PATH"` in your shell profile). +> - **Fedora**: enable [RPM Fusion](https://rpmfusion.org/Configuration) first. Fedora ships `ffmpeg-free`, which conflicts with the full `ffmpeg` package, so swap rather than install — `sudo dnf swap ffmpeg-free ffmpeg --allowerasing`. +> - **Nobara**: codecs are managed by Nobara and manual changes are blocked. Run `nobara-sync install-codecs` (or the Codec Wizard) instead of `dnf swap`. + +### macOS + +```sh +brew install uv ffmpeg + +uv run cadence.py recording.mp4 --preset balanced --captions +``` + +### Windows (PowerShell) + +```powershell +powershell -ExecutionPolicy ByPass -c "irm https://astral.sh/uv/install.ps1 | iex" # uv +winget install -e --id Gyan.FFmpeg # FFmpeg + +uv run cadence.py recording.mp4 --preset balanced --captions +``` + +## Install as a command (Linux & macOS) + +Make `cadence` available everywhere by symlinking the script into a directory on your `PATH`: + +```sh +chmod +x cadence.py +mkdir -p ~/.local/bin +ln -sf "$PWD/cadence.py" ~/.local/bin/cadence +``` + +Now call it from any directory: + +```sh +cadence recording.mp4 --preset balanced --captions +``` + +`~/.local/bin` is added to `PATH` by the uv installer; if it isn't, add `export PATH="$HOME/.local/bin:$PATH"` to your shell profile (`~/.bashrc`, `~/.zshrc`, etc.). + +## Output + +- `recording.kdenlive` — the edited project. +- `recording.srt` and `recording_llm_transcript.txt` — generated when `--captions` is used. + +Presets: `relaxed`, `balanced`, `punchy`. See all options with `uv run cadence.py --help`. + +## Development + +Run the linter and tests locally: + +```sh +uvx ruff check . +uv run --with pytest pytest -q +``` + +The test suite needs only `pytest` — `cadence.py` imports nothing third-party at module level, so no torch/whisper/ffmpeg is required to run it. CI runs both via Forgejo Actions (`.forgejo/workflows/ci.yml`). diff --git a/cadence.py b/cadence.py new file mode 100755 index 0000000..9952455 --- /dev/null +++ b/cadence.py @@ -0,0 +1,1522 @@ +#!/usr/bin/env -S uv run --python 3.13 --script +# /// script +# requires-python = ">=3.13, <3.14" +# dependencies = [ +# "faster-whisper>=1.2.1", +# "mlx-whisper>=0.4.3; platform_system == 'Darwin' and platform_machine == 'arm64'", +# "nvidia-cublas-cu12; platform_system == 'Linux'", +# "nvidia-cudnn-cu12; platform_system == 'Linux'", +# "torch>=2.8.0", +# "torchaudio>=2.8.0", +# "tqdm>=4.70.1", +# "whisperx>=3.8.6", +# ] +# /// +"""cadence: High-retention auto-editor for raw video recordings. + +Features: +- Supports multi-file input concatenated in chronological sequence. +- Selectable audio track extraction for multi-track OBS recordings. +- CTC forced alignment (WhisperX/Wav2Vec2) for phoneme-level cut accuracy. +- Removes silences, regex-matched elongated sounds, and English filler words. +- Eliminates stuttered word repetitions without destroying intentional pauses. +- Frame-locked J-Cuts and L-Cuts to mask visible jump cuts without cumulative A/V drift. +- Generates a native .kdenlive project file with synchronized audio/video cuts. +- Generates synced YouTube .srt captions and an LLM-ready transcript (prompt + timeline) for downstream summaries. +- Built-in pacing presets (relaxed, balanced, punchy) to eliminate the need for shell wrappers. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import logging +import math +import os +import platform +import re +import shutil +import subprocess +import sys +import tempfile +import uuid +import xml.etree.ElementTree as ET +from collections.abc import Callable +from dataclasses import asdict, dataclass +from pathlib import Path +from typing import Literal + +logger = logging.getLogger("cadence") + +__version__ = "0.1.0" + +# Regex to catch elongated sounds like 'aaaa', 'uhhh', 'ummm', 'mmmm', 'eeeh', etc. +HESITATION_REGEX = re.compile( + r"^(a{2,}|e{2,}|u{2,}|m{2,}|h{2,}|o{2,}|" + r"uh+m*|um+|eh+|er+|erm+|ah+|ha+|hm+|mm+|mhm+|huh+)$", + re.IGNORECASE, +) + +# English hesitation sounds (note: single 'a' is excluded as it is a standard article) +HESITATION_FILLERS = { + "ehm", "ehmm", "ehmmm", "uhm", "uhmm", "mh", "mhm", "mmm", "mmmm", "mmh", + "hmm", "eh", "ehh", "ah", "ahh", "uh", "uhh", "eee", "ee", "umm", "um", + "er", "erm", "hm", "uh-huh", "huh", +} + +# Less aggressive discourse markers / mental pause words +DISCOURSE_FILLERS = { + "sort of", "kind of", "you know", "i mean", "to be honest", "like i say", "like i said", +} + +# Built-in pacing profiles to eliminate shell script wrappers +PACING_PRESETS = { + "relaxed": { + "max_silence": 0.85, + "pad": 0.15, + "min_keep": 0.10, + "jl_frames": 0, + }, + "balanced": { + "max_silence": 0.50, + "pad": 0.10, + "min_keep": 0.08, + "jl_frames": 2, + }, + "punchy": { + "max_silence": 0.35, + "pad": 0.08, + "min_keep": 0.08, + "jl_frames": 4, + }, +} + +MLX_MODEL_MAP = { + "tiny": "mlx-community/whisper-tiny-mlx", + "base": "mlx-community/whisper-base-mlx", + "small": "mlx-community/whisper-small-mlx", + "medium": "mlx-community/whisper-medium-mlx", + "large-v3": "mlx-community/whisper-large-v3-mlx", + "large-v3-turbo": "mlx-community/whisper-large-v3-turbo", +} + +VERBATIM_PROMPT = ( + "Verbatim transcription with all hesitations, stutters, and verbal fillers like um, uh, er." +) + + +@dataclass(frozen=True, slots=True) +class Word: + text: str + start: float + end: float + + +@dataclass(frozen=True, slots=True) +class Segment: + start: float + end: float + action: Literal["keep", "drop"] + + +@dataclass(slots=True) +class VideoTrackData: + video: Path + duration: float + fps_num: int + fps_den: int + width: int + height: int + audio_track_idx: int + audio_stream_count: int + words: list[Word] + timeline: list[Segment] + + @property + def fps(self) -> float: + return self.fps_num / self.fps_den + + +def check_dependencies() -> None: + for tool in ("ffmpeg", "ffprobe"): + if shutil.which(tool) is None: + sys.exit(f"[!] Error: Required dependency '{tool}' was not found in PATH.") + + +def run_ff(cmd: list[str], what: str, *, stream: bool = False) -> None: + if stream and logger.isEnabledFor(logging.INFO): + tail: list[str] = [] + proc = subprocess.Popen( + cmd, stderr=subprocess.PIPE, text=True, errors="replace", bufsize=1 + ) + assert proc.stderr is not None + for line in proc.stderr: + sys.stderr.write(line) + tail.append(line) + if len(tail) > 40: + tail = tail[-40:] + proc.wait() + rc, last = proc.returncode, "".join(tail[-20:]) + else: + completed = subprocess.run( + cmd, stderr=subprocess.PIPE, text=True, errors="replace" + ) + rc = completed.returncode + last = "\n".join(completed.stderr.splitlines()[-20:]) + + if rc != 0: + raise SystemExit(f"[!] {what} failed (exit {rc}):\n{last}") + + +def check_ff_output(cmd: list[str], what: str) -> str: + completed = subprocess.run( + cmd, capture_output=True, text=True, errors="replace" + ) + if completed.returncode != 0: + tail = "\n".join(completed.stderr.splitlines()[-20:]) + raise SystemExit(f"[!] {what} failed (exit {completed.returncode}):\n{tail}") + return completed.stdout + + +def probe_duration(path: Path) -> float: + out = check_ff_output([ + "ffprobe", "-v", "error", "-show_entries", "format=duration", + "-of", "default=noprint_wrappers=1:nokey=1", str(path), + ], f"ffprobe duration for {path.name}") + try: + return max(0.0, float(out.strip())) + except ValueError: + raise SystemExit(f"[!] Unable to parse duration for {path.name}: '{out.strip()}'") + + +def probe_video_format(path: Path) -> tuple[int, int, int, int]: + out = check_ff_output([ + "ffprobe", "-v", "error", "-select_streams", "v:0", + "-show_entries", "stream=r_frame_rate,avg_frame_rate,width,height", + "-of", "json", str(path), + ], f"ffprobe video format for {path.name}") + try: + data = json.loads(out) + streams = data.get("streams", []) + if not streams: + raise ValueError(f"No video streams found in {path.name}") + stream = streams[0] + + rate_str = stream.get("r_frame_rate") + + def _valid_rate(value: object) -> str | None: + if not isinstance(value, str): + return None + value = value.strip() + if not value or value in ("N/A", "0/0", "0"): + return None + return value + + rate_str = _valid_rate(rate_str) + if rate_str is None: + rate_str = _valid_rate(stream.get("avg_frame_rate")) + if rate_str is None: + rate_str = "30/1" + num_str, _, den_str = rate_str.partition("/") + try: + num = int(num_str) if num_str else 30 + den = int(den_str) if den_str else 1 + except ValueError: + num, den = 30, 1 + if den == 0 or num == 0: + num, den = 30, 1 + + width = int(stream["width"]) + height = int(stream["height"]) + return num, den, width, height + except (KeyError, IndexError, ValueError) as e: + raise SystemExit(f"[!] Could not determine video format for {path.name}: {e}") + + +def probe_audio_stream_count(path: Path) -> int: + out = check_ff_output([ + "ffprobe", "-v", "error", "-select_streams", "a", + "-show_entries", "stream=index", + "-of", "json", str(path), + ], f"ffprobe audio streams for {path.name}") + try: + data = json.loads(out) + return len(data.get("streams", [])) + except json.JSONDecodeError as e: + raise SystemExit(f"[!] Could not parse audio streams for {path.name}: {e}") + + +def extract_audio(video: Path, wav_16k: Path, track_index: int = 0) -> None: + logger.info(f"[*] Extracting 16kHz audio ({video.name}) from stream a:{track_index}...") + run_ff([ + "ffmpeg", "-y", "-loglevel", "error", "-i", str(video), + "-map", f"0:a:{track_index}", "-ac", "1", "-ar", "16000", + "-c:a", "pcm_s16le", str(wav_16k), + ], "audio extraction") + + if not wav_16k.exists() or wav_16k.stat().st_size == 0: + raise SystemExit( + f"[!] Audio extraction produced an empty file. Verify stream a:{track_index} in {video.name}." + ) + + +def detect_backend() -> str: + if platform.system() == "Darwin" and platform.machine() == "arm64": + try: + import mlx_whisper # noqa: F401 + return "mlx" + except ImportError: + pass + return "faster-whisper" + + +def preload_nvidia_libs() -> None: + if platform.system() != "Linux": + return + try: + import ctypes + import glob + + import nvidia.cublas.lib + import nvidia.cudnn.lib + + for module in (nvidia.cublas.lib, nvidia.cudnn.lib): + for d in getattr(module, "__path__", []): + for lib in glob.glob(os.path.join(d, "lib*.so*")): + try: + ctypes.CDLL(lib, mode=ctypes.RTLD_GLOBAL) + except OSError: + pass + except Exception as e: + logger.debug(f"Note: Preload of NVIDIA dynamic libraries skipped/failed: {e}") + + +def _align_with_whisperx( + raw_segments: list[dict], + wav_16k: Path, + language: str, + align_cache: dict | None = None, +) -> list[Word]: + """Applies CTC Forced Alignment using WhisperX / Wav2Vec2.""" + try: + import torch + import whisperx + except ImportError: + logger.warning("[!] 'whisperx' or 'torch' not installed. Skipping forced alignment.") + return [] + + try: + align_device = ( + "cuda" + if torch.cuda.is_available() + else ("mps" if getattr(torch.backends, "mps", None) and torch.backends.mps.is_available() else "cpu") + ) + + logger.info(f"[*] Running CTC forced alignment ({align_device})...") + if isinstance(align_cache, dict) and language in align_cache: + align_model, align_meta = align_cache[language] + else: + align_model, align_meta = whisperx.load_align_model( + language_code=language, device=align_device + ) + if isinstance(align_cache, dict): + align_cache[language] = (align_model, align_meta) + audio_data = whisperx.load_audio(str(wav_16k)) + aligned_result = whisperx.align( + raw_segments, + align_model, + align_meta, + audio_data, + align_device, + return_char_alignments=False, + ) + + aligned_words: list[Word] = [] + for w in aligned_result.get("word_segments", []): + if "start" in w and "end" in w and w.get("word"): + text = str(w["word"]).strip() + s = float(w["start"]) + e = float(w["end"]) + if text and e > s: + aligned_words.append(Word(text=text, start=s, end=e)) + + logger.info(f"[*] CTC alignment succeeded: {len(aligned_words)} words aligned.") + return aligned_words + except Exception as e: + logger.warning(f"[*] CTC forced alignment failed ({e}). Falling back to Whisper timestamps.") + return [] + + +def _transcribe_mlx( + wav: Path, model_size: str, language: str = "en" +) -> tuple[list[Word], list[dict]]: + import mlx_whisper + + repo = MLX_MODEL_MAP.get(model_size, model_size) + result = mlx_whisper.transcribe( + str(wav), + path_or_hf_repo=repo, + language=language, + word_timestamps=True, + condition_on_previous_text=False, + initial_prompt=VERBATIM_PROMPT, + verbose=False, + ) + words: list[Word] = [] + raw_segments: list[dict] = [] + + for seg in result.get("segments", []): + raw_segments.append({ + "start": float(seg["start"]), + "end": float(seg["end"]), + "text": str(seg.get("text", "")).strip(), + }) + for w in seg.get("words", []) or []: + text = (w.get("word") or w.get("text") or "").strip() + if text: + words.append(Word(text=text, start=float(w["start"]), end=float(w["end"]))) + return words, raw_segments + + +def _load_faster_whisper_model(model_size: str): + preload_nvidia_libs() + from faster_whisper import WhisperModel + + target_device = "cuda" if platform.system() == "Linux" else "cpu" + target_compute = "float16" if target_device == "cuda" else "int8" + + try: + return WhisperModel(model_size, device=target_device, compute_type=target_compute) + except Exception as e: + logger.info(f"[*] {target_device.upper()} init failed ({e}), falling back to CPU.") + return WhisperModel(model_size, device="cpu", compute_type="int8") + + +def _transcribe_faster_whisper( + wav: Path, + model_size: str, + language: str = "en", + model=None, +) -> tuple[list[Word], list[dict]]: + from tqdm import tqdm + + if model is None: + model = _load_faster_whisper_model(model_size) + + segments, meta = model.transcribe( + str(wav), + language=language, + word_timestamps=True, + vad_filter=False, + beam_size=5, + condition_on_previous_text=False, + initial_prompt=VERBATIM_PROMPT, + ) + + words: list[Word] = [] + raw_segments: list[dict] = [] + pbar = tqdm( + total=round(meta.duration, 2), + unit="s", + disable=not logger.isEnabledFor(logging.INFO), + desc="Transcribing", + ) + last = 0.0 + for seg in segments: + raw_segments.append({ + "start": float(seg.start), + "end": float(seg.end), + "text": seg.text.strip(), + }) + if seg.words: + for w in seg.words: + text = w.word.strip() + if text: + words.append(Word(text=text, start=float(w.start), end=float(w.end))) + pbar.update(max(0.0, seg.end - last)) + last = seg.end + pbar.close() + return words, raw_segments + + +def _cache_key(model: str, backend: str, language: str) -> str: + return hashlib.sha256( + json.dumps([backend, language, model]).encode() + ).hexdigest()[:12] + + +def _cache_path(video: Path, track_idx: int, model: str, backend: str, language: str) -> Path: + return video.with_name( + f"{video.name}.trk{track_idx}.{_cache_key(model, backend, language)}.whisper.json" + ) + + +def _read_cache(path: Path) -> list[Word] | None: + try: + data = json.loads(path.read_text(encoding="utf-8")) + words = data.get("words") if isinstance(data, dict) else None + if not isinstance(words, list): + return None + return [Word(**w) for w in words] + except (OSError, ValueError, TypeError, KeyError): + return None + + +def _write_cache(path: Path, words: list[Word]) -> None: + if not words: + return + payload = {"words": [asdict(w) for w in words]} + try: + tmp = path.with_suffix(path.suffix + ".tmp") + tmp.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8") + os.replace(tmp, path) + except OSError as e: + logger.warning(f"[!] Could not write cache {path.name}: {e}") + + +def _write_text_atomic(path: Path, text: str) -> None: + tmp = path.with_suffix(path.suffix + ".tmp") + tmp.write_text(text, encoding="utf-8") + os.replace(tmp, path) + + +def load_or_transcribe( + video: Path, + wav_16k: Path, + model: str, + language: str, + track_idx: int, + no_cache: bool, + backend: str, + stt_model=None, + align_cache: dict | None = None, + extract: Callable[[], None] | None = None, +) -> list[Word]: + cache_path = _cache_path(video, track_idx, model, backend, language) + + if not no_cache: + cached = _read_cache(cache_path) + if cached is not None: + logger.info(f"[*] Reusing cached transcription: {cache_path.name}") + return cached + + if extract is not None: + extract() + + logger.info(f"[*] Transcribing ({backend}) | Model: {model} | File: {video.name}") + + if backend == "mlx": + words, raw_segments = _transcribe_mlx(wav_16k, model, language=language) + else: + words, raw_segments = _transcribe_faster_whisper( + wav_16k, model, language=language, model=stt_model + ) + + if raw_segments: + aligned_words = _align_with_whisperx( + raw_segments=raw_segments, + wav_16k=wav_16k, + language=language, + align_cache=align_cache, + ) + if aligned_words: + words = aligned_words + + _write_cache(cache_path, words) + return words + + +def normalize_token(s: str) -> str: + return s.strip(" .,?!\"'…-—–:;()[]{}*~`").lower() + + +def build_cuts( + words: list[Word], + duration: float, + single_fillers: set[str], + phrase_fillers: set[str], + max_silence: float, + pad: float, +) -> list[tuple[float, float, str]]: + if duration <= 0: + return [] + + cuts: list[tuple[float, float, str]] = [] + n = len(words) + drop_indices: set[int] = set() + + for phrase in phrase_fillers: + tokens = [normalize_token(t) for t in phrase.split() if normalize_token(t)] + plen = len(tokens) + if plen == 0 or plen > n: + continue + for i in range(n - plen + 1): + window = [normalize_token(words[i + k].text) for k in range(plen)] + if window == tokens: + for k in range(plen): + drop_indices.add(i + k) + + for i in range(n - 1): + w1, w2 = words[i], words[i + 1] + t1, t2 = normalize_token(w1.text), normalize_token(w2.text) + if t1 and t1 == t2 and (w2.start - w1.end) < 0.8: + drop_indices.add(i) + + for i in range(n - 3): + pair1 = (normalize_token(words[i].text), normalize_token(words[i + 1].text)) + pair2 = (normalize_token(words[i + 2].text), normalize_token(words[i + 3].text)) + if ( + pair1[0] and pair1[1] + and pair1 == pair2 + and (words[i + 2].start - words[i + 1].end) < 0.8 + ): + drop_indices.add(i) + drop_indices.add(i + 1) + + for i, w in enumerate(words): + tok = normalize_token(w.text) + if not tok: + drop_indices.add(i) + continue + if tok in single_fillers or HESITATION_REGEX.match(tok): + drop_indices.add(i) + + prev_end = 0.0 + for i, w in enumerate(words): + silence_gap = w.start - prev_end + if silence_gap > max_silence: + s = 0.0 if prev_end == 0.0 else prev_end + pad + e = max(s, w.start - pad) + if e - s > 0.04: + cuts.append((s, e, "silence")) + + if i in drop_indices: + left_bound = words[i - 1].end if (i > 0 and (i - 1) not in drop_indices) else 0.0 + right_bound = ( + words[i + 1].start if (i < n - 1 and (i + 1) not in drop_indices) else duration + ) + cut_s = max(left_bound, w.start - pad / 2) + cut_e = min(right_bound, w.end + pad / 2) + if cut_e > cut_s: + cuts.append((cut_s, cut_e, "filler")) + + prev_end = max(prev_end, w.end) + + if duration - prev_end > max_silence: + s = prev_end + pad + if duration - s > 0.04: + cuts.append((min(s, duration), duration, "silence")) + + if not cuts: + return [] + + cuts.sort(key=lambda x: x[0]) + merged: list[tuple[float, float, str]] = [cuts[0]] + for s, e, r in cuts[1:]: + ps, pe, pr = merged[-1] + if s <= pe + 0.01: + merged[-1] = (ps, max(pe, e), pr if pr == r else "mixed") + else: + merged.append((s, e, r)) + + return merged + + +def build_timeline( + cuts: list[tuple[float, float, str]], duration: float, min_keep_dur: float = 0.08 +) -> list[Segment]: + timeline: list[Segment] = [] + cursor = 0.0 + + for s, e, _ in cuts: + s = min(max(s, 0.0), duration) + e = min(max(e, 0.0), duration) + if s > cursor: + timeline.append(Segment(start=cursor, end=s, action="keep")) + if e > s: + timeline.append(Segment(start=s, end=e, action="drop")) + cursor = max(cursor, e) + + if cursor < duration: + timeline.append(Segment(start=cursor, end=duration, action="keep")) + + filtered: list[Segment] = [] + for seg in timeline: + if seg.action == "keep" and (seg.end - seg.start) < min_keep_dur: + filtered.append(Segment(start=seg.start, end=seg.end, action="drop")) + else: + filtered.append(seg) + + if not filtered: + return [Segment(0.0, duration, "keep")] + + consolidated: list[Segment] = [filtered[0]] + for s in filtered[1:]: + last = consolidated[-1] + if s.action == last.action: + consolidated[-1] = Segment(start=last.start, end=s.end, action=last.action) + else: + consolidated.append(s) + + return consolidated + + +def frames_to_tc(frames: int, fps: float) -> str: + """Format frame index securely to Kdenlive/MLT strictly compliant HH:MM:SS.mmm format.""" + if frames <= 0: + return "00:00:00.000" + secs = frames / fps + h = int(secs / 3600) + m = int((secs % 3600) / 60) + s = int(secs % 60) + ms = int(round((secs - int(secs)) * 1000)) + if ms >= 1000: + s += 1 + ms -= 1000 + if s >= 60: + m += 1 + s -= 60 + if m >= 60: + h += 1 + m -= 60 + return f"{h:02d}:{m:02d}:{s:02d}.{ms:03d}" + + +def compute_jl_cut_shifts( + keep_segments: list[Segment], + fps: float, + jl_mode: Literal["j", "l", "off"], + jl_frames: int, +) -> list[int]: + count = len(keep_segments) + shifts = [0] * count + if jl_mode == "off" or jl_frames <= 0 or count < 2: + return shifts + + for i in range(count - 1): + seg = keep_segments[i] + next_seg = keep_segments[i + 1] + + cur_in = int(round(seg.start * fps)) + cur_out = int(round(seg.end * fps)) + cur_dur = max(1, cur_out - cur_in) + + next_in = int(round(next_seg.start * fps)) + next_out = int(round(next_seg.end * fps)) + next_dur = max(1, next_out - next_in) + + handle = max(0, next_in - cur_out - 1) + + if jl_mode == "j": + max_allowed = min(jl_frames, handle, cur_dur // 4, next_dur // 4) + shifts[i] = max_allowed + elif jl_mode == "l": + max_allowed = min(jl_frames, handle, cur_dur // 4, next_dur // 4) + shifts[i] = -max_allowed + + shifts[count - 1] = 0 + return shifts + + +def _generate_multi_kdenlive_project( + track_data_list: list[VideoTrackData], + output: Path, + jl_mode: Literal["j", "l", "off"] = "j", + jl_frames: int = 2, +) -> None: + base = track_data_list[0] + fps_num, fps_den = base.fps_num, base.fps_den + width, height = base.width, base.height + fps = base.fps + + gcd = math.gcd(width, height) + dar_num = width // gcd + dar_den = height // gcd + + total_timeline_frames = 0 + total_kept_segments = 0 + + for td in track_data_list: + for seg in td.timeline: + if seg.action == "keep": + start_f = int(round(seg.start * fps)) + end_f = int(round(seg.end * fps)) + dur_f = max(1, end_f - start_f) + total_timeline_frames += dur_f + total_kept_segments += 1 + + last_frame_index = max(0, total_timeline_frames - 1) + + logger.info( + f"[*] Generating Kdenlive Project: {total_kept_segments} cuts | " + f"{total_timeline_frames} frames ({total_timeline_frames / fps:.2f}s) across {len(track_data_list)} video(s)..." + ) + + mlt = ET.Element("mlt", { + "LC_NUMERIC": "C", + "version": "7.22.0", + "producer": "main_bin", + "root": str(base.video.parent.resolve()), + }) + + ET.SubElement(mlt, "profile", { + "description": "automatic", + "width": str(width), + "height": str(height), + "progressive": "1", + "sample_aspect_num": "1", + "sample_aspect_den": "1", + "display_aspect_num": str(dar_num), + "display_aspect_den": str(dar_den), + "frame_rate_num": str(fps_num), + "frame_rate_den": str(fps_den), + "colorspace": "709", + }) + + def add_prop(parent: ET.Element, name: str, val: str) -> None: + p = ET.SubElement(parent, "property", {"name": name}) + p.text = val + + producer0 = ET.SubElement(mlt, "producer", { + "id": "producer0", + }) + add_prop(producer0, "length", frames_to_tc(total_timeline_frames, fps)) + add_prop(producer0, "eof", "continue") + add_prop(producer0, "resource", "black") + add_prop(producer0, "mlt_service", "color") + add_prop(producer0, "kdenlive:playlistid", "black_track") + add_prop(producer0, "mlt_image_format", "rgba") + add_prop(producer0, "aspect_ratio", "1") + + use_system = bool(track_data_list) and all( + td.audio_stream_count >= 2 for td in track_data_list + ) + + for idx, td in enumerate(track_data_list): + src_path = str(td.video.resolve()) + kdenlive_bin_id = str(idx + 1) + astream_idx = str(td.audio_track_idx) + + # Track 1 (Voice) Producer Chain + chain_a = ET.SubElement(mlt, "chain", {"id": f"chain0_{idx}"}) + add_prop(chain_a, "resource", src_path) + add_prop(chain_a, "mlt_service", "avformat-novalidate") + add_prop(chain_a, "vstream", "-1") + add_prop(chain_a, "astream", astream_idx) + add_prop(chain_a, "audio_index", astream_idx) + add_prop(chain_a, "video_index", "-1") + add_prop(chain_a, "set.test_audio", "0") + add_prop(chain_a, "set.test_video", "1") + add_prop(chain_a, "kdenlive:id", kdenlive_bin_id) + + # Track 2 (System) Producer Chain + if use_system: + sys_stream = "1" if td.audio_track_idx == 0 else "0" + chain_a2 = ET.SubElement(mlt, "chain", {"id": f"chain_a2_{idx}"}) + add_prop(chain_a2, "resource", src_path) + add_prop(chain_a2, "mlt_service", "avformat-novalidate") + add_prop(chain_a2, "vstream", "-1") + add_prop(chain_a2, "astream", sys_stream) + add_prop(chain_a2, "audio_index", sys_stream) + add_prop(chain_a2, "video_index", "-1") + add_prop(chain_a2, "set.test_audio", "0") + add_prop(chain_a2, "set.test_video", "1") + add_prop(chain_a2, "kdenlive:id", kdenlive_bin_id) + + # Video Producer Chain + chain_v = ET.SubElement(mlt, "chain", {"id": f"chain1_{idx}"}) + add_prop(chain_v, "resource", src_path) + add_prop(chain_v, "mlt_service", "avformat-novalidate") + add_prop(chain_v, "vstream", "0") + add_prop(chain_v, "astream", "-1") + add_prop(chain_v, "audio_index", "-1") + add_prop(chain_v, "video_index", "0") + add_prop(chain_v, "set.test_audio", "1") + add_prop(chain_v, "set.test_video", "0") + add_prop(chain_v, "kdenlive:id", kdenlive_bin_id) + + # 1. Audio Track 1 (Voice) + playlist0 = ET.SubElement(mlt, "playlist", {"id": "playlist0"}) + add_prop(playlist0, "kdenlive:audio_track", "1") + playlist1 = ET.SubElement(mlt, "playlist", {"id": "playlist1"}) + add_prop(playlist1, "kdenlive:audio_track", "1") + + tractor0 = ET.SubElement(mlt, "tractor", { + "id": "tractor0", + "in": "00:00:00.000", + "out": frames_to_tc(last_frame_index, fps), + }) + add_prop(tractor0, "kdenlive:audio_track", "1") + add_prop(tractor0, "kdenlive:timeline_active", "1") + add_prop(tractor0, "kdenlive:track_name", "Voice") + ET.SubElement(tractor0, "track", {"hide": "video", "producer": "playlist0"}) + ET.SubElement(tractor0, "track", {"hide": "video", "producer": "playlist1"}) + + # 2. Audio Track 2 (System Audio) + if use_system: + playlist4 = ET.SubElement(mlt, "playlist", {"id": "playlist4"}) + add_prop(playlist4, "kdenlive:audio_track", "1") + playlist5 = ET.SubElement(mlt, "playlist", {"id": "playlist5"}) + add_prop(playlist5, "kdenlive:audio_track", "1") + + tractor_a2 = ET.SubElement(mlt, "tractor", { + "id": "tractor_a2", + "in": "00:00:00.000", + "out": frames_to_tc(last_frame_index, fps), + }) + add_prop(tractor_a2, "kdenlive:audio_track", "1") + add_prop(tractor_a2, "kdenlive:timeline_active", "1") + add_prop(tractor_a2, "kdenlive:track_name", "System") + ET.SubElement(tractor_a2, "track", {"hide": "video", "producer": "playlist4"}) + ET.SubElement(tractor_a2, "track", {"hide": "video", "producer": "playlist5"}) + + # 3. Video Track 1 + playlist2 = ET.SubElement(mlt, "playlist", {"id": "playlist2"}) + ET.SubElement(mlt, "playlist", {"id": "playlist3"}) + + tractor1 = ET.SubElement(mlt, "tractor", { + "id": "tractor1", + "in": "00:00:00.000", + "out": frames_to_tc(last_frame_index, fps), + }) + add_prop(tractor1, "kdenlive:timeline_active", "1") + ET.SubElement(tractor1, "track", {"hide": "audio", "producer": "playlist2"}) + ET.SubElement(tractor1, "track", {"hide": "audio", "producer": "playlist3"}) + + for idx, td in enumerate(track_data_list): + src_path = str(td.video.resolve()) + kdenlive_bin_id = str(idx + 1) + chain_bin = ET.SubElement(mlt, "chain", {"id": f"chain2_{idx}"}) + add_prop(chain_bin, "resource", src_path) + add_prop(chain_bin, "mlt_service", "avformat-novalidate") + add_prop(chain_bin, "audio_index", str(td.audio_track_idx)) + add_prop(chain_bin, "video_index", "0") + add_prop(chain_bin, "vstream", "0") + add_prop(chain_bin, "astream", str(td.audio_track_idx)) + add_prop(chain_bin, "kdenlive:id", kdenlive_bin_id) + + groups = [] + current_tl_audio_frame = 0 + current_tl_video_frame = 0 + + for idx, td in enumerate(track_data_list): + kdenlive_bin_id = str(idx + 1) + keep_segments = [s for s in td.timeline if s.action == "keep"] + shifts = compute_jl_cut_shifts(keep_segments, fps, jl_mode=jl_mode, jl_frames=jl_frames) + + for k, seg in enumerate(keep_segments): + in_frame = int(round(seg.start * fps)) + out_frame_target = int(round(seg.end * fps)) + seg_dur_frames = max(1, out_frame_target - in_frame) + a_out_frame = in_frame + seg_dur_frames - 1 + + # Audio 1 Entry (Voice) + a_entry = ET.SubElement(playlist0, "entry", { + "producer": f"chain0_{idx}", + "in": frames_to_tc(in_frame, fps), + "out": frames_to_tc(a_out_frame, fps), + }) + add_prop(a_entry, "kdenlive:id", kdenlive_bin_id) + + # Audio 2 Entry (System Audio - cut in lockstep with Voice) + if use_system: + a2_entry = ET.SubElement(playlist4, "entry", { + "producer": f"chain_a2_{idx}", + "in": frames_to_tc(in_frame, fps), + "out": frames_to_tc(a_out_frame, fps), + }) + add_prop(a2_entry, "kdenlive:id", kdenlive_bin_id) + + shift_prev = shifts[k - 1] if k > 0 else 0 + shift_cur = shifts[k] + + v_in_frame = in_frame + shift_prev + v_out_frame = a_out_frame + shift_cur + + # Video Entry + v_entry = ET.SubElement(playlist2, "entry", { + "producer": f"chain1_{idx}", + "in": frames_to_tc(v_in_frame, fps), + "out": frames_to_tc(v_out_frame, fps), + }) + add_prop(v_entry, "kdenlive:id", kdenlive_bin_id) + + # Group timeline elements (0: Voice, [1: System], Video last) + children: list[dict] = [ + {"data": f"0:{current_tl_audio_frame}", "leaf": "clip", "type": "Leaf"}, + ] + if use_system: + children.append( + {"data": f"1:{current_tl_audio_frame}", "leaf": "clip", "type": "Leaf"} + ) + video_group_idx = 2 if use_system else 1 + children.append( + {"data": f"{video_group_idx}:{current_tl_video_frame}", "leaf": "clip", "type": "Leaf"} + ) + groups.append({ + "children": children, + "type": "Normal", + }) + + current_tl_audio_frame += seg_dur_frames + current_tl_video_frame += (v_out_frame - v_in_frame + 1) + + seq_uuid = str(uuid.uuid4()) + seq_uuid_str = f"{{{seq_uuid}}}" + + sequence = ET.SubElement(mlt, "tractor", { + "id": seq_uuid_str, + "in": "00:00:00.000", + "out": "00:00:00.000", + }) + add_prop(sequence, "kdenlive:uuid", seq_uuid_str) + add_prop(sequence, "kdenlive:clipname", "Sequence 1") + add_prop(sequence, "kdenlive:sequenceproperties.groups", json.dumps(groups, separators=(',', ':'), indent=4)) + + # Timeline stack order: Black track -> Audio 1 -> Audio 2 -> Video 1 + ET.SubElement(sequence, "track", {"producer": "producer0"}) + ET.SubElement(sequence, "track", {"producer": "tractor0"}) + if use_system: + ET.SubElement(sequence, "track", {"producer": "tractor_a2"}) + ET.SubElement(sequence, "track", {"producer": "tractor1"}) + + main_bin = ET.SubElement(mlt, "playlist", {"id": "main_bin"}) + add_prop(main_bin, "kdenlive:docproperties.uuid", seq_uuid_str) + add_prop(main_bin, "kdenlive:docproperties.version", "1.1") + add_prop(main_bin, "xml_retain", "1") + ET.SubElement(main_bin, "entry", { + "producer": seq_uuid_str, + "in": "00:00:00.000", + "out": frames_to_tc(last_frame_index, fps), + }) + for idx, td in enumerate(track_data_list): + src_out_frame = int(round(td.duration * fps)) - 1 + ET.SubElement(main_bin, "entry", { + "producer": f"chain2_{idx}", + "in": "00:00:00.000", + "out": frames_to_tc(src_out_frame, fps), + }) + + proj_tractor = ET.SubElement(mlt, "tractor", { + "id": "tractor2", + "in": "00:00:00.000", + "out": frames_to_tc(last_frame_index, fps), + }) + add_prop(proj_tractor, "kdenlive:projectTractor", "1") + ET.SubElement(proj_tractor, "track", { + "producer": seq_uuid_str, + "in": "00:00:00.000", + "out": frames_to_tc(last_frame_index, fps), + }) + + tree = ET.ElementTree(mlt) + if hasattr(ET, "indent"): + ET.indent(tree, space=" ", level=0) + + kdenlive_path = output.with_suffix(".kdenlive") + tmp_path = kdenlive_path.with_suffix(".kdenlive.tmp") + tree.write(tmp_path, encoding="utf-8", xml_declaration=True) + os.replace(tmp_path, kdenlive_path) + logger.info(f"[OK] Kdenlive Project saved -> {kdenlive_path.name}") + +def map_words_to_edited_timeline( + track_data_list: list[VideoTrackData], +) -> list[tuple[float, float, str]]: + mapped: list[tuple[float, float, str]] = [] + accumulated_offset = 0.0 + + for td in track_data_list: + fps = td.fps + kept_segs = [s for s in td.timeline if s.action == "keep"] + seg_positions: list[tuple[Segment, float]] = [] + accumulated_offset_frames = 0 + for seg in kept_segs: + seg_tl_start = accumulated_offset_frames / fps + seg_positions.append((seg, seg_tl_start)) + dur_f = max(1, int(round(seg.end * fps)) - int(round(seg.start * fps))) + accumulated_offset_frames += dur_f + + for w in td.words: + w_text = w.text.strip() + if not w_text: + continue + + for seg, seg_tl_start in seg_positions: + overlap_start = max(w.start, seg.start) + overlap_end = min(w.end, seg.end) + + if overlap_end > overlap_start: + overlap_dur = overlap_end - overlap_start + word_dur = max(0.01, w.end - w.start) + if overlap_dur / word_dur >= 0.35 or overlap_dur >= 0.08: + ns = seg_tl_start + (overlap_start - seg.start) + ne = seg_tl_start + (overlap_end - seg.start) + if ne > ns: + mapped.append(( + accumulated_offset + ns, + accumulated_offset + ne, + w_text, + )) + break + + accumulated_offset += accumulated_offset_frames / fps + + return mapped + + +LLM_PROMPT_LINES: tuple[str, ...] = ( + "SYSTEM PROMPT / INSTRUCTIONS:", + "You are an elite YouTube packaging strategist and retention specialist.", + "Your single goal: Maximize Click-Through Rate (CTR), Average Percentage Viewed (APV/Retention), and Subscriber Conversion.", + "", + "Analyze the provided transcript timeline (which has already been cut for pacing with dead silences and hesitations removed) and generate high-performance video assets.", + "", + "---", + "", + "### DELIVERABLES REQUIRED", + "", + "#### 1. PACKAGING CONCEPTS (A/B/C Test Pairs for Maximum CTR)", + "Provide exactly 3 distinct packaging angles. Each angle MUST be an interconnected Title + Thumbnail Concept pair.", + "- Titles must be under 55 characters, front-load high stakes, avoid disappointment, and open compelling curiosity gaps.", + "- Thumbnail concepts must specify clear visual framing, high-contrast focal points, clean backgrounds, emotional expressions, and an optional 1-3 word punchy text overlay.", + "", + "Angle A (High-Stakes Challenge / Conflict / Curiosity):", + "- Title:", + "- Thumbnail Visual:", + "- Text Overlay:", + "", + "Angle B (Transformation / Extreme Result / Value Delivery):", + "- Title:", + "- Thumbnail Visual:", + "- Text Overlay:", + "", + "Angle C (Contrarian / Unconventional / 'I Was Wrong' Angle):", + "- Title:", + "- Thumbnail Visual:", + "- Text Overlay:", + "", + "#### 2. HOOK-FIRST DESCRIPTION SUMMARY", + "- Write exactly one punchy, high-retention paragraph (3-4 sentences max).", + "- Sentence 1: The core hook stating the primary objective or premise.", + "- Sentence 2: The tension, struggle, or key pivot point encountered.", + "- Sentence 3: The outcome/payoff with primary search keywords naturally embedded.", + "", + "#### 3. RETENTION-DRIVEN VIDEO CHAPTERS", + "- Format: `00:00 - Chapter Title`", + "- The first chapter MUST start at `00:00`.", + "- Space chapters every 90 seconds to 4 minutes based on key milestone transitions.", + "- Use curiosity-driven milestone titles (e.g. 'The fatal mistake', 'Refactoring everything', 'The final run').", + "", + "---", + "[EDITED VIDEO TRANSCRIPT TIMELINE STARTS BELOW]", + "", +) + + +def generate_combined_transcripts( + track_data_list: list[VideoTrackData], + output_path: Path, + *, + max_line_chars: int = 42, + max_line_words: int = 7, + max_duration_sec: float = 3.2, + min_duration_sec: float = 0.8, + max_gap_sec: float = 0.5, +) -> None: + def format_ts(ts: float, is_srt: bool = True) -> str: + ts = max(0.0, ts) + ms = int(round(ts * 1000)) % 1000 + s = int(ts) + h, m, sec = s // 3600, (s % 3600) // 60, s % 60 + if is_srt: + return f"{h:02d}:{m:02d}:{sec:02d},{ms:03d}" + return f"{h:02d}:{m:02d}:{sec:02d}" + + mapped_words = map_words_to_edited_timeline(track_data_list) + + if not mapped_words: + logger.warning("[!] No words remaining for transcript/caption generation.") + return + + subs: list[tuple[float, float, str]] = [] + chunk: list[str] = [] + chunk_start: float | None = None + prev_end: float = 0.0 + + terminal_punct = {".", "?", "!"} + clause_punct = {",", ";", ":", "—", "-"} + + for ns, ne, text in mapped_words: + if chunk_start is None: + chunk_start = ns + + time_gap = ns - prev_end + current_dur = ne - chunk_start + prospective_chars = sum(len(w) for w in chunk) + len(chunk) + len(text) + prospective_words = len(chunk) + 1 + + last_char = chunk[-1][-1] if chunk and chunk[-1] else "" + has_clause_break = last_char in (terminal_punct | clause_punct) + is_too_long = ( + prospective_chars > max_line_chars + or prospective_words > max_line_words + or current_dur > max_duration_sec + ) + is_paused = bool(chunk) and (time_gap > max_gap_sec) + + if chunk and (is_too_long or is_paused or (has_clause_break and current_dur >= min_duration_sec)): + sub_end = min(ns, max(prev_end, chunk_start + min_duration_sec)) + subs.append((chunk_start, sub_end, " ".join(chunk))) + chunk = [text] + chunk_start = ns + else: + chunk.append(text) + + prev_end = ne + + if text and text[-1] in terminal_punct and (ne - chunk_start) >= min_duration_sec: + subs.append((chunk_start, ne, " ".join(chunk))) + chunk = [] + chunk_start = None + + if chunk and chunk_start is not None: + subs.append((chunk_start, max(prev_end, chunk_start + min_duration_sec), " ".join(chunk))) + + sanitized_subs: list[tuple[float, float, str]] = [] + for i, (s, e, text) in enumerate(subs): + if sanitized_subs and s < sanitized_subs[-1][1]: + s = sanitized_subs[-1][1] + 0.01 + e = max(s + 0.1, e) + sanitized_subs.append((s, e, text)) + + srt_path = output_path.with_suffix(".srt") + lines: list[str] = [] + for i, (s, e, text) in enumerate(sanitized_subs, 1): + lines.extend([str(i), f"{format_ts(s)} --> {format_ts(e)}", text, ""]) + _write_text_atomic(srt_path, "\n".join(lines)) + logger.info(f"[*] YouTube Captions generated -> {srt_path.name}") + + llm_transcript_path = output_path.with_name(f"{output_path.stem}_llm_transcript.txt") + llm_lines: list[str] = list(LLM_PROMPT_LINES) + + current_para: list[str] = [] + para_start = mapped_words[0][0] + + for ns, ne, text in mapped_words: + if not current_para: + para_start = ns + current_para.append(text) + time_elapsed = ne - para_start + is_sentence_end = bool(text and text[-1] in terminal_punct) + + if (time_elapsed >= 45.0 and is_sentence_end) or time_elapsed >= 75.0: + llm_lines.append(f"[{format_ts(para_start, False)}] {' '.join(current_para)}") + current_para = [] + + if current_para: + llm_lines.append(f"[{format_ts(para_start, False)}] {' '.join(current_para)}") + + _write_text_atomic(llm_transcript_path, "\n\n".join(llm_lines)) + logger.info(f"[*] LLM transcript generated -> {llm_transcript_path.name}") + + +def _positive_int(value: str) -> int: + try: + ivalue = int(value) + except ValueError: + raise argparse.ArgumentTypeError(f"invalid int value: '{value}'") + if ivalue < 1: + raise argparse.ArgumentTypeError(f"must be >= 1 (got {ivalue})") + return ivalue + + +def _non_negative_int(value: str) -> int: + try: + ivalue = int(value) + except ValueError: + raise argparse.ArgumentTypeError(f"invalid int value: '{value}'") + if ivalue < 0: + raise argparse.ArgumentTypeError(f"must be >= 0 (got {ivalue})") + return ivalue + + +def _non_negative_float(value: str) -> float: + try: + fvalue = float(value) + except ValueError: + raise argparse.ArgumentTypeError(f"invalid float value: '{value}'") + if fvalue < 0: + raise argparse.ArgumentTypeError(f"must be >= 0 (got {fvalue})") + return fvalue + + +def parse_args() -> argparse.Namespace: + p = argparse.ArgumentParser( + description="cadence: High-retention audio-visual auto-editor for raw recordings." + ) + p.add_argument( + "--version", + action="version", + version=f"%(prog)s {__version__}", + ) + p.add_argument( + "inputs", + nargs="+", + type=Path, + help="One or more video files in sequence order.", + ) + p.add_argument( + "--preset", + choices=["relaxed", "balanced", "punchy"], + default=None, + help="Apply built-in editing pacing settings (overrides defaults).", + ) + p.add_argument( + "-o", "--output", + type=Path, + default=None, + help="Target output path (defaults to first video path with .kdenlive suffix).", + ) + p.add_argument( + "--audio-track", + type=_positive_int, + default=1, + help="1-based OBS audio track index to analyze and map (default: 1).", + ) + p.add_argument( + "--jl-mode", + choices=["j", "l", "off"], + default="j", + help="Split cut mode: 'j' (audio leads video), 'l' (video leads audio), 'off' (default: j).", + ) + p.add_argument( + "--jl-frames", + type=_non_negative_int, + default=None, + help="Frame lead/lag offset for J/L split cuts to mask jump cuts (default: 2).", + ) + p.add_argument( + "--captions", + action="store_true", + help="Generate synced .srt subtitles and LLM packaging transcript file.", + ) + p.add_argument( + "--max-silence", + type=_non_negative_float, + default=None, + help="Max allowable silence before cutting (default: 0.50s).", + ) + p.add_argument( + "--pad", + type=_non_negative_float, + default=None, + help="Breath padding around cut boundaries (default: 0.10s).", + ) + p.add_argument( + "--min-keep", + type=_non_negative_float, + default=None, + help="Minimum duration for a kept cut to prevent audio popping (default: 0.08s).", + ) + p.add_argument( + "--model", + default="large-v3", + help="Whisper model size/repo (default: large-v3).", + ) + p.add_argument( + "--language", + default="en", + help="Spoken audio language code (default: en).", + ) + p.add_argument( + "--extra-fillers", + type=str, + default="", + help="Comma-separated list of additional single filler words to remove.", + ) + p.add_argument( + "--no-cache", + action="store_true", + help="Force re-transcription by ignoring existing .whisper.json caches.", + ) + p.add_argument( + "-q", "--quiet", + action="store_true", + help="Silence non-essential console status output.", + ) + return p.parse_args() + + +def main() -> None: + args = parse_args() + + # Inject predefined settings based on chosen preset without clobbering user overrides + if args.preset: + preset_cfg = PACING_PRESETS[args.preset] + if args.max_silence is None: + args.max_silence = preset_cfg["max_silence"] + if args.pad is None: + args.pad = preset_cfg["pad"] + if args.min_keep is None: + args.min_keep = preset_cfg["min_keep"] + if args.jl_frames is None: + args.jl_frames = preset_cfg["jl_frames"] + if args.max_silence is None: + args.max_silence = 0.50 + if args.pad is None: + args.pad = 0.10 + if args.min_keep is None: + args.min_keep = 0.08 + if args.jl_frames is None: + args.jl_frames = 2 + + logging.basicConfig( + level=logging.WARNING, + format="%(message)s", + ) + logger.setLevel(logging.WARNING if args.quiet else logging.INFO) + + if args.preset: + logger.info(f"[*] Applying '{args.preset.upper()}' preset: Max Silence={args.max_silence}s, Pad={args.pad}s") + + check_dependencies() + + input_videos: list[Path] = [f.resolve() for f in args.inputs] + for v in input_videos: + if not v.exists() or not v.is_file(): + sys.exit(f"[!] Input file not found or invalid: {v}") + + if args.output: + output = args.output.resolve() + else: + if len(input_videos) == 1: + output = input_videos[0].with_suffix(".kdenlive") + else: + output = input_videos[0].with_name(f"{input_videos[0].stem}_concat.kdenlive") + + single_fillers = set(HESITATION_FILLERS) + if args.extra_fillers: + for f in args.extra_fillers.split(","): + norm = normalize_token(f) + if norm: + single_fillers.add(norm) + + phrase_fillers = set(DISCOURSE_FILLERS) + + backend = detect_backend() + stt_model = _load_faster_whisper_model(args.model) if backend != "mlx" else None + align_cache: dict = {} + + track_data_list: list[VideoTrackData] = [] + user_audio_track_idx = max(0, args.audio_track - 1) + + ref_fps: float | None = None + ref_w = ref_h = 0 + + for idx, video in enumerate(input_videos, 1): + duration = probe_duration(video) + fps_num, fps_den, width, height = probe_video_format(video) + fps = fps_num / fps_den + + if ref_fps is None: + ref_fps, ref_w, ref_h = fps, width, height + elif (fps, width, height) != (ref_fps, ref_w, ref_h): + sys.exit( + f"[!] {video.name} is {fps:g}fps {width}x{height}; all inputs must match " + f"the first ({ref_fps:g}fps {ref_w}x{ref_h})." + ) + + available_audio_streams = probe_audio_stream_count(video) + if available_audio_streams == 0: + sys.exit(f"[!] No audio streams found in {video.name}.") + + if user_audio_track_idx >= available_audio_streams: + logger.warning( + f"[!] Warning: Audio track {args.audio_track} requested, but {video.name} " + f"only has {available_audio_streams} audio stream(s). Falling back to stream 0." + ) + resolved_track_idx = 0 + else: + resolved_track_idx = user_audio_track_idx + + logger.info( + f"\n[*] [{idx}/{len(input_videos)}] {video.name} | " + f"{duration:.2f}s | {fps:.2f} fps | {width}x{height} | Audio stream: a:{resolved_track_idx}" + ) + + with tempfile.TemporaryDirectory(prefix="cadence_") as td: + wav_16k = Path(td) / "audio_16k.wav" + + def do_extract() -> None: + extract_audio(video, wav_16k, track_index=resolved_track_idx) + + words = load_or_transcribe( + video=video, + wav_16k=wav_16k, + model=args.model, + language=args.language, + track_idx=resolved_track_idx, + no_cache=args.no_cache, + backend=backend, + stt_model=stt_model, + align_cache=align_cache, + extract=do_extract, + ) + + if not words: + sys.exit( + f"[!] No speech detected in {video.name}. " + f"Check --language, --audio-track, or the recording." + ) + + cuts = build_cuts( + words=words, + duration=duration, + single_fillers=single_fillers, + phrase_fillers=phrase_fillers, + max_silence=args.max_silence, + pad=args.pad, + ) + + timeline = build_timeline(cuts, duration, min_keep_dur=args.min_keep) + + kept = sum(s.end - s.start for s in timeline if s.action == "keep") + if duration > 0 and kept < 0.05 * duration: + sys.exit( + f"[!] Only {kept:.2f}s of {duration:.2f}s kept for {video.name}; " + f"refusing to write a near-empty project. " + f"Retranscribe or adjust --max-silence/--min-keep." + ) + + track_data_list.append(VideoTrackData( + video=video, + duration=duration, + fps_num=fps_num, + fps_den=fps_den, + width=width, + height=height, + audio_track_idx=resolved_track_idx, + audio_stream_count=available_audio_streams, + words=words, + timeline=timeline, + )) + + output.parent.mkdir(parents=True, exist_ok=True) + + _generate_multi_kdenlive_project( + track_data_list, + output=output, + jl_mode=args.jl_mode, + jl_frames=args.jl_frames, + ) + + if args.captions: + generate_combined_transcripts(track_data_list, output) + + logger.info(f"\n[OK] Processing complete. Master project saved to:\n {output.with_suffix('.kdenlive')}") + + +if __name__ == "__main__": + main() + diff --git a/pytest.ini b/pytest.ini new file mode 100644 index 0000000..5ee6477 --- /dev/null +++ b/pytest.ini @@ -0,0 +1,2 @@ +[pytest] +testpaths = tests diff --git a/ruff.toml b/ruff.toml new file mode 100644 index 0000000..ac68a87 --- /dev/null +++ b/ruff.toml @@ -0,0 +1,8 @@ +line-length = 100 +target-version = "py313" + +[lint] +select = ["E", "F", "I"] +# E501 is intentionally ignored: the file embeds long prose (the LLM prompt +# template) and a few generated-string lines that cannot be wrapped cleanly. +ignore = ["E501"] diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 0000000..15a5176 --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,43 @@ +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +import pytest # noqa: E402 + +import cadence # noqa: E402 + + +@pytest.fixture +def make_track(tmp_path): + """Small factory for building VideoTrackData objects in tests.""" + + def _make( + *, + name: str = "video.mp4", + duration: float = 2.0, + fps_num: int = 30, + fps_den: int = 1, + width: int = 1920, + height: int = 1080, + audio_track_idx: int = 0, + audio_stream_count: int = 1, + words: list | None = None, + timeline: list | None = None, + ) -> cadence.VideoTrackData: + if timeline is None: + timeline = [cadence.Segment(0.0, duration, "keep")] + return cadence.VideoTrackData( + video=tmp_path / name, + duration=duration, + fps_num=fps_num, + fps_den=fps_den, + width=width, + height=height, + audio_track_idx=audio_track_idx, + audio_stream_count=audio_stream_count, + words=list(words) if words else [], + timeline=list(timeline), + ) + + return _make diff --git a/tests/test_core.py b/tests/test_core.py new file mode 100644 index 0000000..72c9dca --- /dev/null +++ b/tests/test_core.py @@ -0,0 +1,290 @@ +import json +import sys + +import pytest + +from cadence import ( + Segment, + Word, + _cache_key, + _read_cache, + _write_cache, + build_timeline, + compute_jl_cut_shifts, + frames_to_tc, + map_words_to_edited_timeline, + normalize_token, + parse_args, +) + +# --------------------------------------------------------------------------- +# normalize_token +# --------------------------------------------------------------------------- + +@pytest.mark.parametrize( + ("raw", "expected"), + [ + ("Uh...", "uh"), + ("—", ""), + ('"Hello,"', "hello"), + (" Word? ", "word"), + ("...", ""), + ("Umm!", "umm"), + ], +) +def test_normalize_token(raw, expected): + assert normalize_token(raw) == expected + + +# --------------------------------------------------------------------------- +# frames_to_tc +# --------------------------------------------------------------------------- + +def test_frames_to_tc_zero_and_negative(): + assert frames_to_tc(0, 30.0) == "00:00:00.000" + assert frames_to_tc(-7, 30.0) == "00:00:00.000" + + +def _parse_tc(tc: str) -> float: + h, m, s = tc.split(":") + return int(h) * 3600 + int(m) * 60 + float(s) + + +@pytest.mark.parametrize("frames", [1, 30, 100, 1000, 30000, 90000]) +def test_frames_to_tc_roundtrip_ntsc(frames): + fps = 30000 / 1001 + tc = frames_to_tc(frames, fps) + secs = _parse_tc(tc) + back = round(secs * fps) + assert abs(back - frames) <= 1 + + +# --------------------------------------------------------------------------- +# build_timeline +# --------------------------------------------------------------------------- + +def _assert_tiling(timeline, duration, min_keep): + assert timeline, "timeline must not be empty" + + # Alternating keep/drop actions. + for prev, nxt in zip(timeline, timeline[1:]): + assert prev.action != nxt.action + + # Contiguous tiling of [0, duration]. + assert timeline[0].start == pytest.approx(0.0) + for prev, nxt in zip(timeline, timeline[1:]): + assert prev.end == pytest.approx(nxt.start) + assert timeline[-1].end == pytest.approx(duration) + + keep = sum(s.end - s.start for s in timeline if s.action == "keep") + drop = sum(s.end - s.start for s in timeline if s.action == "drop") + assert keep == pytest.approx(duration - drop) + + for seg in timeline: + if seg.action == "keep": + assert (seg.end - seg.start) >= min_keep - 1e-9 + + +@pytest.mark.parametrize( + ("cuts", "duration"), + [ + ([], 10.0), + ([(2.0, 3.0, "silence"), (5.0, 6.0, "filler")], 10.0), + ([(0.0, 1.0, "x"), (9.5, 10.0, "y")], 10.0), + ([(0.0, 2.0, "a"), (2.02, 4.0, "b")], 5.0), + ([(1.0, 2.0, "a"), (2.0, 3.0, "b"), (3.0, 4.0, "c")], 4.0), + ], +) +def test_build_timeline_invariants(cuts, duration): + min_keep = 0.08 + timeline = build_timeline(cuts, duration, min_keep_dur=min_keep) + _assert_tiling(timeline, duration, min_keep) + + +def test_build_timeline_no_cuts_single_keep(): + assert build_timeline([], 5.0) == [Segment(0.0, 5.0, "keep")] + + +def test_build_timeline_zero_duration_returns_keep(): + # Verified real behavior: the empty list falls through to a single keep + # segment spanning [0, 0]. + assert build_timeline([], 0.0) == [Segment(0.0, 0.0, "keep")] + + +# --------------------------------------------------------------------------- +# compute_jl_cut_shifts +# --------------------------------------------------------------------------- + +def _keeps(*pairs): + return [Segment(s, e, "keep") for s, e in pairs] + + +def _assert_jl_bounds(segs, shifts, fps, jl_frames): + for i in range(len(segs) - 1): + cur_in = int(round(segs[i].start * fps)) + cur_out = int(round(segs[i].end * fps)) + cur_dur = max(1, cur_out - cur_in) + next_in = int(round(segs[i + 1].start * fps)) + next_out = int(round(segs[i + 1].end * fps)) + next_dur = max(1, next_out - next_in) + handle = max(0, next_in - cur_out - 1) + bound = min(jl_frames, handle, cur_dur // 4, next_dur // 4) + assert abs(shifts[i]) <= bound + assert shifts[-1] == 0 + + +def test_jl_off_and_trivial_cases(): + segs = _keeps((0, 1), (2, 3), (4, 5)) + assert compute_jl_cut_shifts(segs, 30.0, "off", 10) == [0, 0, 0] + assert compute_jl_cut_shifts(segs, 30.0, "j", 0) == [0, 0, 0] + assert compute_jl_cut_shifts(_keeps((0, 1)), 30.0, "j", 10) == [0] + assert compute_jl_cut_shifts([], 30.0, "j", 10) == [] + + +def test_jl_j_signs_and_bounds(): + segs = _keeps((0, 1), (2, 3), (4, 5)) + shifts = compute_jl_cut_shifts(segs, 30.0, "j", 2) + assert shifts == [2, 2, 0] + assert all(x >= 0 for x in shifts) + _assert_jl_bounds(segs, shifts, 30.0, 2) + + +def test_jl_l_signs_and_bounds(): + segs = _keeps((0, 1), (2, 3), (4, 5)) + shifts = compute_jl_cut_shifts(segs, 30.0, "l", 2) + assert shifts == [-2, -2, 0] + assert all(x <= 0 for x in shifts) + _assert_jl_bounds(segs, shifts, 30.0, 2) + + +def test_jl_clamped_by_handle(): + # Touching segments leave no handle, so even a large request becomes 0. + segs = _keeps((0, 1), (1, 2)) + assert compute_jl_cut_shifts(segs, 30.0, "j", 100) == [0, 0] + + +def test_jl_last_element_always_zero(): + segs = _keeps((0, 1), (2, 3), (4, 5), (6, 7)) + for mode in ("j", "l"): + assert compute_jl_cut_shifts(segs, 30.0, mode, 2)[-1] == 0 + + +# --------------------------------------------------------------------------- +# map_words_to_edited_timeline +# --------------------------------------------------------------------------- + +def test_map_words_to_edited_timeline(make_track): + td1 = make_track( + name="a.mp4", + duration=4.0, + timeline=[ + Segment(0.0, 2.0, "keep"), + Segment(2.0, 3.0, "drop"), + Segment(3.0, 4.0, "keep"), + ], + words=[ + Word("a", 1.0, 1.2), + Word("dropped", 2.1, 2.3), + Word("b", 3.1, 3.3), + ], + ) + td2 = make_track( + name="b.mp4", + duration=3.0, + timeline=[Segment(0.0, 3.0, "keep")], + words=[Word("c", 0.5, 0.7)], + ) + + mapped = map_words_to_edited_timeline([td1, td2]) + onsets = {text: (s, e) for s, e, text in mapped} + + assert "dropped" not in onsets + assert onsets["b"][0] == pytest.approx(2.1) + assert onsets["c"][0] == pytest.approx(3.5) + assert len(mapped) == 3 + + +# --------------------------------------------------------------------------- +# cache helpers +# --------------------------------------------------------------------------- + +def test_cache_roundtrip_and_no_version_key(tmp_path): + path = tmp_path / "cache.json" + words = [Word("Hello", 0.0, 0.5), Word("world", 0.5, 1.0)] + + _write_cache(path, words) + + assert _read_cache(path) == words + payload = json.loads(path.read_text(encoding="utf-8")) + assert "version" not in payload + assert payload["words"][0] == {"text": "Hello", "start": 0.0, "end": 0.5} + + +def test_write_cache_empty_words_writes_nothing(tmp_path): + path = tmp_path / "empty.json" + _write_cache(path, []) + assert not path.exists() + + +def test_read_cache_corrupt_returns_none(tmp_path): + path = tmp_path / "bad.json" + path.write_text("{not valid json", encoding="utf-8") + assert _read_cache(path) is None + + +def test_read_cache_empty_words_returns_empty_list(tmp_path): + path = tmp_path / "empty.json" + path.write_text(json.dumps({"words": []}), encoding="utf-8") + assert _read_cache(path) == [] + + +def test_read_cache_bare_list_returns_none(tmp_path): + path = tmp_path / "list.json" + path.write_text(json.dumps([{"text": "x", "start": 0, "end": 1}]), encoding="utf-8") + assert _read_cache(path) is None + + +def test_cache_key_varies_with_model_and_language(): + base = _cache_key("large-v3", "faster-whisper", "en") + assert base != _cache_key("small", "faster-whisper", "en") + assert base != _cache_key("large-v3", "faster-whisper", "fr") + assert base == _cache_key("large-v3", "faster-whisper", "en") + + +# --------------------------------------------------------------------------- +# parse_args +# --------------------------------------------------------------------------- + +def _parse(monkeypatch, argv): + monkeypatch.setattr(sys, "argv", ["cadence.py", *argv]) + return parse_args() + + +@pytest.mark.parametrize( + "argv", + [ + ["video.mp4", "--audio-track", "0"], + ["video.mp4", "--jl-frames", "-1"], + ["video.mp4", "--pad", "-0.1"], + ["video.mp4", "--max-silence", "abc"], + ], +) +def test_parse_args_rejects_invalid(monkeypatch, argv): + with pytest.raises(SystemExit) as exc: + _parse(monkeypatch, argv) + assert exc.value.code != 0 + + +def test_parse_args_defaults_are_none(monkeypatch): + args = _parse(monkeypatch, ["video.mp4"]) + assert args.max_silence is None + assert args.pad is None + assert args.min_keep is None + assert args.jl_frames is None + assert args.preset is None + + +def test_parse_args_version_exits_zero(monkeypatch): + with pytest.raises(SystemExit) as exc: + _parse(monkeypatch, ["--version"]) + assert exc.value.code == 0 diff --git a/tests/test_kdenlive.py b/tests/test_kdenlive.py new file mode 100644 index 0000000..863fdcd --- /dev/null +++ b/tests/test_kdenlive.py @@ -0,0 +1,70 @@ +import json +import xml.etree.ElementTree as ET + +import cadence + + +def _write_and_parse(make_track, tmp_path, **kwargs): + td = make_track(**kwargs) + out = tmp_path / "project" + cadence._generate_multi_kdenlive_project([td], out) + kdenlive = out.with_suffix(".kdenlive") + assert kdenlive.exists() + return ET.parse(kdenlive).getroot() + + +def _sequence_tractor(root): + for tractor in root.findall("tractor"): + if tractor.find("property[@name='kdenlive:uuid']") is not None: + return tractor + raise AssertionError("sequence tractor not found") + + +def _track_producers(tractor): + return [t.get("producer") for t in tractor.findall("track")] + + +def _group_track_indices(tractor): + prop = tractor.find("property[@name='kdenlive:sequenceproperties.groups']") + groups = json.loads(prop.text) + return [child["data"].split(":")[0] for group in groups for child in group["children"]] + + +def test_project_with_system_audio(make_track, tmp_path): + root = _write_and_parse( + make_track, tmp_path, audio_stream_count=2, audio_track_idx=0 + ) + + assert root.find("chain[@id='chain_a2_0']") is not None + + seq = _sequence_tractor(root) + assert _track_producers(seq) == ["producer0", "tractor0", "tractor_a2", "tractor1"] + assert _group_track_indices(seq) == ["0", "1", "2"] + + +def test_project_without_system_audio(make_track, tmp_path): + root = _write_and_parse( + make_track, tmp_path, audio_stream_count=1, audio_track_idx=0 + ) + + assert root.find("chain[@id='chain_a2_0']") is None + + seq = _sequence_tractor(root) + assert _track_producers(seq) == ["producer0", "tractor0", "tractor1"] + assert _group_track_indices(seq) == ["0", "1"] + + +def test_system_stream_wiring_audio_track_0(make_track, tmp_path): + root = _write_and_parse( + make_track, tmp_path, audio_stream_count=2, audio_track_idx=0 + ) + chain = root.find("chain[@id='chain_a2_0']") + assert chain.find("property[@name='astream']").text == "1" + + +def test_system_stream_wiring_audio_track_1(make_track, tmp_path): + root = _write_and_parse( + make_track, tmp_path, audio_stream_count=2, audio_track_idx=1 + ) + chain = root.find("chain[@id='chain_a2_0']") + assert chain.find("property[@name='astream']").text == "0" diff --git a/tests/test_transcripts.py b/tests/test_transcripts.py new file mode 100644 index 0000000..3451d81 --- /dev/null +++ b/tests/test_transcripts.py @@ -0,0 +1,25 @@ +import cadence +from cadence import Segment, Word + + +def test_generate_combined_transcripts(make_track, tmp_path): + td = make_track( + name="video.mp4", + duration=2.0, + timeline=[Segment(0.0, 2.0, "keep")], + words=[Word("Hello.", 0.2, 0.5), Word("world.", 0.6, 0.9)], + ) + out = tmp_path / "captions" + + cadence.generate_combined_transcripts([td], out) + + srt = out.with_suffix(".srt") + assert srt.exists() + assert srt.stat().st_size > 0 + + llm_transcript = out.with_name(f"{out.stem}_llm_transcript.txt") + assert llm_transcript.exists() + assert llm_transcript.stat().st_size > 0 + + # Atomic writes must not leave temp files behind. + assert not list(tmp_path.glob("*.tmp")) diff --git a/tests/test_version.py b/tests/test_version.py new file mode 100644 index 0000000..afc6314 --- /dev/null +++ b/tests/test_version.py @@ -0,0 +1,23 @@ +"""Version metadata checks. + +These keep the CI version-bump gate meaningful: `__version__` must stay a +parseable `MAJOR.MINOR.PATCH` string and be surfaced by `--version`. +""" + +import re +import sys + +import cadence + + +def test_version_is_semver(): + assert re.fullmatch(r"\d+\.\d+\.\d+", cadence.__version__) + + +def test_version_flag_reports_version(monkeypatch, capsys): + monkeypatch.setattr(sys, "argv", ["cadence", "--version"]) + try: + cadence.parse_args() + except SystemExit as exc: + assert exc.code == 0 + assert cadence.__version__ in capsys.readouterr().out