All checks were successful
CI / lint-and-test (push) Successful in 11s
Reviewed-on: #2 Co-authored-by: pedro-bento <mail@pbento.pt> Co-committed-by: pedro-bento <mail@pbento.pt>
1526 lines
51 KiB
Python
Executable file
1526 lines
51 KiB
Python
Executable file
#!/usr/bin/env -S uv run --python 3.13 --script
|
|
# /// script
|
|
# requires-python = ">=3.13, <3.14"
|
|
# dependencies = [
|
|
# "faster-whisper>=1.2.1",
|
|
# # PyAV 19 removed the `metadata_errors` kwarg that faster-whisper's
|
|
# # decode_audio() still passes (unfixed as of faster-whisper 1.2.1).
|
|
# # Pin below 19 until a release includes faster-whisper PR #1495.
|
|
# "av>=11,<19",
|
|
# "mlx-whisper>=0.4.3; platform_system == 'Darwin' and platform_machine == 'arm64'",
|
|
# "nvidia-cublas-cu12; platform_system == 'Linux'",
|
|
# "nvidia-cudnn-cu12; platform_system == 'Linux'",
|
|
# "torch>=2.8.0",
|
|
# "torchaudio>=2.8.0",
|
|
# "tqdm>=4.70.1",
|
|
# "whisperx>=3.8.6",
|
|
# ]
|
|
# ///
|
|
"""cadence: High-retention auto-editor for raw video recordings.
|
|
|
|
Features:
|
|
- Supports multi-file input concatenated in chronological sequence.
|
|
- Selectable audio track extraction for multi-track OBS recordings.
|
|
- CTC forced alignment (WhisperX/Wav2Vec2) for phoneme-level cut accuracy.
|
|
- Removes silences, regex-matched elongated sounds, and English filler words.
|
|
- Eliminates stuttered word repetitions without destroying intentional pauses.
|
|
- Frame-locked J-Cuts and L-Cuts to mask visible jump cuts without cumulative A/V drift.
|
|
- Generates a native .kdenlive project file with synchronized audio/video cuts.
|
|
- Generates synced YouTube .srt captions and an LLM-ready transcript (prompt + timeline) for downstream summaries.
|
|
- Built-in pacing presets (relaxed, balanced, punchy) to eliminate the need for shell wrappers.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import hashlib
|
|
import json
|
|
import logging
|
|
import math
|
|
import os
|
|
import platform
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import sys
|
|
import tempfile
|
|
import uuid
|
|
import xml.etree.ElementTree as ET
|
|
from collections.abc import Callable
|
|
from dataclasses import asdict, dataclass
|
|
from pathlib import Path
|
|
from typing import Literal
|
|
|
|
logger = logging.getLogger("cadence")
|
|
|
|
__version__ = "0.1.1"
|
|
|
|
# Regex to catch elongated sounds like 'aaaa', 'uhhh', 'ummm', 'mmmm', 'eeeh', etc.
|
|
HESITATION_REGEX = re.compile(
|
|
r"^(a{2,}|e{2,}|u{2,}|m{2,}|h{2,}|o{2,}|"
|
|
r"uh+m*|um+|eh+|er+|erm+|ah+|ha+|hm+|mm+|mhm+|huh+)$",
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
# English hesitation sounds (note: single 'a' is excluded as it is a standard article)
|
|
HESITATION_FILLERS = {
|
|
"ehm", "ehmm", "ehmmm", "uhm", "uhmm", "mh", "mhm", "mmm", "mmmm", "mmh",
|
|
"hmm", "eh", "ehh", "ah", "ahh", "uh", "uhh", "eee", "ee", "umm", "um",
|
|
"er", "erm", "hm", "uh-huh", "huh",
|
|
}
|
|
|
|
# Less aggressive discourse markers / mental pause words
|
|
DISCOURSE_FILLERS = {
|
|
"sort of", "kind of", "you know", "i mean", "to be honest", "like i say", "like i said",
|
|
}
|
|
|
|
# Built-in pacing profiles to eliminate shell script wrappers
|
|
PACING_PRESETS = {
|
|
"relaxed": {
|
|
"max_silence": 0.85,
|
|
"pad": 0.15,
|
|
"min_keep": 0.10,
|
|
"jl_frames": 0,
|
|
},
|
|
"balanced": {
|
|
"max_silence": 0.50,
|
|
"pad": 0.10,
|
|
"min_keep": 0.08,
|
|
"jl_frames": 2,
|
|
},
|
|
"punchy": {
|
|
"max_silence": 0.35,
|
|
"pad": 0.08,
|
|
"min_keep": 0.08,
|
|
"jl_frames": 4,
|
|
},
|
|
}
|
|
|
|
MLX_MODEL_MAP = {
|
|
"tiny": "mlx-community/whisper-tiny-mlx",
|
|
"base": "mlx-community/whisper-base-mlx",
|
|
"small": "mlx-community/whisper-small-mlx",
|
|
"medium": "mlx-community/whisper-medium-mlx",
|
|
"large-v3": "mlx-community/whisper-large-v3-mlx",
|
|
"large-v3-turbo": "mlx-community/whisper-large-v3-turbo",
|
|
}
|
|
|
|
VERBATIM_PROMPT = (
|
|
"Verbatim transcription with all hesitations, stutters, and verbal fillers like um, uh, er."
|
|
)
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class Word:
|
|
text: str
|
|
start: float
|
|
end: float
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class Segment:
|
|
start: float
|
|
end: float
|
|
action: Literal["keep", "drop"]
|
|
|
|
|
|
@dataclass(slots=True)
|
|
class VideoTrackData:
|
|
video: Path
|
|
duration: float
|
|
fps_num: int
|
|
fps_den: int
|
|
width: int
|
|
height: int
|
|
audio_track_idx: int
|
|
audio_stream_count: int
|
|
words: list[Word]
|
|
timeline: list[Segment]
|
|
|
|
@property
|
|
def fps(self) -> float:
|
|
return self.fps_num / self.fps_den
|
|
|
|
|
|
def check_dependencies() -> None:
|
|
for tool in ("ffmpeg", "ffprobe"):
|
|
if shutil.which(tool) is None:
|
|
sys.exit(f"[!] Error: Required dependency '{tool}' was not found in PATH.")
|
|
|
|
|
|
def run_ff(cmd: list[str], what: str, *, stream: bool = False) -> None:
|
|
if stream and logger.isEnabledFor(logging.INFO):
|
|
tail: list[str] = []
|
|
proc = subprocess.Popen(
|
|
cmd, stderr=subprocess.PIPE, text=True, errors="replace", bufsize=1
|
|
)
|
|
assert proc.stderr is not None
|
|
for line in proc.stderr:
|
|
sys.stderr.write(line)
|
|
tail.append(line)
|
|
if len(tail) > 40:
|
|
tail = tail[-40:]
|
|
proc.wait()
|
|
rc, last = proc.returncode, "".join(tail[-20:])
|
|
else:
|
|
completed = subprocess.run(
|
|
cmd, stderr=subprocess.PIPE, text=True, errors="replace"
|
|
)
|
|
rc = completed.returncode
|
|
last = "\n".join(completed.stderr.splitlines()[-20:])
|
|
|
|
if rc != 0:
|
|
raise SystemExit(f"[!] {what} failed (exit {rc}):\n{last}")
|
|
|
|
|
|
def check_ff_output(cmd: list[str], what: str) -> str:
|
|
completed = subprocess.run(
|
|
cmd, capture_output=True, text=True, errors="replace"
|
|
)
|
|
if completed.returncode != 0:
|
|
tail = "\n".join(completed.stderr.splitlines()[-20:])
|
|
raise SystemExit(f"[!] {what} failed (exit {completed.returncode}):\n{tail}")
|
|
return completed.stdout
|
|
|
|
|
|
def probe_duration(path: Path) -> float:
|
|
out = check_ff_output([
|
|
"ffprobe", "-v", "error", "-show_entries", "format=duration",
|
|
"-of", "default=noprint_wrappers=1:nokey=1", str(path),
|
|
], f"ffprobe duration for {path.name}")
|
|
try:
|
|
return max(0.0, float(out.strip()))
|
|
except ValueError:
|
|
raise SystemExit(f"[!] Unable to parse duration for {path.name}: '{out.strip()}'")
|
|
|
|
|
|
def probe_video_format(path: Path) -> tuple[int, int, int, int]:
|
|
out = check_ff_output([
|
|
"ffprobe", "-v", "error", "-select_streams", "v:0",
|
|
"-show_entries", "stream=r_frame_rate,avg_frame_rate,width,height",
|
|
"-of", "json", str(path),
|
|
], f"ffprobe video format for {path.name}")
|
|
try:
|
|
data = json.loads(out)
|
|
streams = data.get("streams", [])
|
|
if not streams:
|
|
raise ValueError(f"No video streams found in {path.name}")
|
|
stream = streams[0]
|
|
|
|
rate_str = stream.get("r_frame_rate")
|
|
|
|
def _valid_rate(value: object) -> str | None:
|
|
if not isinstance(value, str):
|
|
return None
|
|
value = value.strip()
|
|
if not value or value in ("N/A", "0/0", "0"):
|
|
return None
|
|
return value
|
|
|
|
rate_str = _valid_rate(rate_str)
|
|
if rate_str is None:
|
|
rate_str = _valid_rate(stream.get("avg_frame_rate"))
|
|
if rate_str is None:
|
|
rate_str = "30/1"
|
|
num_str, _, den_str = rate_str.partition("/")
|
|
try:
|
|
num = int(num_str) if num_str else 30
|
|
den = int(den_str) if den_str else 1
|
|
except ValueError:
|
|
num, den = 30, 1
|
|
if den == 0 or num == 0:
|
|
num, den = 30, 1
|
|
|
|
width = int(stream["width"])
|
|
height = int(stream["height"])
|
|
return num, den, width, height
|
|
except (KeyError, IndexError, ValueError) as e:
|
|
raise SystemExit(f"[!] Could not determine video format for {path.name}: {e}")
|
|
|
|
|
|
def probe_audio_stream_count(path: Path) -> int:
|
|
out = check_ff_output([
|
|
"ffprobe", "-v", "error", "-select_streams", "a",
|
|
"-show_entries", "stream=index",
|
|
"-of", "json", str(path),
|
|
], f"ffprobe audio streams for {path.name}")
|
|
try:
|
|
data = json.loads(out)
|
|
return len(data.get("streams", []))
|
|
except json.JSONDecodeError as e:
|
|
raise SystemExit(f"[!] Could not parse audio streams for {path.name}: {e}")
|
|
|
|
|
|
def extract_audio(video: Path, wav_16k: Path, track_index: int = 0) -> None:
|
|
logger.info(f"[*] Extracting 16kHz audio ({video.name}) from stream a:{track_index}...")
|
|
run_ff([
|
|
"ffmpeg", "-y", "-loglevel", "error", "-i", str(video),
|
|
"-map", f"0:a:{track_index}", "-ac", "1", "-ar", "16000",
|
|
"-c:a", "pcm_s16le", str(wav_16k),
|
|
], "audio extraction")
|
|
|
|
if not wav_16k.exists() or wav_16k.stat().st_size == 0:
|
|
raise SystemExit(
|
|
f"[!] Audio extraction produced an empty file. Verify stream a:{track_index} in {video.name}."
|
|
)
|
|
|
|
|
|
def detect_backend() -> str:
|
|
if platform.system() == "Darwin" and platform.machine() == "arm64":
|
|
try:
|
|
import mlx_whisper # noqa: F401
|
|
return "mlx"
|
|
except ImportError:
|
|
pass
|
|
return "faster-whisper"
|
|
|
|
|
|
def preload_nvidia_libs() -> None:
|
|
if platform.system() != "Linux":
|
|
return
|
|
try:
|
|
import ctypes
|
|
import glob
|
|
|
|
import nvidia.cublas.lib
|
|
import nvidia.cudnn.lib
|
|
|
|
for module in (nvidia.cublas.lib, nvidia.cudnn.lib):
|
|
for d in getattr(module, "__path__", []):
|
|
for lib in glob.glob(os.path.join(d, "lib*.so*")):
|
|
try:
|
|
ctypes.CDLL(lib, mode=ctypes.RTLD_GLOBAL)
|
|
except OSError:
|
|
pass
|
|
except Exception as e:
|
|
logger.debug(f"Note: Preload of NVIDIA dynamic libraries skipped/failed: {e}")
|
|
|
|
|
|
def _align_with_whisperx(
|
|
raw_segments: list[dict],
|
|
wav_16k: Path,
|
|
language: str,
|
|
align_cache: dict | None = None,
|
|
) -> list[Word]:
|
|
"""Applies CTC Forced Alignment using WhisperX / Wav2Vec2."""
|
|
try:
|
|
import torch
|
|
import whisperx
|
|
except ImportError:
|
|
logger.warning("[!] 'whisperx' or 'torch' not installed. Skipping forced alignment.")
|
|
return []
|
|
|
|
try:
|
|
align_device = (
|
|
"cuda"
|
|
if torch.cuda.is_available()
|
|
else ("mps" if getattr(torch.backends, "mps", None) and torch.backends.mps.is_available() else "cpu")
|
|
)
|
|
|
|
logger.info(f"[*] Running CTC forced alignment ({align_device})...")
|
|
if isinstance(align_cache, dict) and language in align_cache:
|
|
align_model, align_meta = align_cache[language]
|
|
else:
|
|
align_model, align_meta = whisperx.load_align_model(
|
|
language_code=language, device=align_device
|
|
)
|
|
if isinstance(align_cache, dict):
|
|
align_cache[language] = (align_model, align_meta)
|
|
audio_data = whisperx.load_audio(str(wav_16k))
|
|
aligned_result = whisperx.align(
|
|
raw_segments,
|
|
align_model,
|
|
align_meta,
|
|
audio_data,
|
|
align_device,
|
|
return_char_alignments=False,
|
|
)
|
|
|
|
aligned_words: list[Word] = []
|
|
for w in aligned_result.get("word_segments", []):
|
|
if "start" in w and "end" in w and w.get("word"):
|
|
text = str(w["word"]).strip()
|
|
s = float(w["start"])
|
|
e = float(w["end"])
|
|
if text and e > s:
|
|
aligned_words.append(Word(text=text, start=s, end=e))
|
|
|
|
logger.info(f"[*] CTC alignment succeeded: {len(aligned_words)} words aligned.")
|
|
return aligned_words
|
|
except Exception as e:
|
|
logger.warning(f"[*] CTC forced alignment failed ({e}). Falling back to Whisper timestamps.")
|
|
return []
|
|
|
|
|
|
def _transcribe_mlx(
|
|
wav: Path, model_size: str, language: str = "en"
|
|
) -> tuple[list[Word], list[dict]]:
|
|
import mlx_whisper
|
|
|
|
repo = MLX_MODEL_MAP.get(model_size, model_size)
|
|
result = mlx_whisper.transcribe(
|
|
str(wav),
|
|
path_or_hf_repo=repo,
|
|
language=language,
|
|
word_timestamps=True,
|
|
condition_on_previous_text=False,
|
|
initial_prompt=VERBATIM_PROMPT,
|
|
verbose=False,
|
|
)
|
|
words: list[Word] = []
|
|
raw_segments: list[dict] = []
|
|
|
|
for seg in result.get("segments", []):
|
|
raw_segments.append({
|
|
"start": float(seg["start"]),
|
|
"end": float(seg["end"]),
|
|
"text": str(seg.get("text", "")).strip(),
|
|
})
|
|
for w in seg.get("words", []) or []:
|
|
text = (w.get("word") or w.get("text") or "").strip()
|
|
if text:
|
|
words.append(Word(text=text, start=float(w["start"]), end=float(w["end"])))
|
|
return words, raw_segments
|
|
|
|
|
|
def _load_faster_whisper_model(model_size: str):
|
|
preload_nvidia_libs()
|
|
from faster_whisper import WhisperModel
|
|
|
|
target_device = "cuda" if platform.system() == "Linux" else "cpu"
|
|
target_compute = "float16" if target_device == "cuda" else "int8"
|
|
|
|
try:
|
|
return WhisperModel(model_size, device=target_device, compute_type=target_compute)
|
|
except Exception as e:
|
|
logger.info(f"[*] {target_device.upper()} init failed ({e}), falling back to CPU.")
|
|
return WhisperModel(model_size, device="cpu", compute_type="int8")
|
|
|
|
|
|
def _transcribe_faster_whisper(
|
|
wav: Path,
|
|
model_size: str,
|
|
language: str = "en",
|
|
model=None,
|
|
) -> tuple[list[Word], list[dict]]:
|
|
from tqdm import tqdm
|
|
|
|
if model is None:
|
|
model = _load_faster_whisper_model(model_size)
|
|
|
|
segments, meta = model.transcribe(
|
|
str(wav),
|
|
language=language,
|
|
word_timestamps=True,
|
|
vad_filter=False,
|
|
beam_size=5,
|
|
condition_on_previous_text=False,
|
|
initial_prompt=VERBATIM_PROMPT,
|
|
)
|
|
|
|
words: list[Word] = []
|
|
raw_segments: list[dict] = []
|
|
pbar = tqdm(
|
|
total=round(meta.duration, 2),
|
|
unit="s",
|
|
disable=not logger.isEnabledFor(logging.INFO),
|
|
desc="Transcribing",
|
|
)
|
|
last = 0.0
|
|
for seg in segments:
|
|
raw_segments.append({
|
|
"start": float(seg.start),
|
|
"end": float(seg.end),
|
|
"text": seg.text.strip(),
|
|
})
|
|
if seg.words:
|
|
for w in seg.words:
|
|
text = w.word.strip()
|
|
if text:
|
|
words.append(Word(text=text, start=float(w.start), end=float(w.end)))
|
|
pbar.update(max(0.0, seg.end - last))
|
|
last = seg.end
|
|
pbar.close()
|
|
return words, raw_segments
|
|
|
|
|
|
def _cache_key(model: str, backend: str, language: str) -> str:
|
|
return hashlib.sha256(
|
|
json.dumps([backend, language, model]).encode()
|
|
).hexdigest()[:12]
|
|
|
|
|
|
def _cache_path(video: Path, track_idx: int, model: str, backend: str, language: str) -> Path:
|
|
return video.with_name(
|
|
f"{video.name}.trk{track_idx}.{_cache_key(model, backend, language)}.whisper.json"
|
|
)
|
|
|
|
|
|
def _read_cache(path: Path) -> list[Word] | None:
|
|
try:
|
|
data = json.loads(path.read_text(encoding="utf-8"))
|
|
words = data.get("words") if isinstance(data, dict) else None
|
|
if not isinstance(words, list):
|
|
return None
|
|
return [Word(**w) for w in words]
|
|
except (OSError, ValueError, TypeError, KeyError):
|
|
return None
|
|
|
|
|
|
def _write_cache(path: Path, words: list[Word]) -> None:
|
|
if not words:
|
|
return
|
|
payload = {"words": [asdict(w) for w in words]}
|
|
try:
|
|
tmp = path.with_suffix(path.suffix + ".tmp")
|
|
tmp.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8")
|
|
os.replace(tmp, path)
|
|
except OSError as e:
|
|
logger.warning(f"[!] Could not write cache {path.name}: {e}")
|
|
|
|
|
|
def _write_text_atomic(path: Path, text: str) -> None:
|
|
tmp = path.with_suffix(path.suffix + ".tmp")
|
|
tmp.write_text(text, encoding="utf-8")
|
|
os.replace(tmp, path)
|
|
|
|
|
|
def load_or_transcribe(
|
|
video: Path,
|
|
wav_16k: Path,
|
|
model: str,
|
|
language: str,
|
|
track_idx: int,
|
|
no_cache: bool,
|
|
backend: str,
|
|
stt_model=None,
|
|
align_cache: dict | None = None,
|
|
extract: Callable[[], None] | None = None,
|
|
) -> list[Word]:
|
|
cache_path = _cache_path(video, track_idx, model, backend, language)
|
|
|
|
if not no_cache:
|
|
cached = _read_cache(cache_path)
|
|
if cached is not None:
|
|
logger.info(f"[*] Reusing cached transcription: {cache_path.name}")
|
|
return cached
|
|
|
|
if extract is not None:
|
|
extract()
|
|
|
|
logger.info(f"[*] Transcribing ({backend}) | Model: {model} | File: {video.name}")
|
|
|
|
if backend == "mlx":
|
|
words, raw_segments = _transcribe_mlx(wav_16k, model, language=language)
|
|
else:
|
|
words, raw_segments = _transcribe_faster_whisper(
|
|
wav_16k, model, language=language, model=stt_model
|
|
)
|
|
|
|
if raw_segments:
|
|
aligned_words = _align_with_whisperx(
|
|
raw_segments=raw_segments,
|
|
wav_16k=wav_16k,
|
|
language=language,
|
|
align_cache=align_cache,
|
|
)
|
|
if aligned_words:
|
|
words = aligned_words
|
|
|
|
_write_cache(cache_path, words)
|
|
return words
|
|
|
|
|
|
def normalize_token(s: str) -> str:
|
|
return s.strip(" .,?!\"'…-—–:;()[]{}*~`").lower()
|
|
|
|
|
|
def build_cuts(
|
|
words: list[Word],
|
|
duration: float,
|
|
single_fillers: set[str],
|
|
phrase_fillers: set[str],
|
|
max_silence: float,
|
|
pad: float,
|
|
) -> list[tuple[float, float, str]]:
|
|
if duration <= 0:
|
|
return []
|
|
|
|
cuts: list[tuple[float, float, str]] = []
|
|
n = len(words)
|
|
drop_indices: set[int] = set()
|
|
|
|
for phrase in phrase_fillers:
|
|
tokens = [normalize_token(t) for t in phrase.split() if normalize_token(t)]
|
|
plen = len(tokens)
|
|
if plen == 0 or plen > n:
|
|
continue
|
|
for i in range(n - plen + 1):
|
|
window = [normalize_token(words[i + k].text) for k in range(plen)]
|
|
if window == tokens:
|
|
for k in range(plen):
|
|
drop_indices.add(i + k)
|
|
|
|
for i in range(n - 1):
|
|
w1, w2 = words[i], words[i + 1]
|
|
t1, t2 = normalize_token(w1.text), normalize_token(w2.text)
|
|
if t1 and t1 == t2 and (w2.start - w1.end) < 0.8:
|
|
drop_indices.add(i)
|
|
|
|
for i in range(n - 3):
|
|
pair1 = (normalize_token(words[i].text), normalize_token(words[i + 1].text))
|
|
pair2 = (normalize_token(words[i + 2].text), normalize_token(words[i + 3].text))
|
|
if (
|
|
pair1[0] and pair1[1]
|
|
and pair1 == pair2
|
|
and (words[i + 2].start - words[i + 1].end) < 0.8
|
|
):
|
|
drop_indices.add(i)
|
|
drop_indices.add(i + 1)
|
|
|
|
for i, w in enumerate(words):
|
|
tok = normalize_token(w.text)
|
|
if not tok:
|
|
drop_indices.add(i)
|
|
continue
|
|
if tok in single_fillers or HESITATION_REGEX.match(tok):
|
|
drop_indices.add(i)
|
|
|
|
prev_end = 0.0
|
|
for i, w in enumerate(words):
|
|
silence_gap = w.start - prev_end
|
|
if silence_gap > max_silence:
|
|
s = 0.0 if prev_end == 0.0 else prev_end + pad
|
|
e = max(s, w.start - pad)
|
|
if e - s > 0.04:
|
|
cuts.append((s, e, "silence"))
|
|
|
|
if i in drop_indices:
|
|
left_bound = words[i - 1].end if (i > 0 and (i - 1) not in drop_indices) else 0.0
|
|
right_bound = (
|
|
words[i + 1].start if (i < n - 1 and (i + 1) not in drop_indices) else duration
|
|
)
|
|
cut_s = max(left_bound, w.start - pad / 2)
|
|
cut_e = min(right_bound, w.end + pad / 2)
|
|
if cut_e > cut_s:
|
|
cuts.append((cut_s, cut_e, "filler"))
|
|
|
|
prev_end = max(prev_end, w.end)
|
|
|
|
if duration - prev_end > max_silence:
|
|
s = prev_end + pad
|
|
if duration - s > 0.04:
|
|
cuts.append((min(s, duration), duration, "silence"))
|
|
|
|
if not cuts:
|
|
return []
|
|
|
|
cuts.sort(key=lambda x: x[0])
|
|
merged: list[tuple[float, float, str]] = [cuts[0]]
|
|
for s, e, r in cuts[1:]:
|
|
ps, pe, pr = merged[-1]
|
|
if s <= pe + 0.01:
|
|
merged[-1] = (ps, max(pe, e), pr if pr == r else "mixed")
|
|
else:
|
|
merged.append((s, e, r))
|
|
|
|
return merged
|
|
|
|
|
|
def build_timeline(
|
|
cuts: list[tuple[float, float, str]], duration: float, min_keep_dur: float = 0.08
|
|
) -> list[Segment]:
|
|
timeline: list[Segment] = []
|
|
cursor = 0.0
|
|
|
|
for s, e, _ in cuts:
|
|
s = min(max(s, 0.0), duration)
|
|
e = min(max(e, 0.0), duration)
|
|
if s > cursor:
|
|
timeline.append(Segment(start=cursor, end=s, action="keep"))
|
|
if e > s:
|
|
timeline.append(Segment(start=s, end=e, action="drop"))
|
|
cursor = max(cursor, e)
|
|
|
|
if cursor < duration:
|
|
timeline.append(Segment(start=cursor, end=duration, action="keep"))
|
|
|
|
filtered: list[Segment] = []
|
|
for seg in timeline:
|
|
if seg.action == "keep" and (seg.end - seg.start) < min_keep_dur:
|
|
filtered.append(Segment(start=seg.start, end=seg.end, action="drop"))
|
|
else:
|
|
filtered.append(seg)
|
|
|
|
if not filtered:
|
|
return [Segment(0.0, duration, "keep")]
|
|
|
|
consolidated: list[Segment] = [filtered[0]]
|
|
for s in filtered[1:]:
|
|
last = consolidated[-1]
|
|
if s.action == last.action:
|
|
consolidated[-1] = Segment(start=last.start, end=s.end, action=last.action)
|
|
else:
|
|
consolidated.append(s)
|
|
|
|
return consolidated
|
|
|
|
|
|
def frames_to_tc(frames: int, fps: float) -> str:
|
|
"""Format frame index securely to Kdenlive/MLT strictly compliant HH:MM:SS.mmm format."""
|
|
if frames <= 0:
|
|
return "00:00:00.000"
|
|
secs = frames / fps
|
|
h = int(secs / 3600)
|
|
m = int((secs % 3600) / 60)
|
|
s = int(secs % 60)
|
|
ms = int(round((secs - int(secs)) * 1000))
|
|
if ms >= 1000:
|
|
s += 1
|
|
ms -= 1000
|
|
if s >= 60:
|
|
m += 1
|
|
s -= 60
|
|
if m >= 60:
|
|
h += 1
|
|
m -= 60
|
|
return f"{h:02d}:{m:02d}:{s:02d}.{ms:03d}"
|
|
|
|
|
|
def compute_jl_cut_shifts(
|
|
keep_segments: list[Segment],
|
|
fps: float,
|
|
jl_mode: Literal["j", "l", "off"],
|
|
jl_frames: int,
|
|
) -> list[int]:
|
|
count = len(keep_segments)
|
|
shifts = [0] * count
|
|
if jl_mode == "off" or jl_frames <= 0 or count < 2:
|
|
return shifts
|
|
|
|
for i in range(count - 1):
|
|
seg = keep_segments[i]
|
|
next_seg = keep_segments[i + 1]
|
|
|
|
cur_in = int(round(seg.start * fps))
|
|
cur_out = int(round(seg.end * fps))
|
|
cur_dur = max(1, cur_out - cur_in)
|
|
|
|
next_in = int(round(next_seg.start * fps))
|
|
next_out = int(round(next_seg.end * fps))
|
|
next_dur = max(1, next_out - next_in)
|
|
|
|
handle = max(0, next_in - cur_out - 1)
|
|
|
|
if jl_mode == "j":
|
|
max_allowed = min(jl_frames, handle, cur_dur // 4, next_dur // 4)
|
|
shifts[i] = max_allowed
|
|
elif jl_mode == "l":
|
|
max_allowed = min(jl_frames, handle, cur_dur // 4, next_dur // 4)
|
|
shifts[i] = -max_allowed
|
|
|
|
shifts[count - 1] = 0
|
|
return shifts
|
|
|
|
|
|
def _generate_multi_kdenlive_project(
|
|
track_data_list: list[VideoTrackData],
|
|
output: Path,
|
|
jl_mode: Literal["j", "l", "off"] = "j",
|
|
jl_frames: int = 2,
|
|
) -> None:
|
|
base = track_data_list[0]
|
|
fps_num, fps_den = base.fps_num, base.fps_den
|
|
width, height = base.width, base.height
|
|
fps = base.fps
|
|
|
|
gcd = math.gcd(width, height)
|
|
dar_num = width // gcd
|
|
dar_den = height // gcd
|
|
|
|
total_timeline_frames = 0
|
|
total_kept_segments = 0
|
|
|
|
for td in track_data_list:
|
|
for seg in td.timeline:
|
|
if seg.action == "keep":
|
|
start_f = int(round(seg.start * fps))
|
|
end_f = int(round(seg.end * fps))
|
|
dur_f = max(1, end_f - start_f)
|
|
total_timeline_frames += dur_f
|
|
total_kept_segments += 1
|
|
|
|
last_frame_index = max(0, total_timeline_frames - 1)
|
|
|
|
logger.info(
|
|
f"[*] Generating Kdenlive Project: {total_kept_segments} cuts | "
|
|
f"{total_timeline_frames} frames ({total_timeline_frames / fps:.2f}s) across {len(track_data_list)} video(s)..."
|
|
)
|
|
|
|
mlt = ET.Element("mlt", {
|
|
"LC_NUMERIC": "C",
|
|
"version": "7.22.0",
|
|
"producer": "main_bin",
|
|
"root": str(base.video.parent.resolve()),
|
|
})
|
|
|
|
ET.SubElement(mlt, "profile", {
|
|
"description": "automatic",
|
|
"width": str(width),
|
|
"height": str(height),
|
|
"progressive": "1",
|
|
"sample_aspect_num": "1",
|
|
"sample_aspect_den": "1",
|
|
"display_aspect_num": str(dar_num),
|
|
"display_aspect_den": str(dar_den),
|
|
"frame_rate_num": str(fps_num),
|
|
"frame_rate_den": str(fps_den),
|
|
"colorspace": "709",
|
|
})
|
|
|
|
def add_prop(parent: ET.Element, name: str, val: str) -> None:
|
|
p = ET.SubElement(parent, "property", {"name": name})
|
|
p.text = val
|
|
|
|
producer0 = ET.SubElement(mlt, "producer", {
|
|
"id": "producer0",
|
|
})
|
|
add_prop(producer0, "length", frames_to_tc(total_timeline_frames, fps))
|
|
add_prop(producer0, "eof", "continue")
|
|
add_prop(producer0, "resource", "black")
|
|
add_prop(producer0, "mlt_service", "color")
|
|
add_prop(producer0, "kdenlive:playlistid", "black_track")
|
|
add_prop(producer0, "mlt_image_format", "rgba")
|
|
add_prop(producer0, "aspect_ratio", "1")
|
|
|
|
use_system = bool(track_data_list) and all(
|
|
td.audio_stream_count >= 2 for td in track_data_list
|
|
)
|
|
|
|
for idx, td in enumerate(track_data_list):
|
|
src_path = str(td.video.resolve())
|
|
kdenlive_bin_id = str(idx + 1)
|
|
astream_idx = str(td.audio_track_idx)
|
|
|
|
# Track 1 (Voice) Producer Chain
|
|
chain_a = ET.SubElement(mlt, "chain", {"id": f"chain0_{idx}"})
|
|
add_prop(chain_a, "resource", src_path)
|
|
add_prop(chain_a, "mlt_service", "avformat-novalidate")
|
|
add_prop(chain_a, "vstream", "-1")
|
|
add_prop(chain_a, "astream", astream_idx)
|
|
add_prop(chain_a, "audio_index", astream_idx)
|
|
add_prop(chain_a, "video_index", "-1")
|
|
add_prop(chain_a, "set.test_audio", "0")
|
|
add_prop(chain_a, "set.test_video", "1")
|
|
add_prop(chain_a, "kdenlive:id", kdenlive_bin_id)
|
|
|
|
# Track 2 (System) Producer Chain
|
|
if use_system:
|
|
sys_stream = "1" if td.audio_track_idx == 0 else "0"
|
|
chain_a2 = ET.SubElement(mlt, "chain", {"id": f"chain_a2_{idx}"})
|
|
add_prop(chain_a2, "resource", src_path)
|
|
add_prop(chain_a2, "mlt_service", "avformat-novalidate")
|
|
add_prop(chain_a2, "vstream", "-1")
|
|
add_prop(chain_a2, "astream", sys_stream)
|
|
add_prop(chain_a2, "audio_index", sys_stream)
|
|
add_prop(chain_a2, "video_index", "-1")
|
|
add_prop(chain_a2, "set.test_audio", "0")
|
|
add_prop(chain_a2, "set.test_video", "1")
|
|
add_prop(chain_a2, "kdenlive:id", kdenlive_bin_id)
|
|
|
|
# Video Producer Chain
|
|
chain_v = ET.SubElement(mlt, "chain", {"id": f"chain1_{idx}"})
|
|
add_prop(chain_v, "resource", src_path)
|
|
add_prop(chain_v, "mlt_service", "avformat-novalidate")
|
|
add_prop(chain_v, "vstream", "0")
|
|
add_prop(chain_v, "astream", "-1")
|
|
add_prop(chain_v, "audio_index", "-1")
|
|
add_prop(chain_v, "video_index", "0")
|
|
add_prop(chain_v, "set.test_audio", "1")
|
|
add_prop(chain_v, "set.test_video", "0")
|
|
add_prop(chain_v, "kdenlive:id", kdenlive_bin_id)
|
|
|
|
# 1. Audio Track 1 (Voice)
|
|
playlist0 = ET.SubElement(mlt, "playlist", {"id": "playlist0"})
|
|
add_prop(playlist0, "kdenlive:audio_track", "1")
|
|
playlist1 = ET.SubElement(mlt, "playlist", {"id": "playlist1"})
|
|
add_prop(playlist1, "kdenlive:audio_track", "1")
|
|
|
|
tractor0 = ET.SubElement(mlt, "tractor", {
|
|
"id": "tractor0",
|
|
"in": "00:00:00.000",
|
|
"out": frames_to_tc(last_frame_index, fps),
|
|
})
|
|
add_prop(tractor0, "kdenlive:audio_track", "1")
|
|
add_prop(tractor0, "kdenlive:timeline_active", "1")
|
|
add_prop(tractor0, "kdenlive:track_name", "Voice")
|
|
ET.SubElement(tractor0, "track", {"hide": "video", "producer": "playlist0"})
|
|
ET.SubElement(tractor0, "track", {"hide": "video", "producer": "playlist1"})
|
|
|
|
# 2. Audio Track 2 (System Audio)
|
|
if use_system:
|
|
playlist4 = ET.SubElement(mlt, "playlist", {"id": "playlist4"})
|
|
add_prop(playlist4, "kdenlive:audio_track", "1")
|
|
playlist5 = ET.SubElement(mlt, "playlist", {"id": "playlist5"})
|
|
add_prop(playlist5, "kdenlive:audio_track", "1")
|
|
|
|
tractor_a2 = ET.SubElement(mlt, "tractor", {
|
|
"id": "tractor_a2",
|
|
"in": "00:00:00.000",
|
|
"out": frames_to_tc(last_frame_index, fps),
|
|
})
|
|
add_prop(tractor_a2, "kdenlive:audio_track", "1")
|
|
add_prop(tractor_a2, "kdenlive:timeline_active", "1")
|
|
add_prop(tractor_a2, "kdenlive:track_name", "System")
|
|
ET.SubElement(tractor_a2, "track", {"hide": "video", "producer": "playlist4"})
|
|
ET.SubElement(tractor_a2, "track", {"hide": "video", "producer": "playlist5"})
|
|
|
|
# 3. Video Track 1
|
|
playlist2 = ET.SubElement(mlt, "playlist", {"id": "playlist2"})
|
|
ET.SubElement(mlt, "playlist", {"id": "playlist3"})
|
|
|
|
tractor1 = ET.SubElement(mlt, "tractor", {
|
|
"id": "tractor1",
|
|
"in": "00:00:00.000",
|
|
"out": frames_to_tc(last_frame_index, fps),
|
|
})
|
|
add_prop(tractor1, "kdenlive:timeline_active", "1")
|
|
ET.SubElement(tractor1, "track", {"hide": "audio", "producer": "playlist2"})
|
|
ET.SubElement(tractor1, "track", {"hide": "audio", "producer": "playlist3"})
|
|
|
|
for idx, td in enumerate(track_data_list):
|
|
src_path = str(td.video.resolve())
|
|
kdenlive_bin_id = str(idx + 1)
|
|
chain_bin = ET.SubElement(mlt, "chain", {"id": f"chain2_{idx}"})
|
|
add_prop(chain_bin, "resource", src_path)
|
|
add_prop(chain_bin, "mlt_service", "avformat-novalidate")
|
|
add_prop(chain_bin, "audio_index", str(td.audio_track_idx))
|
|
add_prop(chain_bin, "video_index", "0")
|
|
add_prop(chain_bin, "vstream", "0")
|
|
add_prop(chain_bin, "astream", str(td.audio_track_idx))
|
|
add_prop(chain_bin, "kdenlive:id", kdenlive_bin_id)
|
|
|
|
groups = []
|
|
current_tl_audio_frame = 0
|
|
current_tl_video_frame = 0
|
|
|
|
for idx, td in enumerate(track_data_list):
|
|
kdenlive_bin_id = str(idx + 1)
|
|
keep_segments = [s for s in td.timeline if s.action == "keep"]
|
|
shifts = compute_jl_cut_shifts(keep_segments, fps, jl_mode=jl_mode, jl_frames=jl_frames)
|
|
|
|
for k, seg in enumerate(keep_segments):
|
|
in_frame = int(round(seg.start * fps))
|
|
out_frame_target = int(round(seg.end * fps))
|
|
seg_dur_frames = max(1, out_frame_target - in_frame)
|
|
a_out_frame = in_frame + seg_dur_frames - 1
|
|
|
|
# Audio 1 Entry (Voice)
|
|
a_entry = ET.SubElement(playlist0, "entry", {
|
|
"producer": f"chain0_{idx}",
|
|
"in": frames_to_tc(in_frame, fps),
|
|
"out": frames_to_tc(a_out_frame, fps),
|
|
})
|
|
add_prop(a_entry, "kdenlive:id", kdenlive_bin_id)
|
|
|
|
# Audio 2 Entry (System Audio - cut in lockstep with Voice)
|
|
if use_system:
|
|
a2_entry = ET.SubElement(playlist4, "entry", {
|
|
"producer": f"chain_a2_{idx}",
|
|
"in": frames_to_tc(in_frame, fps),
|
|
"out": frames_to_tc(a_out_frame, fps),
|
|
})
|
|
add_prop(a2_entry, "kdenlive:id", kdenlive_bin_id)
|
|
|
|
shift_prev = shifts[k - 1] if k > 0 else 0
|
|
shift_cur = shifts[k]
|
|
|
|
v_in_frame = in_frame + shift_prev
|
|
v_out_frame = a_out_frame + shift_cur
|
|
|
|
# Video Entry
|
|
v_entry = ET.SubElement(playlist2, "entry", {
|
|
"producer": f"chain1_{idx}",
|
|
"in": frames_to_tc(v_in_frame, fps),
|
|
"out": frames_to_tc(v_out_frame, fps),
|
|
})
|
|
add_prop(v_entry, "kdenlive:id", kdenlive_bin_id)
|
|
|
|
# Group timeline elements (0: Voice, [1: System], Video last)
|
|
children: list[dict] = [
|
|
{"data": f"0:{current_tl_audio_frame}", "leaf": "clip", "type": "Leaf"},
|
|
]
|
|
if use_system:
|
|
children.append(
|
|
{"data": f"1:{current_tl_audio_frame}", "leaf": "clip", "type": "Leaf"}
|
|
)
|
|
video_group_idx = 2 if use_system else 1
|
|
children.append(
|
|
{"data": f"{video_group_idx}:{current_tl_video_frame}", "leaf": "clip", "type": "Leaf"}
|
|
)
|
|
groups.append({
|
|
"children": children,
|
|
"type": "Normal",
|
|
})
|
|
|
|
current_tl_audio_frame += seg_dur_frames
|
|
current_tl_video_frame += (v_out_frame - v_in_frame + 1)
|
|
|
|
seq_uuid = str(uuid.uuid4())
|
|
seq_uuid_str = f"{{{seq_uuid}}}"
|
|
|
|
sequence = ET.SubElement(mlt, "tractor", {
|
|
"id": seq_uuid_str,
|
|
"in": "00:00:00.000",
|
|
"out": "00:00:00.000",
|
|
})
|
|
add_prop(sequence, "kdenlive:uuid", seq_uuid_str)
|
|
add_prop(sequence, "kdenlive:clipname", "Sequence 1")
|
|
add_prop(sequence, "kdenlive:sequenceproperties.groups", json.dumps(groups, separators=(',', ':'), indent=4))
|
|
|
|
# Timeline stack order: Black track -> Audio 1 -> Audio 2 -> Video 1
|
|
ET.SubElement(sequence, "track", {"producer": "producer0"})
|
|
ET.SubElement(sequence, "track", {"producer": "tractor0"})
|
|
if use_system:
|
|
ET.SubElement(sequence, "track", {"producer": "tractor_a2"})
|
|
ET.SubElement(sequence, "track", {"producer": "tractor1"})
|
|
|
|
main_bin = ET.SubElement(mlt, "playlist", {"id": "main_bin"})
|
|
add_prop(main_bin, "kdenlive:docproperties.uuid", seq_uuid_str)
|
|
add_prop(main_bin, "kdenlive:docproperties.version", "1.1")
|
|
add_prop(main_bin, "xml_retain", "1")
|
|
ET.SubElement(main_bin, "entry", {
|
|
"producer": seq_uuid_str,
|
|
"in": "00:00:00.000",
|
|
"out": frames_to_tc(last_frame_index, fps),
|
|
})
|
|
for idx, td in enumerate(track_data_list):
|
|
src_out_frame = int(round(td.duration * fps)) - 1
|
|
ET.SubElement(main_bin, "entry", {
|
|
"producer": f"chain2_{idx}",
|
|
"in": "00:00:00.000",
|
|
"out": frames_to_tc(src_out_frame, fps),
|
|
})
|
|
|
|
proj_tractor = ET.SubElement(mlt, "tractor", {
|
|
"id": "tractor2",
|
|
"in": "00:00:00.000",
|
|
"out": frames_to_tc(last_frame_index, fps),
|
|
})
|
|
add_prop(proj_tractor, "kdenlive:projectTractor", "1")
|
|
ET.SubElement(proj_tractor, "track", {
|
|
"producer": seq_uuid_str,
|
|
"in": "00:00:00.000",
|
|
"out": frames_to_tc(last_frame_index, fps),
|
|
})
|
|
|
|
tree = ET.ElementTree(mlt)
|
|
if hasattr(ET, "indent"):
|
|
ET.indent(tree, space=" ", level=0)
|
|
|
|
kdenlive_path = output.with_suffix(".kdenlive")
|
|
tmp_path = kdenlive_path.with_suffix(".kdenlive.tmp")
|
|
tree.write(tmp_path, encoding="utf-8", xml_declaration=True)
|
|
os.replace(tmp_path, kdenlive_path)
|
|
logger.info(f"[OK] Kdenlive Project saved -> {kdenlive_path.name}")
|
|
|
|
def map_words_to_edited_timeline(
|
|
track_data_list: list[VideoTrackData],
|
|
) -> list[tuple[float, float, str]]:
|
|
mapped: list[tuple[float, float, str]] = []
|
|
accumulated_offset = 0.0
|
|
|
|
for td in track_data_list:
|
|
fps = td.fps
|
|
kept_segs = [s for s in td.timeline if s.action == "keep"]
|
|
seg_positions: list[tuple[Segment, float]] = []
|
|
accumulated_offset_frames = 0
|
|
for seg in kept_segs:
|
|
seg_tl_start = accumulated_offset_frames / fps
|
|
seg_positions.append((seg, seg_tl_start))
|
|
dur_f = max(1, int(round(seg.end * fps)) - int(round(seg.start * fps)))
|
|
accumulated_offset_frames += dur_f
|
|
|
|
for w in td.words:
|
|
w_text = w.text.strip()
|
|
if not w_text:
|
|
continue
|
|
|
|
for seg, seg_tl_start in seg_positions:
|
|
overlap_start = max(w.start, seg.start)
|
|
overlap_end = min(w.end, seg.end)
|
|
|
|
if overlap_end > overlap_start:
|
|
overlap_dur = overlap_end - overlap_start
|
|
word_dur = max(0.01, w.end - w.start)
|
|
if overlap_dur / word_dur >= 0.35 or overlap_dur >= 0.08:
|
|
ns = seg_tl_start + (overlap_start - seg.start)
|
|
ne = seg_tl_start + (overlap_end - seg.start)
|
|
if ne > ns:
|
|
mapped.append((
|
|
accumulated_offset + ns,
|
|
accumulated_offset + ne,
|
|
w_text,
|
|
))
|
|
break
|
|
|
|
accumulated_offset += accumulated_offset_frames / fps
|
|
|
|
return mapped
|
|
|
|
|
|
LLM_PROMPT_LINES: tuple[str, ...] = (
|
|
"SYSTEM PROMPT / INSTRUCTIONS:",
|
|
"You are an elite YouTube packaging strategist and retention specialist.",
|
|
"Your single goal: Maximize Click-Through Rate (CTR), Average Percentage Viewed (APV/Retention), and Subscriber Conversion.",
|
|
"",
|
|
"Analyze the provided transcript timeline (which has already been cut for pacing with dead silences and hesitations removed) and generate high-performance video assets.",
|
|
"",
|
|
"---",
|
|
"",
|
|
"### DELIVERABLES REQUIRED",
|
|
"",
|
|
"#### 1. PACKAGING CONCEPTS (A/B/C Test Pairs for Maximum CTR)",
|
|
"Provide exactly 3 distinct packaging angles. Each angle MUST be an interconnected Title + Thumbnail Concept pair.",
|
|
"- Titles must be under 55 characters, front-load high stakes, avoid disappointment, and open compelling curiosity gaps.",
|
|
"- Thumbnail concepts must specify clear visual framing, high-contrast focal points, clean backgrounds, emotional expressions, and an optional 1-3 word punchy text overlay.",
|
|
"",
|
|
"Angle A (High-Stakes Challenge / Conflict / Curiosity):",
|
|
"- Title:",
|
|
"- Thumbnail Visual:",
|
|
"- Text Overlay:",
|
|
"",
|
|
"Angle B (Transformation / Extreme Result / Value Delivery):",
|
|
"- Title:",
|
|
"- Thumbnail Visual:",
|
|
"- Text Overlay:",
|
|
"",
|
|
"Angle C (Contrarian / Unconventional / 'I Was Wrong' Angle):",
|
|
"- Title:",
|
|
"- Thumbnail Visual:",
|
|
"- Text Overlay:",
|
|
"",
|
|
"#### 2. HOOK-FIRST DESCRIPTION SUMMARY",
|
|
"- Write exactly one punchy, high-retention paragraph (3-4 sentences max).",
|
|
"- Sentence 1: The core hook stating the primary objective or premise.",
|
|
"- Sentence 2: The tension, struggle, or key pivot point encountered.",
|
|
"- Sentence 3: The outcome/payoff with primary search keywords naturally embedded.",
|
|
"",
|
|
"#### 3. RETENTION-DRIVEN VIDEO CHAPTERS",
|
|
"- Format: `00:00 - Chapter Title`",
|
|
"- The first chapter MUST start at `00:00`.",
|
|
"- Space chapters every 90 seconds to 4 minutes based on key milestone transitions.",
|
|
"- Use curiosity-driven milestone titles (e.g. 'The fatal mistake', 'Refactoring everything', 'The final run').",
|
|
"",
|
|
"---",
|
|
"[EDITED VIDEO TRANSCRIPT TIMELINE STARTS BELOW]",
|
|
"",
|
|
)
|
|
|
|
|
|
def generate_combined_transcripts(
|
|
track_data_list: list[VideoTrackData],
|
|
output_path: Path,
|
|
*,
|
|
max_line_chars: int = 42,
|
|
max_line_words: int = 7,
|
|
max_duration_sec: float = 3.2,
|
|
min_duration_sec: float = 0.8,
|
|
max_gap_sec: float = 0.5,
|
|
) -> None:
|
|
def format_ts(ts: float, is_srt: bool = True) -> str:
|
|
ts = max(0.0, ts)
|
|
ms = int(round(ts * 1000)) % 1000
|
|
s = int(ts)
|
|
h, m, sec = s // 3600, (s % 3600) // 60, s % 60
|
|
if is_srt:
|
|
return f"{h:02d}:{m:02d}:{sec:02d},{ms:03d}"
|
|
return f"{h:02d}:{m:02d}:{sec:02d}"
|
|
|
|
mapped_words = map_words_to_edited_timeline(track_data_list)
|
|
|
|
if not mapped_words:
|
|
logger.warning("[!] No words remaining for transcript/caption generation.")
|
|
return
|
|
|
|
subs: list[tuple[float, float, str]] = []
|
|
chunk: list[str] = []
|
|
chunk_start: float | None = None
|
|
prev_end: float = 0.0
|
|
|
|
terminal_punct = {".", "?", "!"}
|
|
clause_punct = {",", ";", ":", "—", "-"}
|
|
|
|
for ns, ne, text in mapped_words:
|
|
if chunk_start is None:
|
|
chunk_start = ns
|
|
|
|
time_gap = ns - prev_end
|
|
current_dur = ne - chunk_start
|
|
prospective_chars = sum(len(w) for w in chunk) + len(chunk) + len(text)
|
|
prospective_words = len(chunk) + 1
|
|
|
|
last_char = chunk[-1][-1] if chunk and chunk[-1] else ""
|
|
has_clause_break = last_char in (terminal_punct | clause_punct)
|
|
is_too_long = (
|
|
prospective_chars > max_line_chars
|
|
or prospective_words > max_line_words
|
|
or current_dur > max_duration_sec
|
|
)
|
|
is_paused = bool(chunk) and (time_gap > max_gap_sec)
|
|
|
|
if chunk and (is_too_long or is_paused or (has_clause_break and current_dur >= min_duration_sec)):
|
|
sub_end = min(ns, max(prev_end, chunk_start + min_duration_sec))
|
|
subs.append((chunk_start, sub_end, " ".join(chunk)))
|
|
chunk = [text]
|
|
chunk_start = ns
|
|
else:
|
|
chunk.append(text)
|
|
|
|
prev_end = ne
|
|
|
|
if text and text[-1] in terminal_punct and (ne - chunk_start) >= min_duration_sec:
|
|
subs.append((chunk_start, ne, " ".join(chunk)))
|
|
chunk = []
|
|
chunk_start = None
|
|
|
|
if chunk and chunk_start is not None:
|
|
subs.append((chunk_start, max(prev_end, chunk_start + min_duration_sec), " ".join(chunk)))
|
|
|
|
sanitized_subs: list[tuple[float, float, str]] = []
|
|
for i, (s, e, text) in enumerate(subs):
|
|
if sanitized_subs and s < sanitized_subs[-1][1]:
|
|
s = sanitized_subs[-1][1] + 0.01
|
|
e = max(s + 0.1, e)
|
|
sanitized_subs.append((s, e, text))
|
|
|
|
srt_path = output_path.with_suffix(".srt")
|
|
lines: list[str] = []
|
|
for i, (s, e, text) in enumerate(sanitized_subs, 1):
|
|
lines.extend([str(i), f"{format_ts(s)} --> {format_ts(e)}", text, ""])
|
|
_write_text_atomic(srt_path, "\n".join(lines))
|
|
logger.info(f"[*] YouTube Captions generated -> {srt_path.name}")
|
|
|
|
llm_transcript_path = output_path.with_name(f"{output_path.stem}_llm_transcript.txt")
|
|
llm_lines: list[str] = list(LLM_PROMPT_LINES)
|
|
|
|
current_para: list[str] = []
|
|
para_start = mapped_words[0][0]
|
|
|
|
for ns, ne, text in mapped_words:
|
|
if not current_para:
|
|
para_start = ns
|
|
current_para.append(text)
|
|
time_elapsed = ne - para_start
|
|
is_sentence_end = bool(text and text[-1] in terminal_punct)
|
|
|
|
if (time_elapsed >= 45.0 and is_sentence_end) or time_elapsed >= 75.0:
|
|
llm_lines.append(f"[{format_ts(para_start, False)}] {' '.join(current_para)}")
|
|
current_para = []
|
|
|
|
if current_para:
|
|
llm_lines.append(f"[{format_ts(para_start, False)}] {' '.join(current_para)}")
|
|
|
|
_write_text_atomic(llm_transcript_path, "\n\n".join(llm_lines))
|
|
logger.info(f"[*] LLM transcript generated -> {llm_transcript_path.name}")
|
|
|
|
|
|
def _positive_int(value: str) -> int:
|
|
try:
|
|
ivalue = int(value)
|
|
except ValueError:
|
|
raise argparse.ArgumentTypeError(f"invalid int value: '{value}'")
|
|
if ivalue < 1:
|
|
raise argparse.ArgumentTypeError(f"must be >= 1 (got {ivalue})")
|
|
return ivalue
|
|
|
|
|
|
def _non_negative_int(value: str) -> int:
|
|
try:
|
|
ivalue = int(value)
|
|
except ValueError:
|
|
raise argparse.ArgumentTypeError(f"invalid int value: '{value}'")
|
|
if ivalue < 0:
|
|
raise argparse.ArgumentTypeError(f"must be >= 0 (got {ivalue})")
|
|
return ivalue
|
|
|
|
|
|
def _non_negative_float(value: str) -> float:
|
|
try:
|
|
fvalue = float(value)
|
|
except ValueError:
|
|
raise argparse.ArgumentTypeError(f"invalid float value: '{value}'")
|
|
if fvalue < 0:
|
|
raise argparse.ArgumentTypeError(f"must be >= 0 (got {fvalue})")
|
|
return fvalue
|
|
|
|
|
|
def parse_args() -> argparse.Namespace:
|
|
p = argparse.ArgumentParser(
|
|
description="cadence: High-retention audio-visual auto-editor for raw recordings."
|
|
)
|
|
p.add_argument(
|
|
"--version",
|
|
action="version",
|
|
version=f"%(prog)s {__version__}",
|
|
)
|
|
p.add_argument(
|
|
"inputs",
|
|
nargs="+",
|
|
type=Path,
|
|
help="One or more video files in sequence order.",
|
|
)
|
|
p.add_argument(
|
|
"--preset",
|
|
choices=["relaxed", "balanced", "punchy"],
|
|
default=None,
|
|
help="Apply built-in editing pacing settings (overrides defaults).",
|
|
)
|
|
p.add_argument(
|
|
"-o", "--output",
|
|
type=Path,
|
|
default=None,
|
|
help="Target output path (defaults to first video path with .kdenlive suffix).",
|
|
)
|
|
p.add_argument(
|
|
"--audio-track",
|
|
type=_positive_int,
|
|
default=1,
|
|
help="1-based OBS audio track index to analyze and map (default: 1).",
|
|
)
|
|
p.add_argument(
|
|
"--jl-mode",
|
|
choices=["j", "l", "off"],
|
|
default="j",
|
|
help="Split cut mode: 'j' (audio leads video), 'l' (video leads audio), 'off' (default: j).",
|
|
)
|
|
p.add_argument(
|
|
"--jl-frames",
|
|
type=_non_negative_int,
|
|
default=None,
|
|
help="Frame lead/lag offset for J/L split cuts to mask jump cuts (default: 2).",
|
|
)
|
|
p.add_argument(
|
|
"--captions",
|
|
action="store_true",
|
|
help="Generate synced .srt subtitles and LLM packaging transcript file.",
|
|
)
|
|
p.add_argument(
|
|
"--max-silence",
|
|
type=_non_negative_float,
|
|
default=None,
|
|
help="Max allowable silence before cutting (default: 0.50s).",
|
|
)
|
|
p.add_argument(
|
|
"--pad",
|
|
type=_non_negative_float,
|
|
default=None,
|
|
help="Breath padding around cut boundaries (default: 0.10s).",
|
|
)
|
|
p.add_argument(
|
|
"--min-keep",
|
|
type=_non_negative_float,
|
|
default=None,
|
|
help="Minimum duration for a kept cut to prevent audio popping (default: 0.08s).",
|
|
)
|
|
p.add_argument(
|
|
"--model",
|
|
default="large-v3",
|
|
help="Whisper model size/repo (default: large-v3).",
|
|
)
|
|
p.add_argument(
|
|
"--language",
|
|
default="en",
|
|
help="Spoken audio language code (default: en).",
|
|
)
|
|
p.add_argument(
|
|
"--extra-fillers",
|
|
type=str,
|
|
default="",
|
|
help="Comma-separated list of additional single filler words to remove.",
|
|
)
|
|
p.add_argument(
|
|
"--no-cache",
|
|
action="store_true",
|
|
help="Force re-transcription by ignoring existing .whisper.json caches.",
|
|
)
|
|
p.add_argument(
|
|
"-q", "--quiet",
|
|
action="store_true",
|
|
help="Silence non-essential console status output.",
|
|
)
|
|
return p.parse_args()
|
|
|
|
|
|
def main() -> None:
|
|
args = parse_args()
|
|
|
|
# Inject predefined settings based on chosen preset without clobbering user overrides
|
|
if args.preset:
|
|
preset_cfg = PACING_PRESETS[args.preset]
|
|
if args.max_silence is None:
|
|
args.max_silence = preset_cfg["max_silence"]
|
|
if args.pad is None:
|
|
args.pad = preset_cfg["pad"]
|
|
if args.min_keep is None:
|
|
args.min_keep = preset_cfg["min_keep"]
|
|
if args.jl_frames is None:
|
|
args.jl_frames = preset_cfg["jl_frames"]
|
|
if args.max_silence is None:
|
|
args.max_silence = 0.50
|
|
if args.pad is None:
|
|
args.pad = 0.10
|
|
if args.min_keep is None:
|
|
args.min_keep = 0.08
|
|
if args.jl_frames is None:
|
|
args.jl_frames = 2
|
|
|
|
logging.basicConfig(
|
|
level=logging.WARNING,
|
|
format="%(message)s",
|
|
)
|
|
logger.setLevel(logging.WARNING if args.quiet else logging.INFO)
|
|
|
|
if args.preset:
|
|
logger.info(f"[*] Applying '{args.preset.upper()}' preset: Max Silence={args.max_silence}s, Pad={args.pad}s")
|
|
|
|
check_dependencies()
|
|
|
|
input_videos: list[Path] = [f.resolve() for f in args.inputs]
|
|
for v in input_videos:
|
|
if not v.exists() or not v.is_file():
|
|
sys.exit(f"[!] Input file not found or invalid: {v}")
|
|
|
|
if args.output:
|
|
output = args.output.resolve()
|
|
else:
|
|
if len(input_videos) == 1:
|
|
output = input_videos[0].with_suffix(".kdenlive")
|
|
else:
|
|
output = input_videos[0].with_name(f"{input_videos[0].stem}_concat.kdenlive")
|
|
|
|
single_fillers = set(HESITATION_FILLERS)
|
|
if args.extra_fillers:
|
|
for f in args.extra_fillers.split(","):
|
|
norm = normalize_token(f)
|
|
if norm:
|
|
single_fillers.add(norm)
|
|
|
|
phrase_fillers = set(DISCOURSE_FILLERS)
|
|
|
|
backend = detect_backend()
|
|
stt_model = _load_faster_whisper_model(args.model) if backend != "mlx" else None
|
|
align_cache: dict = {}
|
|
|
|
track_data_list: list[VideoTrackData] = []
|
|
user_audio_track_idx = max(0, args.audio_track - 1)
|
|
|
|
ref_fps: float | None = None
|
|
ref_w = ref_h = 0
|
|
|
|
for idx, video in enumerate(input_videos, 1):
|
|
duration = probe_duration(video)
|
|
fps_num, fps_den, width, height = probe_video_format(video)
|
|
fps = fps_num / fps_den
|
|
|
|
if ref_fps is None:
|
|
ref_fps, ref_w, ref_h = fps, width, height
|
|
elif (fps, width, height) != (ref_fps, ref_w, ref_h):
|
|
sys.exit(
|
|
f"[!] {video.name} is {fps:g}fps {width}x{height}; all inputs must match "
|
|
f"the first ({ref_fps:g}fps {ref_w}x{ref_h})."
|
|
)
|
|
|
|
available_audio_streams = probe_audio_stream_count(video)
|
|
if available_audio_streams == 0:
|
|
sys.exit(f"[!] No audio streams found in {video.name}.")
|
|
|
|
if user_audio_track_idx >= available_audio_streams:
|
|
logger.warning(
|
|
f"[!] Warning: Audio track {args.audio_track} requested, but {video.name} "
|
|
f"only has {available_audio_streams} audio stream(s). Falling back to stream 0."
|
|
)
|
|
resolved_track_idx = 0
|
|
else:
|
|
resolved_track_idx = user_audio_track_idx
|
|
|
|
logger.info(
|
|
f"\n[*] [{idx}/{len(input_videos)}] {video.name} | "
|
|
f"{duration:.2f}s | {fps:.2f} fps | {width}x{height} | Audio stream: a:{resolved_track_idx}"
|
|
)
|
|
|
|
with tempfile.TemporaryDirectory(prefix="cadence_") as td:
|
|
wav_16k = Path(td) / "audio_16k.wav"
|
|
|
|
def do_extract() -> None:
|
|
extract_audio(video, wav_16k, track_index=resolved_track_idx)
|
|
|
|
words = load_or_transcribe(
|
|
video=video,
|
|
wav_16k=wav_16k,
|
|
model=args.model,
|
|
language=args.language,
|
|
track_idx=resolved_track_idx,
|
|
no_cache=args.no_cache,
|
|
backend=backend,
|
|
stt_model=stt_model,
|
|
align_cache=align_cache,
|
|
extract=do_extract,
|
|
)
|
|
|
|
if not words:
|
|
sys.exit(
|
|
f"[!] No speech detected in {video.name}. "
|
|
f"Check --language, --audio-track, or the recording."
|
|
)
|
|
|
|
cuts = build_cuts(
|
|
words=words,
|
|
duration=duration,
|
|
single_fillers=single_fillers,
|
|
phrase_fillers=phrase_fillers,
|
|
max_silence=args.max_silence,
|
|
pad=args.pad,
|
|
)
|
|
|
|
timeline = build_timeline(cuts, duration, min_keep_dur=args.min_keep)
|
|
|
|
kept = sum(s.end - s.start for s in timeline if s.action == "keep")
|
|
if duration > 0 and kept < 0.05 * duration:
|
|
sys.exit(
|
|
f"[!] Only {kept:.2f}s of {duration:.2f}s kept for {video.name}; "
|
|
f"refusing to write a near-empty project. "
|
|
f"Retranscribe or adjust --max-silence/--min-keep."
|
|
)
|
|
|
|
track_data_list.append(VideoTrackData(
|
|
video=video,
|
|
duration=duration,
|
|
fps_num=fps_num,
|
|
fps_den=fps_den,
|
|
width=width,
|
|
height=height,
|
|
audio_track_idx=resolved_track_idx,
|
|
audio_stream_count=available_audio_streams,
|
|
words=words,
|
|
timeline=timeline,
|
|
))
|
|
|
|
output.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
_generate_multi_kdenlive_project(
|
|
track_data_list,
|
|
output=output,
|
|
jl_mode=args.jl_mode,
|
|
jl_frames=args.jl_frames,
|
|
)
|
|
|
|
if args.captions:
|
|
generate_combined_transcripts(track_data_list, output)
|
|
|
|
logger.info(f"\n[OK] Processing complete. Master project saved to:\n {output.with_suffix('.kdenlive')}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|
|
|