Write talking-head scripts and produce Instagram reels and YouTube shorts
93
94%
Does it follow best practices?
Impact
88%
1.23xAverage score across 3 eval scenarios
Low
Low-risk findings worth noting
#!/usr/bin/env python3
"""Transcribe a take with DaVinci Resolve and hand the result to the pipeline.
Resolve's transcription returns frame-precise words AND marks every silence
as an explicit `(...)` span — measured on a synthetic take: a 4.00s silent
head and two 1.73s pauses, each reported with start and end. That is the
speech-aware silence detection tighten_vo.py's level threshold approximates.
Nothing is rendered here. Three files come out, all in shapes the pipeline
already reads, so the rest of the talking-head path is unchanged:
--out words JSON, flat [{word, start, end}] in seconds — what
gen_captions.py, plan_broll.py, trim_words.py and check_yap.py take
--cuts keep-range cut list in tighten_vo.py's schema. Replay it with
`tighten_vo.py TAKE --apply-cuts CUTS` and ffmpeg does the cutting;
verify_cut.py then proves the words survived. Resolve decides, ffmpeg
cuts — one renderer, no second edit.
--srt Resolve's auto-captions as SRT, for render_reel.py --srt
Voice isolation and the dialogue leveler are Resolve properties on the AUDIO
track item — the video item refuses them (measured) — and belong to a Resolve
render, not here. --keep-project leaves the timeline for that.
Requires Resolve STUDIO running. Transcription of a 24s take took over a
minute the first time (model load), so this never runs under a short timeout.
Usage:
resolve_transcribe.py raw/take.mp4 --out work/take.words.json
resolve_transcribe.py raw/take.mp4 --out work/take.words.json \\
--cuts work/cuts.json --srt work/take.srt --min-silence 0.5
"""
import argparse, json, subprocess, sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from resolve_bridge import connect, project_fps, DEFAULT_FPS # noqa: E402
from tighten_vo import keep_ranges, CUTS_SCHEMA # noqa: E402
SILENCE = "(...)"
DEFAULT_MARGIN = 0.12 # seconds kept on each side of a cut; 0 sounds clipped
DEFAULT_MIN_SILENCE = 0.5 # shorter pauses are speech rhythm, not dead air
def tc_to_seconds(tc, fps):
"""'HH:MM:SS:FF' at fps to seconds. Frames are exact, so no rounding."""
h, m, s, f = (int(x) for x in tc.split(":"))
return h * 3600 + m * 60 + s + f / fps
def words_from(transcription, fps):
"""Flat word list in seconds, silences dropped, matching whisper's shape."""
out = []
for seg in (transcription or {}).get("segments", []):
for w in seg.get("words", []):
text = (w.get("text") or "").strip()
if not text or text == SILENCE:
continue
out.append({"word": text, "start": round(tc_to_seconds(w["start"], fps), 3),
"end": round(tc_to_seconds(w["end"], fps), 3)})
return out
def silences_from(transcription, fps, min_s=DEFAULT_MIN_SILENCE):
"""(start, end) spans Resolve marked as silence, at least min_s long."""
spans = []
for seg in (transcription or {}).get("segments", []):
for w in seg.get("words", []):
if (w.get("text") or "").strip() == SILENCE:
a, b = tc_to_seconds(w["start"], fps), tc_to_seconds(w["end"], fps)
if b - a >= min_s:
spans.append((round(a, 3), round(b, 3)))
return spans
def srt_from(cues):
"""SRT text from (start_s, end_s, text) cues, in order."""
def stamp(t):
ms = int(round(t * 1000))
h, ms = divmod(ms, 3600000); m, ms = divmod(ms, 60000); s, ms = divmod(ms, 1000)
return f"{h:02}:{m:02}:{s:02},{ms:03}"
lines = []
for i, (a, b, text) in enumerate(cues, 1):
lines += [str(i), f"{stamp(a)} --> {stamp(b)}", text.strip(), ""]
return "\n".join(lines)
def probe_duration(path):
r = subprocess.run(["ffprobe", "-v", "error", "-show_entries", "format=duration",
"-of", "csv=p=0", str(path)], capture_output=True, text=True)
return float(r.stdout.strip() or 0)
def main():
ap = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("take")
ap.add_argument("--out", required=True, help="words JSON (flat list, seconds)")
ap.add_argument("--cuts", help="also write a tighten_vo.py cut list here")
ap.add_argument("--srt", help="also write Resolve's auto-captions as SRT here")
ap.add_argument("--min-silence", type=float, default=DEFAULT_MIN_SILENCE)
ap.add_argument("--margin", type=float, default=DEFAULT_MARGIN,
help=f"seconds kept either side of a cut (default {DEFAULT_MARGIN})")
ap.add_argument("--speaker-detection", action="store_true")
ap.add_argument("--project", default="reel-builder_transcribe")
ap.add_argument("--keep-project", action="store_true")
a = ap.parse_args()
take = Path(a.take).resolve()
if not take.exists():
sys.exit(f"Take not found: {a.take}")
resolve = connect()
pm = resolve.GetProjectManager()
if pm.GetCurrentProject():
pm.SaveProject()
proj = pm.LoadProject(a.project) or pm.CreateProject(a.project)
if proj is None:
sys.exit(f"Could not open Resolve project {a.project!r}. If Resolve is "
f"showing a dialog, close it and retry.")
proj.SetSetting("timelineFrameRate", str(DEFAULT_FPS))
fps = project_fps(proj)
mp = proj.GetMediaPool()
items = resolve.GetMediaStorage().AddItemListToMediaPool([str(take)]) or []
mpi = items[0] if items else next(
(c for c in mp.GetRootFolder().GetClipList() or []
if str(Path(c.GetClipProperty("File Path") or "").resolve()) == str(take)), None)
if mpi is None:
sys.exit(f"Resolve did not import {take}.")
tl = mp.CreateEmptyTimeline(take.stem) or proj.GetCurrentTimeline()
if tl and not tl.GetItemListInTrack("video", 1):
mp.AppendToTimeline([{"mediaPoolItem": mpi}])
proj.SetCurrentTimeline(tl)
if not mpi.TranscribeAudio(a.speaker_detection):
sys.exit("Resolve refused to transcribe. Studio edition and its speech "
"model are required; check Extras Download Manager.")
tr = mpi.GetTranscription()
words = words_from(tr, fps)
if not words:
sys.exit("Transcription came back empty — no speech found in the take.")
out = Path(a.out); out.parent.mkdir(parents=True, exist_ok=True)
out.write_text(json.dumps(words, indent=1))
summary = {"take": str(take), "words": len(words), "out": str(out),
"language": tr.get("language"), "fps": fps,
"speakers": sorted({s.get("speaker") for s in tr.get("segments", [])
if s.get("speaker")})}
if a.cuts:
duration = probe_duration(take)
sil = silences_from(tr, fps, a.min_silence)
keeps = keep_ranges(sil, duration, a.margin)
cuts = Path(a.cuts); cuts.parent.mkdir(parents=True, exist_ok=True)
cuts.write_text(json.dumps({
"schema_version": CUTS_SCHEMA, "source": str(take),
"source_duration": round(duration, 3), "threshold": None,
"margin": a.margin, "engine": "resolve-transcription",
"keep": [[round(x, 3), round(y, 3)] for x, y in keeps]}, indent=2))
summary.update({"silences": len(sil),
"removed_seconds": round(sum(b - x for x, b in sil), 2),
"cuts": str(cuts)})
if a.srt:
if tl.CreateSubtitlesFromAudio({"language": resolve.AUTO_CAPTION_AUTO}):
start = tl.GetStartFrame()
cues = [((s.GetStart() - start) / fps, (s.GetEnd() - start) / fps, s.GetName() or "")
for s in tl.GetItemListInTrack("subtitle", 1) or []]
srt = Path(a.srt); srt.parent.mkdir(parents=True, exist_ok=True)
srt.write_text(srt_from(cues))
summary.update({"captions": len(cues), "srt": str(srt)})
else:
print("WARNING: Resolve did not create captions; words JSON still "
"feeds gen_captions.py.", file=sys.stderr)
if not a.keep_project:
pm.CloseProject(proj); pm.DeleteProject(a.project)
else:
summary["project"] = a.project
print(json.dumps(summary, indent=2))
if a.cuts:
print(f"\nApply with: tighten_vo.py {a.take} --out work/extracts/tight.mp4 "
f"--apply-cuts {a.cuts}; then verify_cut.py.", file=sys.stderr)
if __name__ == "__main__":
main().tessl-plugin
skills
reel-builder
assets
references
scripts
yap-writer