CtrlK
BlogDocsLog inGet started
Tessl Logo

gamussa/reels-producer-skill

Write talking-head scripts and produce Instagram reels and YouTube shorts

93

1.23x
Quality

94%

Does it follow best practices?

Impact

88%

1.23x

Average score across 3 eval scenarios

SecuritybySnyk

Low

Low-risk findings worth noting

Overview
Quality
Evals
Security
Files

resolve_transcribe.pyskills/reel-builder/scripts/

#!/usr/bin/env python3
"""Transcribe a take with DaVinci Resolve and hand the result to the pipeline.

Resolve's transcription returns frame-precise words AND marks every silence
as an explicit `(...)` span — measured on a synthetic take: a 4.00s silent
head and two 1.73s pauses, each reported with start and end. That is the
speech-aware silence detection tighten_vo.py's level threshold approximates.

Nothing is rendered here. Three files come out, all in shapes the pipeline
already reads, so the rest of the talking-head path is unchanged:

  --out    words JSON, flat [{word, start, end}] in seconds — what
           gen_captions.py, plan_broll.py, trim_words.py and check_yap.py take
  --cuts   keep-range cut list in tighten_vo.py's schema. Replay it with
           `tighten_vo.py TAKE --apply-cuts CUTS` and ffmpeg does the cutting;
           verify_cut.py then proves the words survived. Resolve decides, ffmpeg
           cuts — one renderer, no second edit.
  --srt    Resolve's auto-captions as SRT, for render_reel.py --srt

Voice isolation and the dialogue leveler are Resolve properties on the AUDIO
track item — the video item refuses them (measured) — and belong to a Resolve
render, not here. --keep-project leaves the timeline for that.

Requires Resolve STUDIO running. Transcription of a 24s take took over a
minute the first time (model load), so this never runs under a short timeout.

Usage:
  resolve_transcribe.py raw/take.mp4 --out work/take.words.json
  resolve_transcribe.py raw/take.mp4 --out work/take.words.json \\
      --cuts work/cuts.json --srt work/take.srt --min-silence 0.5
"""
import argparse, json, subprocess, sys
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parent))
from resolve_bridge import connect, project_fps, DEFAULT_FPS   # noqa: E402
from tighten_vo import keep_ranges, CUTS_SCHEMA               # noqa: E402

SILENCE = "(...)"
DEFAULT_MARGIN = 0.12      # seconds kept on each side of a cut; 0 sounds clipped
DEFAULT_MIN_SILENCE = 0.5  # shorter pauses are speech rhythm, not dead air


def tc_to_seconds(tc, fps):
    """'HH:MM:SS:FF' at fps to seconds. Frames are exact, so no rounding."""
    h, m, s, f = (int(x) for x in tc.split(":"))
    return h * 3600 + m * 60 + s + f / fps


def words_from(transcription, fps):
    """Flat word list in seconds, silences dropped, matching whisper's shape."""
    out = []
    for seg in (transcription or {}).get("segments", []):
        for w in seg.get("words", []):
            text = (w.get("text") or "").strip()
            if not text or text == SILENCE:
                continue
            out.append({"word": text, "start": round(tc_to_seconds(w["start"], fps), 3),
                        "end": round(tc_to_seconds(w["end"], fps), 3)})
    return out


def silences_from(transcription, fps, min_s=DEFAULT_MIN_SILENCE):
    """(start, end) spans Resolve marked as silence, at least min_s long."""
    spans = []
    for seg in (transcription or {}).get("segments", []):
        for w in seg.get("words", []):
            if (w.get("text") or "").strip() == SILENCE:
                a, b = tc_to_seconds(w["start"], fps), tc_to_seconds(w["end"], fps)
                if b - a >= min_s:
                    spans.append((round(a, 3), round(b, 3)))
    return spans


def srt_from(cues):
    """SRT text from (start_s, end_s, text) cues, in order."""
    def stamp(t):
        ms = int(round(t * 1000))
        h, ms = divmod(ms, 3600000); m, ms = divmod(ms, 60000); s, ms = divmod(ms, 1000)
        return f"{h:02}:{m:02}:{s:02},{ms:03}"
    lines = []
    for i, (a, b, text) in enumerate(cues, 1):
        lines += [str(i), f"{stamp(a)} --> {stamp(b)}", text.strip(), ""]
    return "\n".join(lines)


def probe_duration(path):
    r = subprocess.run(["ffprobe", "-v", "error", "-show_entries", "format=duration",
                        "-of", "csv=p=0", str(path)], capture_output=True, text=True)
    return float(r.stdout.strip() or 0)


def main():
    ap = argparse.ArgumentParser(
        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
    ap.add_argument("take")
    ap.add_argument("--out", required=True, help="words JSON (flat list, seconds)")
    ap.add_argument("--cuts", help="also write a tighten_vo.py cut list here")
    ap.add_argument("--srt", help="also write Resolve's auto-captions as SRT here")
    ap.add_argument("--min-silence", type=float, default=DEFAULT_MIN_SILENCE)
    ap.add_argument("--margin", type=float, default=DEFAULT_MARGIN,
                    help=f"seconds kept either side of a cut (default {DEFAULT_MARGIN})")
    ap.add_argument("--speaker-detection", action="store_true")
    ap.add_argument("--project", default="reel-builder_transcribe")
    ap.add_argument("--keep-project", action="store_true")
    a = ap.parse_args()

    take = Path(a.take).resolve()
    if not take.exists():
        sys.exit(f"Take not found: {a.take}")

    resolve = connect()
    pm = resolve.GetProjectManager()
    if pm.GetCurrentProject():
        pm.SaveProject()
    proj = pm.LoadProject(a.project) or pm.CreateProject(a.project)
    if proj is None:
        sys.exit(f"Could not open Resolve project {a.project!r}. If Resolve is "
                 f"showing a dialog, close it and retry.")
    proj.SetSetting("timelineFrameRate", str(DEFAULT_FPS))
    fps = project_fps(proj)
    mp = proj.GetMediaPool()
    items = resolve.GetMediaStorage().AddItemListToMediaPool([str(take)]) or []
    mpi = items[0] if items else next(
        (c for c in mp.GetRootFolder().GetClipList() or []
         if str(Path(c.GetClipProperty("File Path") or "").resolve()) == str(take)), None)
    if mpi is None:
        sys.exit(f"Resolve did not import {take}.")
    tl = mp.CreateEmptyTimeline(take.stem) or proj.GetCurrentTimeline()
    if tl and not tl.GetItemListInTrack("video", 1):
        mp.AppendToTimeline([{"mediaPoolItem": mpi}])
    proj.SetCurrentTimeline(tl)

    if not mpi.TranscribeAudio(a.speaker_detection):
        sys.exit("Resolve refused to transcribe. Studio edition and its speech "
                 "model are required; check Extras Download Manager.")
    tr = mpi.GetTranscription()
    words = words_from(tr, fps)
    if not words:
        sys.exit("Transcription came back empty — no speech found in the take.")

    out = Path(a.out); out.parent.mkdir(parents=True, exist_ok=True)
    out.write_text(json.dumps(words, indent=1))
    summary = {"take": str(take), "words": len(words), "out": str(out),
               "language": tr.get("language"), "fps": fps,
               "speakers": sorted({s.get("speaker") for s in tr.get("segments", [])
                                   if s.get("speaker")})}

    if a.cuts:
        duration = probe_duration(take)
        sil = silences_from(tr, fps, a.min_silence)
        keeps = keep_ranges(sil, duration, a.margin)
        cuts = Path(a.cuts); cuts.parent.mkdir(parents=True, exist_ok=True)
        cuts.write_text(json.dumps({
            "schema_version": CUTS_SCHEMA, "source": str(take),
            "source_duration": round(duration, 3), "threshold": None,
            "margin": a.margin, "engine": "resolve-transcription",
            "keep": [[round(x, 3), round(y, 3)] for x, y in keeps]}, indent=2))
        summary.update({"silences": len(sil),
                        "removed_seconds": round(sum(b - x for x, b in sil), 2),
                        "cuts": str(cuts)})

    if a.srt:
        if tl.CreateSubtitlesFromAudio({"language": resolve.AUTO_CAPTION_AUTO}):
            start = tl.GetStartFrame()
            cues = [((s.GetStart() - start) / fps, (s.GetEnd() - start) / fps, s.GetName() or "")
                    for s in tl.GetItemListInTrack("subtitle", 1) or []]
            srt = Path(a.srt); srt.parent.mkdir(parents=True, exist_ok=True)
            srt.write_text(srt_from(cues))
            summary.update({"captions": len(cues), "srt": str(srt)})
        else:
            print("WARNING: Resolve did not create captions; words JSON still "
                  "feeds gen_captions.py.", file=sys.stderr)

    if not a.keep_project:
        pm.CloseProject(proj); pm.DeleteProject(a.project)
    else:
        summary["project"] = a.project
    print(json.dumps(summary, indent=2))
    if a.cuts:
        print(f"\nApply with: tighten_vo.py {a.take} --out work/extracts/tight.mp4 "
              f"--apply-cuts {a.cuts}; then verify_cut.py.", file=sys.stderr)


if __name__ == "__main__":
    main()

.mcp.json

tile.json