CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/speaker-toolkit

Seven-skill presentation system: ingest talks into a rhetoric vault, run interactive clarification, generate a speaker profile, create presentations that match your documented patterns, produce the deck illustrations + thumbnail visual layer, create and publish talk-content Agent Skills with talk pages to a Jekyll shownotes site, and verify a recorded screencast against its storyboard. Includes a 113-entry Presentation Patterns taxonomy (83 observable: 64 patterns + 19 antipatterns; 30 unobservable: 21 patterns + 9 antipatterns) for scoring, brainstorming, and go-live preparation.

75

Quality

94%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide

SecuritybySnyk

Passed

No findings from the security scan

Overview
Quality
Evals
Security
Files

local_media_words.pyskills/vault-ingress/scripts/

"""Closed sampled-word receipts owned by the ingress media-acquisition lane.

This pure boundary validates and normalizes provider data; it does not acquire
media or authenticate a caller-supplied digest. Times are sample-relative, never
interpolated from segment timestamps. Punctuation-only tokens and degenerate
zero-or-negative-span tokens are omitted, each with an explicit index/reason
record. No timestamp is ever repaired; a retained word always carries a positive
span, and a sample whose degenerate share exceeds the bound below refuses whole.
"""

from __future__ import annotations

from collections.abc import Mapping
import copy
import math
from numbers import Real
import re
from typing import Any, NoReturn

from local_media_contract import LocalMediaError


WORDS_PIPELINE_VERSION = "sampled-words-v3"
WORDS_MAX_SAMPLE_SECONDS = 1200
WORDS_MAX_SOURCE_SECONDS = 14400
WORDS_MAX_COUNT = 50000
WORDS_MAX_SEGMENTS = 5000
WORDS_MAX_TOKEN_BYTES = 1024

# Share of lexical tokens permitted to carry a zero-or-negative span before the
# whole sample refuses. Whisper emits zero-duration words routinely on short
# tokens, so a handful is ordinary provider output, not corruption; a cluster is
# a different animal, because a misaligned transcript degrades many spans at once
# and its other timestamps are untrustworthy too. Measured against the bound in
# the CHANGELOG entry that introduced it.
WORDS_MAX_NONPOSITIVE_SHARE = 0.01

# Reasons a provider token may be recorded as excluded rather than retained.
TOKEN_EXCLUSION_REASONS = frozenset({"punctuation_only", "nonpositive_span"})
WORD_VALIDATION_FAILURES = frozenset(
    {
        "whisper_word_sample_invalid",
        "whisper_word_sample_invalid_token",
        "whisper_word_sample_invalid_word_span",
        "whisper_word_sample_invalid_word_overlap",
        "whisper_word_sample_invalid_word_nonpositive_span",
        "whisper_word_sample_invalid_word_probability",
        "whisper_word_sample_invalid_segment_span",
        "whisper_word_sample_invalid_word_segment",
    }
)
_SHA = re.compile(r"[0-9a-f]{64}\Z")
_REVISION = re.compile(r"[0-9a-f]{40}\Z")
_MODEL = re.compile(r"[A-Za-z0-9._-]+/[A-Za-z0-9._-]+\Z")
_VERSION = re.compile(r"[0-9]+\.[0-9]+\.[0-9]+(?:[-+][A-Za-z0-9.-]+)?\Z")
_LANGUAGE = re.compile(r"[a-z]{2,3}(?:-[a-z0-9]{2,8})*\Z")

# Renew the model pin quarterly in a dedicated dependency/capability change.
# Verify native word alignment and language detection before accepting a new
# revision. The Python packages retain their own manifest/Dependabot pins.
DEFAULT_WORD_MODEL = {
    "id": "mlx-community/whisper-large-v3-turbo",
    "revision": "a4aaeec0636e6fef84abdcbe3544cb2bf7e9f6fb",
}


def _refuse(detail: str = "") -> NoReturn:
    code = "whisper_word_sample_invalid" + ("_" + detail if detail else "")
    if code not in WORD_VALIDATION_FAILURES:
        raise ValueError("unsupported word-validation diagnostic")
    raise LocalMediaError(code)


def _token_bytes(text: str) -> int:
    try:
        return len(text.encode("utf-8"))
    except UnicodeEncodeError:
        _refuse()


def _object(value: Any, fields: set[str]) -> dict:
    if not isinstance(value, dict) or set(value) != fields:
        _refuse()
    return value


def _number(value: Any, low: float, high: float) -> float:
    if type(value) not in (int, float) or not low <= value <= high:
        _refuse()
    result = float(value)
    if not math.isfinite(result):
        _refuse()
    return result


def _degenerate_span(begin: Any, end: Any, bound: float | None) -> bool:
    """True only for a span the receipt would otherwise accept as in-bounds.

    A timestamp that is non-finite, negative, or past the sample is malformed
    rather than degenerate. Returning False leaves it in ``words``, where
    ``validate_word_sample`` refuses it with its own diagnostic.
    """
    if bound is None:
        return False
    try:
        low = _number(begin, 0, bound)
        high = _number(end, 0, bound)
    except LocalMediaError:
        return False
    return high <= low


def validate_word_diagnostic(value: Any) -> dict:
    """Closed numeric-only failure evidence, never source locators or words."""
    diagnostic = _object(
        value,
        {
            "schema_version",
            "word_index",
            "word_count",
            "word_start_seconds",
            "word_end_seconds",
            "segment_index",
            "segment_count",
            "segment_start_seconds",
            "segment_end_seconds",
        },
    )
    if (
        type(diagnostic["schema_version"]) is not int
        or diagnostic["schema_version"] != 1
    ):
        _refuse()
    for field, maximum in (
        ("word_count", WORDS_MAX_COUNT),
        ("segment_count", WORDS_MAX_SEGMENTS),
    ):
        if type(diagnostic[field]) is not int or not 1 <= diagnostic[field] <= maximum:
            _refuse()
    index = diagnostic["word_index"]
    if type(index) is not int or not 0 <= index < diagnostic["word_count"]:
        _refuse()
    begin = _number(diagnostic["word_start_seconds"], 0, WORDS_MAX_SAMPLE_SECONDS)
    end = _number(diagnostic["word_end_seconds"], begin, WORDS_MAX_SAMPLE_SECONDS)
    if end <= begin:
        _refuse()
    index = diagnostic["segment_index"]
    if index is None:
        if (
            diagnostic["segment_start_seconds"] is not None
            or diagnostic["segment_end_seconds"] is not None
        ):
            _refuse()
    else:
        if type(index) is not int or not 0 <= index < diagnostic["segment_count"]:
            _refuse()
        start = _number(
            diagnostic["segment_start_seconds"], 0, WORDS_MAX_SAMPLE_SECONDS
        )
        finish = _number(
            diagnostic["segment_end_seconds"], start, WORDS_MAX_SAMPLE_SECONDS
        )
        if finish <= start:
            _refuse()
    return diagnostic


class WordSampleError(LocalMediaError):
    """A segment-membership refusal with bounded numeric worker diagnostics."""

    def __init__(self, diagnostic: Any):
        self.word_timing = copy.deepcopy(validate_word_diagnostic(diagnostic))
        super().__init__("whisper_word_sample_invalid_word_segment")


def _provider_number(value: Any) -> Any:
    # Native alignment returns NumPy real scalars. Project those losslessly to
    # Python floats at this boundary; persisted receipts still require plain
    # JSON numbers. This does not alter, interpolate or repair a timestamp.
    if type(value) in (int, float) or not isinstance(value, Real):
        return value
    if isinstance(value, bool):
        return value
    try:
        converted = float(value)
    except (TypeError, ValueError, OverflowError):
        _refuse()
    if converted != value:
        _refuse()
    return converted


def validate_word_model(value: Any) -> dict:
    model = _object(value, {"id", "revision"})
    if (
        not isinstance(model["id"], str)
        or len(model["id"]) > 120
        or _MODEL.fullmatch(model["id"]) is None
        or not isinstance(model["revision"], str)
        or _REVISION.fullmatch(model["revision"]) is None
    ):
        _refuse()
    return model


def validate_word_sample(value: Any) -> dict:
    sample = _object(
        value,
        {
            "schema_version",
            "pipeline_version",
            "source_sha256",
            "sample_sha256",
            "source_duration_seconds",
            "sample_start_seconds",
            "sample_duration_seconds",
            "provider",
            "provider_version",
            "model",
            "language",
            "language_probability",
            "language_probe_seconds",
            "words",
            "segments",
            "token_exclusions",
        },
    )
    if (
        type(sample["schema_version"]) is not int
        or sample["schema_version"] != 3
        or sample["pipeline_version"] != WORDS_PIPELINE_VERSION
        or sample["provider"] != "mlx-whisper"
        or not isinstance(sample["provider_version"], str)
        or _VERSION.fullmatch(sample["provider_version"]) is None
    ):
        _refuse()
    for key in ("source_sha256", "sample_sha256"):
        if not isinstance(sample[key], str) or _SHA.fullmatch(sample[key]) is None:
            _refuse()
    validate_word_model(sample["model"])
    duration = _number(sample["sample_duration_seconds"], 0, WORDS_MAX_SAMPLE_SECONDS)
    source_duration = _number(
        sample["source_duration_seconds"], 0, WORDS_MAX_SOURCE_SECONDS
    )
    start = _number(sample["sample_start_seconds"], 0, source_duration)
    if duration <= 0 or start + duration > source_duration:
        _refuse()
    if (
        not isinstance(sample["language"], str)
        or _LANGUAGE.fullmatch(sample["language"]) is None
    ):
        _refuse()
    _number(sample["language_probability"], 0, 1)
    probe_seconds = _number(sample["language_probe_seconds"], 0, min(duration, 30))
    if probe_seconds <= 0:
        _refuse()
    words = sample["words"]
    if not isinstance(words, list) or not 1 <= len(words) <= WORDS_MAX_COUNT:
        _refuse()
    previous_end = 0.0
    for word in words:
        word = _object(
            word,
            {"text", "start_seconds", "end_seconds", "probability", "segment_index"},
        )
        text = word["text"]
        if (
            not isinstance(text, str)
            or not text
            or _token_bytes(text) > WORDS_MAX_TOKEN_BYTES
            or any(ord(char) < 32 for char in text)
            or any(char.isspace() for char in text)
            or not any(char.isalnum() for char in text)
        ):
            _refuse("token")
        try:
            begin = _number(word["start_seconds"], 0, duration)
            end = _number(word["end_seconds"], 0, duration)
        except LocalMediaError:
            _refuse("word_span")
        if begin < previous_end:
            _refuse("word_overlap")
        if end <= begin:
            _refuse("word_nonpositive_span")
        try:
            _number(word["probability"], 0, 1)
        except LocalMediaError:
            _refuse("word_probability")
        previous_end = end
    segments = sample["segments"]
    if not isinstance(segments, list) or not 1 <= len(segments) <= WORDS_MAX_SEGMENTS:
        _refuse()
    previous_end = 0.0
    for segment in segments:
        segment = _object(
            segment,
            {
                "start_seconds",
                "end_seconds",
                "compression_ratio",
                "average_log_probability",
                "no_speech_probability",
            },
        )
        try:
            begin = _number(segment["start_seconds"], previous_end, duration)
            end = _number(segment["end_seconds"], begin, duration)
        except LocalMediaError:
            _refuse("segment_span")
        if end <= begin:
            _refuse("segment_span")
        _number(segment["compression_ratio"], 0, 10000)
        _number(segment["average_log_probability"], -10000, 0)
        _number(segment["no_speech_probability"], 0, 1)
        previous_end = end
    # Preserve the provider's explicit membership, not a guessed containing box.
    # MLX median-duration adjustments can place boundary words across the retained
    # segment timestamp. Positive overlap binds quality metadata without changing
    # either timestamp. Word/segment ordering and sample bounds still apply.
    previous_index = 0
    for word_index, word in enumerate(words):
        index = word["segment_index"]
        if type(index) is not int or not previous_index <= index < len(segments):
            _refuse("word_segment")
        previous_index = index
        segment = segments[index]
        if not (
            segment["start_seconds"] < word["end_seconds"]
            and word["start_seconds"] < segment["end_seconds"]
        ):
            raise WordSampleError(
                {
                    "schema_version": 1,
                    "word_index": word_index,
                    "word_count": len(words),
                    "word_start_seconds": word["start_seconds"],
                    "word_end_seconds": word["end_seconds"],
                    "segment_index": index,
                    "segment_count": len(segments),
                    "segment_start_seconds": segment["start_seconds"],
                    "segment_end_seconds": segment["end_seconds"],
                }
            )
    exclusions = sample["token_exclusions"]
    if not isinstance(exclusions, list) or len(exclusions) > WORDS_MAX_COUNT:
        _refuse()
    previous_index = -1
    for item in exclusions:
        item = _object(item, {"token_index", "reason"})
        if (
            type(item["token_index"]) is not int
            or not previous_index < item["token_index"] < WORDS_MAX_COUNT
            or not isinstance(item["reason"], str)
            or item["reason"] not in TOKEN_EXCLUSION_REASONS
        ):
            _refuse()
        previous_index = item["token_index"]
    degenerate = sum(1 for item in exclusions if item["reason"] == "nonpositive_span")
    # Same bound, same denominator as normalize_word_result: a receipt that
    # excluded more than the admission share is not a valid receipt, whoever
    # wrote it.
    if degenerate and degenerate / (len(words) + degenerate) > (
        WORDS_MAX_NONPOSITIVE_SHARE
    ):
        _refuse("word_nonpositive_span")
    return sample


def normalize_word_result(
    value: Any,
    *,
    source_sha256: str,
    sample_sha256: str,
    source_duration_seconds: float,
    sample_start_seconds: float,
    sample_duration_seconds: float,
    provider_version: str,
    model: dict,
    language_probability: float,
) -> dict:
    """Project bounded provider output; never make up word times or confidence."""
    if not isinstance(value, Mapping) or not isinstance(value.get("segments"), list):
        _refuse()
    if not 1 <= len(value["segments"]) <= WORDS_MAX_SEGMENTS:
        _refuse()
    words, segments, exclusions = [], [], []
    token_index = 0
    lexical = degenerate = 0
    try:
        bound = _number(sample_duration_seconds, 0, WORDS_MAX_SAMPLE_SECONDS)
    except LocalMediaError:
        bound = None
    if bound is not None and bound <= 0:
        bound = None
    for segment_index, segment in enumerate(value["segments"]):
        if not isinstance(segment, Mapping) or not isinstance(
            segment.get("words"), list
        ):
            _refuse()
        segments.append(
            {
                "start_seconds": _provider_number(segment.get("start")),
                "end_seconds": _provider_number(segment.get("end")),
                "compression_ratio": _provider_number(segment.get("compression_ratio")),
                "average_log_probability": _provider_number(segment.get("avg_logprob")),
                "no_speech_probability": _provider_number(
                    segment.get("no_speech_prob")
                ),
            }
        )
        if token_index + len(segment["words"]) > WORDS_MAX_COUNT:
            _refuse()
        for word in segment["words"]:
            if not isinstance(word, Mapping) or not isinstance(word.get("word"), str):
                _refuse()
            text = word["word"].strip()
            if (
                not text
                or _token_bytes(text) > WORDS_MAX_TOKEN_BYTES
                or any(ord(char) < 32 for char in text)
            ):
                _refuse()
            if not any(char.isalnum() for char in text):
                exclusions.append(
                    {"token_index": token_index, "reason": "punctuation_only"}
                )
            else:
                lexical += 1
                retained = {
                    "text": text,
                    "start_seconds": _provider_number(word.get("start")),
                    "end_seconds": _provider_number(word.get("end")),
                    "probability": _provider_number(word.get("probability")),
                    "segment_index": segment_index,
                }
                # A zero-or-negative span carries no duration, so omitting the
                # token leaves the elapsed-time denominator untouched and moves
                # the word count by one. Recorded, never repaired.
                #
                # Only a pair the receipt would otherwise accept is judged here.
                # A malformed timestamp is retained and refused by
                # validate_word_sample below, keeping its own diagnostic.
                if _degenerate_span(
                    retained["start_seconds"], retained["end_seconds"], bound
                ):
                    degenerate += 1
                    exclusions.append(
                        {"token_index": token_index, "reason": "nonpositive_span"}
                    )
                else:
                    words.append(retained)
            token_index += 1
    if degenerate and (
        lexical == 0 or degenerate / lexical > WORDS_MAX_NONPOSITIVE_SHARE
    ):
        _refuse("word_nonpositive_span")
    return validate_word_sample(
        {
            "schema_version": 3,
            "pipeline_version": WORDS_PIPELINE_VERSION,
            "source_sha256": source_sha256,
            "sample_sha256": sample_sha256,
            "source_duration_seconds": source_duration_seconds,
            "sample_start_seconds": sample_start_seconds,
            "sample_duration_seconds": sample_duration_seconds,
            "provider": "mlx-whisper",
            "provider_version": provider_version,
            "model": copy.deepcopy(model),
            "language": value.get("language"),
            "language_probability": language_probability,
            # Comparing against an unvalidated duration would raise before the
            # reader can refuse it; pass the caller's value through instead.
            "language_probe_seconds": (
                min(30.0, bound) if bound is not None else sample_duration_seconds
            ),
            "words": words,
            "segments": segments,
            "token_exclusions": exclusions,
        }
    )

skills

vault-ingress

scripts

adherence_baseline.py

aggregate-catalog-feedback.py

apply-source-repairs.py

artifact_locator.py

artifact_metadata.py

artifact_supervisor.py

audit-pattern-catalog.py

audit-persisted-pattern-observations.py

audit-source-identities.py

batch-download-videos.py

build-contact-sheet.py

build-crop-reviewer.py

build-score-basis.py

catalog_dimension_registry.py

catalog_io.py

catalog_normalization.py

check-runtime.py

classify-pptx-evidence.py

cloud_artifacts.py

cooperative_lock.py

crop_frames.py

crop-reviewer-shell.html

crop-reviewer-shell.html.txt

crop-reviewer.js

crop-reviewer.js.txt

establish-date-provenance.py

failure_diagnostics.py

fetch-transcript.py

ingress_contract.py

local_media_contract.py

local_media_download.py

local_media_evidence.py

local_media_process.py

local_media_sampling.py

local_media_transcription.py

local_media_words.py

markdown_deck.py

migrate-tracking-database.py

mutate-tracking-database.py

pattern_evidence.py

pdf_evidence.py

persist-results.py

persisted_pattern_observations.py

pptx_catalog_selection.py

pptx_deck_facts.py

pptx_discovery_contract.py

pptx_evidence.py

pptx_talk_identity.py

pptx-extraction.py

preflight-vault.py

queue_claim_contract.py

queue-state.py

read-tracking-database.py

render-markdown-deck.py

render-vault-status.py

retained_stage.py

return_validation.py

run-obligations.py

scan-shownotes.py

source_alias_contract.py

source_identity_matching.py

summary_lock.py

sweep-pptx-talk-identity.py

tracking_database_io.py

tracking_database.py

transcript_quality.py

transcript_timing.py

validate-returns.py

vault_root_authority.py

video_evidence.py

video_integrity.py

video-slide-extraction.py

vtt-cleanup.py

write-analysis.py

ytdlp_runtime.py

SKILL.md

README.md

tile.json