CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/speaker-toolkit

Seven-skill presentation system: ingest talks into a rhetoric vault, run interactive clarification, generate a speaker profile, create presentations that match your documented patterns, produce the deck illustrations + thumbnail visual layer, publish talk pages to a Jekyll shownotes site, and verify a recorded screencast against its storyboard. Includes a 113-entry Presentation Patterns taxonomy (83 observable: 64 patterns + 19 antipatterns; 30 unobservable: 21 patterns + 9 antipatterns) for scoring, brainstorming, and go-live preparation.

74

Quality

93%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide

SecuritybySnyk

Low

Low-risk findings worth noting

Overview
Quality
Evals
Security
Files

local_media_words.pyskills/vault-ingress/scripts/

"""Closed sampled-word receipts owned by the ingress media-acquisition lane.

This pure boundary validates and normalizes provider data; it does not acquire
media or authenticate a caller-supplied digest. Times are sample-relative, never
interpolated from segment timestamps. Punctuation-only tokens and degenerate
zero-or-negative-span tokens are omitted, each with an explicit index/reason
record. No timestamp is ever repaired; a retained word always carries a positive
span, and a sample whose degenerate share exceeds the bound below refuses whole.
"""

from __future__ import annotations

from collections.abc import Mapping
import copy
import math
from numbers import Real
import re
from typing import Any, NoReturn

from local_media_contract import LocalMediaError


WORDS_PIPELINE_VERSION = "sampled-words-v3"
WORDS_MAX_SAMPLE_SECONDS = 1200
WORDS_MAX_SOURCE_SECONDS = 14400
WORDS_MAX_COUNT = 50000
WORDS_MAX_SEGMENTS = 5000
WORDS_MAX_TOKEN_BYTES = 1024

# Share of lexical tokens permitted to carry a zero-or-negative span before the
# whole sample refuses. Whisper emits zero-duration words routinely on short
# tokens, so a handful is ordinary provider output, not corruption; a cluster is
# a different animal, because a misaligned transcript degrades many spans at once
# and its other timestamps are untrustworthy too. Measured against the bound in
# the CHANGELOG entry that introduced it.
WORDS_MAX_NONPOSITIVE_SHARE = 0.01

# Reasons a provider token may be recorded as excluded rather than retained.
TOKEN_EXCLUSION_REASONS = frozenset({"punctuation_only", "nonpositive_span"})
WORD_VALIDATION_FAILURES = frozenset(
    {
        "whisper_word_sample_invalid",
        "whisper_word_sample_invalid_token",
        "whisper_word_sample_invalid_word_span",
        "whisper_word_sample_invalid_word_overlap",
        "whisper_word_sample_invalid_word_nonpositive_span",
        "whisper_word_sample_invalid_word_probability",
        "whisper_word_sample_invalid_segment_span",
        "whisper_word_sample_invalid_word_segment",
    }
)
_SHA = re.compile(r"[0-9a-f]{64}\Z")
_REVISION = re.compile(r"[0-9a-f]{40}\Z")
_MODEL = re.compile(r"[A-Za-z0-9._-]+/[A-Za-z0-9._-]+\Z")
_VERSION = re.compile(r"[0-9]+\.[0-9]+\.[0-9]+(?:[-+][A-Za-z0-9.-]+)?\Z")
_LANGUAGE = re.compile(r"[a-z]{2,3}(?:-[a-z0-9]{2,8})*\Z")

# Renew the model pin quarterly in a dedicated dependency/capability change.
# Verify native word alignment and language detection before accepting a new
# revision. The Python packages retain their own manifest/Dependabot pins.
DEFAULT_WORD_MODEL = {
    "id": "mlx-community/whisper-large-v3-turbo",
    "revision": "a4aaeec0636e6fef84abdcbe3544cb2bf7e9f6fb",
}


def _refuse(detail: str = "") -> NoReturn:
    code = "whisper_word_sample_invalid" + ("_" + detail if detail else "")
    if code not in WORD_VALIDATION_FAILURES:
        raise ValueError("unsupported word-validation diagnostic")
    raise LocalMediaError(code)


def _token_bytes(text: str) -> int:
    try:
        return len(text.encode("utf-8"))
    except UnicodeEncodeError:
        _refuse()


def _object(value: Any, fields: set[str]) -> dict:
    if not isinstance(value, dict) or set(value) != fields:
        _refuse()
    return value


def _number(value: Any, low: float, high: float) -> float:
    if type(value) not in (int, float) or not low <= value <= high:
        _refuse()
    result = float(value)
    if not math.isfinite(result):
        _refuse()
    return result


def _degenerate_span(begin: Any, end: Any, bound: float | None) -> bool:
    """True only for a span the receipt would otherwise accept as in-bounds.

    A timestamp that is non-finite, negative, or past the sample is malformed
    rather than degenerate. Returning False leaves it in ``words``, where
    ``validate_word_sample`` refuses it with its own diagnostic.
    """
    if bound is None:
        return False
    try:
        low = _number(begin, 0, bound)
        high = _number(end, 0, bound)
    except LocalMediaError:
        return False
    return high <= low


def validate_word_diagnostic(value: Any) -> dict:
    """Closed numeric-only failure evidence, never source locators or words."""
    diagnostic = _object(
        value,
        {
            "schema_version",
            "word_index",
            "word_count",
            "word_start_seconds",
            "word_end_seconds",
            "segment_index",
            "segment_count",
            "segment_start_seconds",
            "segment_end_seconds",
        },
    )
    if (
        type(diagnostic["schema_version"]) is not int
        or diagnostic["schema_version"] != 1
    ):
        _refuse()
    for field, maximum in (
        ("word_count", WORDS_MAX_COUNT),
        ("segment_count", WORDS_MAX_SEGMENTS),
    ):
        if type(diagnostic[field]) is not int or not 1 <= diagnostic[field] <= maximum:
            _refuse()
    index = diagnostic["word_index"]
    if type(index) is not int or not 0 <= index < diagnostic["word_count"]:
        _refuse()
    begin = _number(diagnostic["word_start_seconds"], 0, WORDS_MAX_SAMPLE_SECONDS)
    end = _number(diagnostic["word_end_seconds"], begin, WORDS_MAX_SAMPLE_SECONDS)
    if end <= begin:
        _refuse()
    index = diagnostic["segment_index"]
    if index is None:
        if (
            diagnostic["segment_start_seconds"] is not None
            or diagnostic["segment_end_seconds"] is not None
        ):
            _refuse()
    else:
        if type(index) is not int or not 0 <= index < diagnostic["segment_count"]:
            _refuse()
        start = _number(
            diagnostic["segment_start_seconds"], 0, WORDS_MAX_SAMPLE_SECONDS
        )
        finish = _number(
            diagnostic["segment_end_seconds"], start, WORDS_MAX_SAMPLE_SECONDS
        )
        if finish <= start:
            _refuse()
    return diagnostic


class WordSampleError(LocalMediaError):
    """A segment-membership refusal with bounded numeric worker diagnostics."""

    def __init__(self, diagnostic: Any):
        self.word_timing = copy.deepcopy(validate_word_diagnostic(diagnostic))
        super().__init__("whisper_word_sample_invalid_word_segment")


def _provider_number(value: Any) -> Any:
    # Native alignment returns NumPy real scalars. Project those losslessly to
    # Python floats at this boundary; persisted receipts still require plain
    # JSON numbers. This does not alter, interpolate or repair a timestamp.
    if type(value) in (int, float) or not isinstance(value, Real):
        return value
    if isinstance(value, bool):
        return value
    try:
        converted = float(value)
    except (TypeError, ValueError, OverflowError):
        _refuse()
    if converted != value:
        _refuse()
    return converted


def validate_word_model(value: Any) -> dict:
    model = _object(value, {"id", "revision"})
    if (
        not isinstance(model["id"], str)
        or len(model["id"]) > 120
        or _MODEL.fullmatch(model["id"]) is None
        or not isinstance(model["revision"], str)
        or _REVISION.fullmatch(model["revision"]) is None
    ):
        _refuse()
    return model


def validate_word_sample(value: Any) -> dict:
    sample = _object(
        value,
        {
            "schema_version",
            "pipeline_version",
            "source_sha256",
            "sample_sha256",
            "source_duration_seconds",
            "sample_start_seconds",
            "sample_duration_seconds",
            "provider",
            "provider_version",
            "model",
            "language",
            "language_probability",
            "language_probe_seconds",
            "words",
            "segments",
            "token_exclusions",
        },
    )
    if (
        type(sample["schema_version"]) is not int
        or sample["schema_version"] != 3
        or sample["pipeline_version"] != WORDS_PIPELINE_VERSION
        or sample["provider"] != "mlx-whisper"
        or not isinstance(sample["provider_version"], str)
        or _VERSION.fullmatch(sample["provider_version"]) is None
    ):
        _refuse()
    for key in ("source_sha256", "sample_sha256"):
        if not isinstance(sample[key], str) or _SHA.fullmatch(sample[key]) is None:
            _refuse()
    validate_word_model(sample["model"])
    duration = _number(sample["sample_duration_seconds"], 0, WORDS_MAX_SAMPLE_SECONDS)
    source_duration = _number(
        sample["source_duration_seconds"], 0, WORDS_MAX_SOURCE_SECONDS
    )
    start = _number(sample["sample_start_seconds"], 0, source_duration)
    if duration <= 0 or start + duration > source_duration:
        _refuse()
    if (
        not isinstance(sample["language"], str)
        or _LANGUAGE.fullmatch(sample["language"]) is None
    ):
        _refuse()
    _number(sample["language_probability"], 0, 1)
    probe_seconds = _number(sample["language_probe_seconds"], 0, min(duration, 30))
    if probe_seconds <= 0:
        _refuse()
    words = sample["words"]
    if not isinstance(words, list) or not 1 <= len(words) <= WORDS_MAX_COUNT:
        _refuse()
    previous_end = 0.0
    for word in words:
        word = _object(
            word,
            {"text", "start_seconds", "end_seconds", "probability", "segment_index"},
        )
        text = word["text"]
        if (
            not isinstance(text, str)
            or not text
            or _token_bytes(text) > WORDS_MAX_TOKEN_BYTES
            or any(ord(char) < 32 for char in text)
            or any(char.isspace() for char in text)
            or not any(char.isalnum() for char in text)
        ):
            _refuse("token")
        try:
            begin = _number(word["start_seconds"], 0, duration)
            end = _number(word["end_seconds"], 0, duration)
        except LocalMediaError:
            _refuse("word_span")
        if begin < previous_end:
            _refuse("word_overlap")
        if end <= begin:
            _refuse("word_nonpositive_span")
        try:
            _number(word["probability"], 0, 1)
        except LocalMediaError:
            _refuse("word_probability")
        previous_end = end
    segments = sample["segments"]
    if not isinstance(segments, list) or not 1 <= len(segments) <= WORDS_MAX_SEGMENTS:
        _refuse()
    previous_end = 0.0
    for segment in segments:
        segment = _object(
            segment,
            {
                "start_seconds",
                "end_seconds",
                "compression_ratio",
                "average_log_probability",
                "no_speech_probability",
            },
        )
        try:
            begin = _number(segment["start_seconds"], previous_end, duration)
            end = _number(segment["end_seconds"], begin, duration)
        except LocalMediaError:
            _refuse("segment_span")
        if end <= begin:
            _refuse("segment_span")
        _number(segment["compression_ratio"], 0, 10000)
        _number(segment["average_log_probability"], -10000, 0)
        _number(segment["no_speech_probability"], 0, 1)
        previous_end = end
    # Preserve the provider's explicit membership, not a guessed containing box.
    # MLX median-duration adjustments can place boundary words across the retained
    # segment timestamp. Positive overlap binds quality metadata without changing
    # either timestamp. Word/segment ordering and sample bounds still apply.
    previous_index = 0
    for word_index, word in enumerate(words):
        index = word["segment_index"]
        if type(index) is not int or not previous_index <= index < len(segments):
            _refuse("word_segment")
        previous_index = index
        segment = segments[index]
        if not (
            segment["start_seconds"] < word["end_seconds"]
            and word["start_seconds"] < segment["end_seconds"]
        ):
            raise WordSampleError(
                {
                    "schema_version": 1,
                    "word_index": word_index,
                    "word_count": len(words),
                    "word_start_seconds": word["start_seconds"],
                    "word_end_seconds": word["end_seconds"],
                    "segment_index": index,
                    "segment_count": len(segments),
                    "segment_start_seconds": segment["start_seconds"],
                    "segment_end_seconds": segment["end_seconds"],
                }
            )
    exclusions = sample["token_exclusions"]
    if not isinstance(exclusions, list) or len(exclusions) > WORDS_MAX_COUNT:
        _refuse()
    previous_index = -1
    for item in exclusions:
        item = _object(item, {"token_index", "reason"})
        if (
            type(item["token_index"]) is not int
            or not previous_index < item["token_index"] < WORDS_MAX_COUNT
            or not isinstance(item["reason"], str)
            or item["reason"] not in TOKEN_EXCLUSION_REASONS
        ):
            _refuse()
        previous_index = item["token_index"]
    degenerate = sum(1 for item in exclusions if item["reason"] == "nonpositive_span")
    # Same bound, same denominator as normalize_word_result: a receipt that
    # excluded more than the admission share is not a valid receipt, whoever
    # wrote it.
    if degenerate and degenerate / (len(words) + degenerate) > (
        WORDS_MAX_NONPOSITIVE_SHARE
    ):
        _refuse("word_nonpositive_span")
    return sample


def normalize_word_result(
    value: Any,
    *,
    source_sha256: str,
    sample_sha256: str,
    source_duration_seconds: float,
    sample_start_seconds: float,
    sample_duration_seconds: float,
    provider_version: str,
    model: dict,
    language_probability: float,
) -> dict:
    """Project bounded provider output; never make up word times or confidence."""
    if not isinstance(value, Mapping) or not isinstance(value.get("segments"), list):
        _refuse()
    if not 1 <= len(value["segments"]) <= WORDS_MAX_SEGMENTS:
        _refuse()
    words, segments, exclusions = [], [], []
    token_index = 0
    lexical = degenerate = 0
    try:
        bound = _number(sample_duration_seconds, 0, WORDS_MAX_SAMPLE_SECONDS)
    except LocalMediaError:
        bound = None
    if bound is not None and bound <= 0:
        bound = None
    for segment_index, segment in enumerate(value["segments"]):
        if not isinstance(segment, Mapping) or not isinstance(
            segment.get("words"), list
        ):
            _refuse()
        segments.append(
            {
                "start_seconds": _provider_number(segment.get("start")),
                "end_seconds": _provider_number(segment.get("end")),
                "compression_ratio": _provider_number(segment.get("compression_ratio")),
                "average_log_probability": _provider_number(segment.get("avg_logprob")),
                "no_speech_probability": _provider_number(
                    segment.get("no_speech_prob")
                ),
            }
        )
        if token_index + len(segment["words"]) > WORDS_MAX_COUNT:
            _refuse()
        for word in segment["words"]:
            if not isinstance(word, Mapping) or not isinstance(word.get("word"), str):
                _refuse()
            text = word["word"].strip()
            if (
                not text
                or _token_bytes(text) > WORDS_MAX_TOKEN_BYTES
                or any(ord(char) < 32 for char in text)
            ):
                _refuse()
            if not any(char.isalnum() for char in text):
                exclusions.append(
                    {"token_index": token_index, "reason": "punctuation_only"}
                )
            else:
                lexical += 1
                retained = {
                    "text": text,
                    "start_seconds": _provider_number(word.get("start")),
                    "end_seconds": _provider_number(word.get("end")),
                    "probability": _provider_number(word.get("probability")),
                    "segment_index": segment_index,
                }
                # A zero-or-negative span carries no duration, so omitting the
                # token leaves the elapsed-time denominator untouched and moves
                # the word count by one. Recorded, never repaired.
                #
                # Only a pair the receipt would otherwise accept is judged here.
                # A malformed timestamp is retained and refused by
                # validate_word_sample below, keeping its own diagnostic.
                if _degenerate_span(
                    retained["start_seconds"], retained["end_seconds"], bound
                ):
                    degenerate += 1
                    exclusions.append(
                        {"token_index": token_index, "reason": "nonpositive_span"}
                    )
                else:
                    words.append(retained)
            token_index += 1
    if degenerate and (
        lexical == 0 or degenerate / lexical > WORDS_MAX_NONPOSITIVE_SHARE
    ):
        _refuse("word_nonpositive_span")
    return validate_word_sample(
        {
            "schema_version": 3,
            "pipeline_version": WORDS_PIPELINE_VERSION,
            "source_sha256": source_sha256,
            "sample_sha256": sample_sha256,
            "source_duration_seconds": source_duration_seconds,
            "sample_start_seconds": sample_start_seconds,
            "sample_duration_seconds": sample_duration_seconds,
            "provider": "mlx-whisper",
            "provider_version": provider_version,
            "model": copy.deepcopy(model),
            "language": value.get("language"),
            "language_probability": language_probability,
            # Comparing against an unvalidated duration would raise before the
            # reader can refuse it; pass the caller's value through instead.
            "language_probe_seconds": (
                min(30.0, bound) if bound is not None else sample_duration_seconds
            ),
            "words": words,
            "segments": segments,
            "token_exclusions": exclusions,
        }
    )

skills

vault-ingress

scripts

adherence_baseline.py

aggregate-catalog-feedback.py

apply-source-repairs.py

artifact_locator.py

artifact_metadata.py

artifact_supervisor.py

audit-pattern-catalog.py

audit-persisted-pattern-observations.py

audit-source-identities.py

batch-download-videos.py

build-contact-sheet.py

build-crop-reviewer.py

build-score-basis.py

catalog_dimension_registry.py

catalog_io.py

catalog_normalization.py

check-runtime.py

classify-pptx-evidence.py

cloud_artifacts.py

cooperative_lock.py

crop_frames.py

crop-reviewer-shell.html

crop-reviewer-shell.html.txt

crop-reviewer.js

crop-reviewer.js.txt

establish-date-provenance.py

failure_diagnostics.py

fetch-transcript.py

ingress_contract.py

local_media_contract.py

local_media_download.py

local_media_evidence.py

local_media_process.py

local_media_sampling.py

local_media_transcription.py

local_media_words.py

markdown_deck.py

migrate-tracking-database.py

mutate-tracking-database.py

pattern_evidence.py

pdf_evidence.py

persist-results.py

persisted_pattern_observations.py

pptx_catalog_selection.py

pptx_deck_facts.py

pptx_discovery_contract.py

pptx_evidence.py

pptx_talk_identity.py

pptx-extraction.py

preflight-vault.py

queue_claim_contract.py

queue-state.py

read-tracking-database.py

render-markdown-deck.py

render-vault-status.py

retained_stage.py

return_validation.py

scan-shownotes.py

source_alias_contract.py

source_identity_matching.py

summary_lock.py

sweep-pptx-talk-identity.py

tracking_database_io.py

tracking_database.py

transcript_quality.py

transcript_timing.py

validate-returns.py

vault_root_authority.py

video_evidence.py

video_integrity.py

video-slide-extraction.py

vtt-cleanup.py

write-analysis.py

ytdlp_runtime.py

SKILL.md

README.md

tile.json