CtrlK
BlogDocsLog inGet started
Tessl Logo

spec-driven-development/spec-as-source

Spec-driven development on OpenSpec, with mechanical spec-as-source enforcement: a custom 'spec-as-source' OpenSpec schema adds file-ownership (targets) and test-verification ([@test]) metadata to every capability spec, three scripts (link check, ownership check, manifest build) keep code and specs from drifting apart, plus requirement-gathering, spec-writer, work-review, and a session-handoff skill with a proactive context-warning hook and a packaged handoff memory: the skill ships the exporter, importer, graph model, facts pipeline, Neo4j Compose runtime and operating guide to load handoffs into a local, authenticated Neo4j graph and query them.

68

Quality

85%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide
SecuritybySnyk

Low

Low-risk findings worth noting

Overview
Quality
Evals
Security
Files

graph_export.pyskills/handoff/scripts/

#!/usr/bin/env python3
# GENERATED FROM SPEC — DO NOT EDIT DIRECTLY
# Source: openspec/specs/handoff-graph-export/spec.md
"""Create a deterministic, privacy-conscious graph sidecar for a handoff.

Schema version 2 carries the approved structured handoff metadata (Markdown
frontmatter, then a sibling ``HANDOFF-NNN.meta.yaml``, then null), failure
records and source-folder provenance. Version 1 remains selectable and
validates unchanged through version dispatch.

The built-in validator intentionally supports the small JSON Schema keyword
set used by the checked-in schemas. It is not a general JSON Schema validator.

Supported YAML subset (metadata only, never evaluated):
  * a JSON document (machine-written ``.meta.yaml``), duplicate keys rejected;
  * or block mappings indented by two spaces, ``key: value`` scalars, flow
    lists ``[a, "b"]`` of scalars, block lists ``- item`` of scalars, full-line
    and `` #`` trailing comments;
  * scalars: ``null``/``~``/empty -> null, ``true``/``false`` -> booleans,
    single- or double-quoted strings, everything else is a plain string;
  * anchors, aliases, tags, block scalars, flow mappings, tabs, multiple
    documents and duplicate keys are rejected.

Commands:
  graph_export.py HANDOFF-NNN.md            export (v2 when metadata exists, else v1)
  graph_export.py --new HANDOFF-NNN.md      strict new-save profile, v2 only
  graph_export.py --check-only [--as HANDOFF-NNN.md] DRAFT.md
                                            validate new-save frontmatter, write nothing
  graph_export.py recover --manifest M --inventory I --classification C [--out DIR]
                                            deterministic historical recovery to local candidates
                                            (default DIR: output/private/ of the current git work tree;
                                            a DIR inside a work tree must be ignored by git)
  graph_export.py publish --candidates DIR --manifest M [--collection C]
                  [--replace PATH=SHA256 ...] create-only publication of candidates
"""

from __future__ import annotations

import argparse
import datetime as _dt
import hashlib
import json
import os
import re
import subprocess
import sys
import tempfile
import uuid
from pathlib import Path
from typing import Any


REFERENCES = Path(__file__).resolve().parents[1] / "references"
SCHEMA_PATH = REFERENCES / "graph-export.schema.json"
SCHEMA_PATHS = {1: REFERENCES / "graph-export-v1.schema.json", 2: SCHEMA_PATH}
CURRENT_VERSION = 2
SCHEMA_KEYWORDS = {
    "$comment",
    "$schema",
    "$id",
    "title",
    "type",
    "required",
    "additionalProperties",
    "properties",
    "items",
    "enum",
    "pattern",
    "minLength",
}
HEADING_RE = re.compile(r"^\s{0,3}(#{1,6})\s+(.+?)\s*#*\s*$")
NUMBERED_RE = re.compile(r"^\s*\d+[.)]\s+(.+?)\s*$")
BULLET_RE = re.compile(r"^(\s{0,3})(?:[-*+]|\d+[.)])\s+(.*)$")
LABEL_RE = re.compile(r"^\s*(?:[-*+]\s+|\d+[.)]\s+)?(?:\[(decision|fact)\]|(decision|fact)\s*:)\s*(.*?)\s*$", re.I)
HANDOFF_NAME_RE = re.compile(r"^HANDOFF-([0-9]+)\.md$")
HANDOFF_REF_RE = re.compile(r"^HANDOFF-([0-9]+)$")
SECRET_PATTERNS = (
    re.compile(r"(?i)\b(?:api[_ -]?key|access[_ -]?token|refresh[_ -]?token|password|passwd|secret|client[_ -]?secret|authorization)\b\s*[:=]\s*[^\s,;]+"),
    re.compile(r"(?i)\bbearer\s+[a-z0-9._~+/=-]{8,}"),
    re.compile(r"\bAKIA[0-9A-Z]{16}\b"),
    re.compile(r"\bgh[pousr]_[A-Za-z0-9_]{20,}\b"),
    re.compile(r"\bgithub_pat_[A-Za-z0-9_]{20,}\b"),
    re.compile(r"\bsk-[A-Za-z0-9_-]{20,}\b"),
    re.compile(r"-----BEGIN [A-Z ]*PRIVATE KEY-----"),
    re.compile(r"\b[A-Za-z0-9_-]{16,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\b"),
)

METADATA_FIELDS = (
    "handoff",
    "project_id",
    "date",
    "continues",
    "agent_session",
    "parent_session",
    "operator",
    "client",
    "work_context",
    "track",
    "plan_entry",
    "change",
    "deadline",
)
OPERATOR_FIELDS = ("person", "provider", "models")
NULLABLE_STRING_FIELDS = ("agent_session", "parent_session", "client", "track", "plan_entry", "change", "deadline")
PROFILES = ("new", "historical")
OFFSET_DATETIME_RE = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(?::\d{2}(?:\.\d+)?)?(?:Z|[+-]\d{2}:\d{2})$")
LOCAL_DATETIME_RE = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(?::\d{2})?$")
DATE_ONLY_RE = re.compile(r"^\d{4}-\d{2}-\d{2}$")
META_ENVELOPE_KEYS = {"meta_version", "profile", "identity", "metadata", "evidence", "diagnostics", "classification"}
NO_FAILURE_PLACEHOLDERS = {
    "",
    "-",
    "—",
    "n/a",
    "na",
    "none",
    "nothing",
    "nessuno",
    "nessuna",
    "niente",
    "nulla",
    "nessun problema",
    "nessun fallimento",
    "nessun errore",
    "no failures",
    "nothing failed",
}


class ExportError(Exception):
    """An error safe to report without exposing source text."""


class SchemaValidationError(ExportError):
    pass


class MetadataError(ExportError):
    pass


class PublicationConflict(ExportError):
    pass


# --------------------------------------------------------------------------- schema


def _type_matches(value: Any, expected: Any) -> bool:
    if isinstance(expected, list):
        return any(_type_matches(value, item) for item in expected)
    if expected == "object":
        return isinstance(value, dict)
    if expected == "array":
        return isinstance(value, list)
    if expected == "integer":
        return isinstance(value, int) and not isinstance(value, bool)
    if expected == "string":
        return isinstance(value, str)
    if expected == "boolean":
        return isinstance(value, bool)
    if expected == "null":
        return value is None
    if expected == "number":
        return isinstance(value, (int, float)) and not isinstance(value, bool)
    raise SchemaValidationError(f"unsupported JSON Schema type at {expected}")


def _schema_keywords(schema: dict[str, Any], path: str) -> None:
    unknown = set(schema) - SCHEMA_KEYWORDS
    if unknown:
        keyword = sorted(unknown)[0]
        raise SchemaValidationError(f"unsupported JSON Schema keyword at {path} ({keyword})")


def _assert_supported_schema(schema: Any, path: str = "$") -> None:
    if not isinstance(schema, dict):
        raise SchemaValidationError(f"invalid JSON Schema at {path}")
    _schema_keywords(schema, path)
    for key, child in schema.get("properties", {}).items():
        _assert_supported_schema(child, f"{path}.properties.{key}")
    if isinstance(schema.get("items"), dict):
        _assert_supported_schema(schema["items"], f"{path}.items")
    if isinstance(schema.get("additionalProperties"), dict):
        _assert_supported_schema(schema["additionalProperties"], f"{path}.additionalProperties")


def _validate(instance: Any, schema: dict[str, Any], path: str = "$") -> None:
    if not isinstance(schema, dict):
        raise SchemaValidationError(f"invalid JSON Schema at {path}")
    _schema_keywords(schema, path)
    expected = schema.get("type")
    if expected is not None and not _type_matches(instance, expected):
        raise SchemaValidationError(f"schema mismatch at {path} (type)")
    if "enum" in schema and instance not in schema["enum"]:
        raise SchemaValidationError(f"schema mismatch at {path} (enum)")
    if isinstance(instance, str):
        if len(instance) < schema.get("minLength", 0):
            raise SchemaValidationError(f"schema mismatch at {path} (minLength)")
        pattern = schema.get("pattern")
        if pattern is not None and re.search(pattern, instance) is None:
            raise SchemaValidationError(f"schema mismatch at {path} (pattern)")
    if isinstance(instance, dict):
        for key in schema.get("required", []):
            if key not in instance:
                raise SchemaValidationError(f"schema mismatch at {path}.{key} (required)")
        properties = schema.get("properties", {})
        additional = schema.get("additionalProperties", True)
        for key, value in instance.items():
            if key in properties:
                _validate(value, properties[key], f"{path}.{key}")
            elif additional is False:
                raise SchemaValidationError(f"schema mismatch at {path} (additionalProperties)")
            elif isinstance(additional, dict):
                _validate(value, additional, f"{path}.{key}")
    if isinstance(instance, list) and "items" in schema:
        for index, value in enumerate(instance):
            _validate(value, schema["items"], f"{path}[{index}]")


def validate_schema_instance(instance: Any, schema: dict[str, Any]) -> None:
    """Validate against the supported keyword subset in a loaded schema."""
    _assert_supported_schema(schema)
    _validate(instance, schema)


def load_schema(path: Path = SCHEMA_PATH) -> dict[str, Any]:
    try:
        schema = json.loads(path.read_text(encoding="utf-8"))
    except (OSError, json.JSONDecodeError) as exc:
        raise ExportError("could not load the graph export schema") from exc
    if not isinstance(schema, dict):
        raise ExportError("the graph export schema must be a JSON object")
    _assert_supported_schema(schema)
    return schema


def schema_for_version(version: Any) -> dict[str, Any]:
    """Return the checked-in schema for a declared sidecar version."""
    if isinstance(version, bool) or version not in SCHEMA_PATHS:
        raise SchemaValidationError("unsupported schema version")
    return load_schema(SCHEMA_PATHS[version])


def validate_document(document: Any) -> None:
    """Validate a sidecar against the schema selected by its declared version."""
    if not isinstance(document, dict) or "schema_version" not in document:
        raise SchemaValidationError("schema mismatch at $.schema_version (required)")
    validate_schema_instance(document, schema_for_version(document["schema_version"]))


# --------------------------------------------------------------------------- identities


def _valid_uuid(value: str) -> bool:
    try:
        return str(uuid.UUID(value)) == value
    except (ValueError, AttributeError, TypeError):
        return False


def _read_project_identity(handoff_dir: Path) -> uuid.UUID | None:
    identity_file = handoff_dir / "graph-project-id"
    try:
        current = identity_file.read_text(encoding="ascii").strip()
    except FileNotFoundError:
        return None
    except (OSError, UnicodeError) as exc:
        raise ExportError("could not read the stored project identity") from exc
    if not current:
        return None
    if not _valid_uuid(current):
        raise ExportError("the stored project identity is invalid")
    return uuid.UUID(current)


def _project_id(handoff_dir: Path) -> uuid.UUID:
    existing = _read_project_identity(handoff_dir)
    if existing is not None:
        return existing
    handoff_dir.mkdir(parents=True, exist_ok=True)
    identity_file = handoff_dir / "graph-project-id"
    generated = str(uuid.uuid4())
    try:
        fd = os.open(identity_file, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600)
    except FileExistsError:
        raced = _read_project_identity(handoff_dir)
        if raced is None:
            raise ExportError("the stored project identity is invalid")
        return raced
    try:
        with os.fdopen(fd, "w", encoding="ascii") as identity:
            identity.write(generated + "\n")
            identity.flush()
            os.fsync(identity.fileno())
    except OSError as exc:
        try:
            identity_file.unlink()
        except OSError:
            pass
        raise ExportError("could not save the project identity") from exc
    return uuid.UUID(generated)


# --------------------------------------------------------------------------- text helpers


def _normalise_text(text: str) -> str:
    return " ".join(text.strip().split())


def _contains_secret(text: str) -> bool:
    return any(pattern.search(text) for pattern in SECRET_PATTERNS)


def _section_key(section: str) -> str:
    return re.sub(r"[^a-z0-9]+", " ", section.casefold()).strip()


def split_frontmatter(markdown: str) -> tuple[str | None, str]:
    """Return (frontmatter text or None, body). Frontmatter must open the file."""
    if not markdown.startswith("---\n"):
        return None, markdown
    end = markdown.find("\n---\n", 3)
    if end == -1:
        if markdown.endswith("\n---"):
            end = len(markdown) - 4
        else:
            raise MetadataError("frontmatter is not closed")
    return markdown[4:end + 1], markdown[end + 5:]


def _failure_text(block: list[str]) -> str:
    return _normalise_text(" ".join(block))


def _is_placeholder(text: str) -> bool:
    simplified = re.sub(r"[.!*_`]+", "", text).strip().casefold()
    simplified = re.sub(r"\s*\(.*\)\s*$", "", simplified).strip().strip('"“”').strip()
    return simplified in NO_FAILURE_PLACEHOLDERS or simplified.startswith(("nessun vicolo cieco", "nessun fallimento", "nessun problema", "no dead end"))


def _extract_candidates(markdown: str, *, failures: bool = False) -> tuple[list[tuple[str, str, str]], int]:
    elements, omitted = extract_elements(markdown, failures=failures)
    return [(record_type, text, section) for record_type, text, section, _, _ in elements], omitted


def extract_elements(markdown: str, *, failures: bool = False) -> tuple[list[tuple[str, str, str, int, int]], int]:
    """Scan record candidates as (type, text, section, first line, last line); lines are 1-based in ``markdown``."""
    section = ""
    in_fence = False
    in_comment = False
    candidates: list[tuple[str, str, str, int, int]] = []
    omitted = 0
    block: list[str] = []
    block_start = block_end = 0

    def add(record_type: str, text: str, start: int, end: int) -> None:
        nonlocal omitted
        if not text:
            return
        if _contains_secret(text):
            omitted += 1
        else:
            candidates.append((record_type, text, section, start, end))

    def flush() -> None:
        nonlocal block
        if block:
            text = _failure_text(block)
            if not _is_placeholder(text):
                add("failure", text, block_start, block_end)
        block = []

    for number, line in enumerate(markdown.splitlines(), start=1):
        stripped = line.strip()
        failure_section = failures and _section_key(section) == "what didn t work"
        if stripped.startswith("```") or stripped.startswith("~~~"):
            if failure_section:
                flush()
            in_fence = not in_fence
            continue
        if in_fence:
            continue
        if failure_section and (in_comment or stripped.startswith("<!--")):
            in_comment = "-->" not in stripped
            continue
        heading = HEADING_RE.match(line)
        if heading:
            flush()
            section = heading.group(2).strip()
            continue
        if not section:
            continue
        label = LABEL_RE.match(line)
        if label:
            flush()
            record_type = (label.group(1) or label.group(2)).casefold()
            add("decision" if record_type == "decision" else "durable_fact", _normalise_text(label.group(3)), number, number)
            continue
        if _section_key(section) == "next steps":
            item = NUMBERED_RE.match(line)
            if item:
                add("next_action", _normalise_text(item.group(1)), number, number)
            continue
        if failure_section:
            if not stripped or re.fullmatch(r"-{3,}|\*{3,}|_{3,}", stripped):
                flush()
                continue
            bullet = BULLET_RE.match(line)
            if bullet:
                flush()
                block = [bullet.group(2)]
                block_start = block_end = number
            else:
                block.append(stripped)
                block_end = number
    flush()
    return candidates, omitted


# --------------------------------------------------------------------------- YAML subset


def _reject_duplicates(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
    result: dict[str, Any] = {}
    for key, value in pairs:
        if key in result:
            raise MetadataError("metadata contains a duplicate key")
        result[key] = value
    return result


def _strip_comment(text: str) -> str:
    quote = None
    for index, char in enumerate(text):
        if quote:
            if char == quote:
                quote = None
        elif char in "\"'":
            quote = char
        elif char == "#" and (index == 0 or text[index - 1] in " \t"):
            return text[:index].rstrip()
    return text.rstrip()


def _yaml_scalar(raw: str) -> Any:
    value = raw.strip()
    if value in ("", "null", "~", "Null", "NULL"):
        return None
    if value in ("true", "false"):
        return value == "true"
    if value[0] in "&*!|>{":
        raise MetadataError("metadata uses an unsupported YAML construct")
    if value[0] == '"':
        try:
            parsed = json.loads(value)
        except json.JSONDecodeError as exc:
            raise MetadataError("metadata contains a malformed quoted string") from exc
        if not isinstance(parsed, str):
            raise MetadataError("metadata contains a malformed quoted string")
        return parsed
    if value[0] == "'":
        if len(value) < 2 or not value.endswith("'"):
            raise MetadataError("metadata contains a malformed quoted string")
        inner = value[1:-1]
        if re.search(r"(?<!')'(?!')", inner):
            raise MetadataError("metadata contains a malformed quoted string")
        return inner.replace("''", "'")
    if value.startswith("- ") or value == "-":
        raise MetadataError("metadata contains malformed YAML")
    return value


def _split_flow(inner: str) -> list[str]:
    items, current, quote = [], "", None
    for char in inner:
        if quote:
            current += char
            if char == quote:
                quote = None
        elif char in "\"'":
            quote = char
            current += char
        elif char == ",":
            items.append(current)
            current = ""
        elif char in "[]{}":
            raise MetadataError("metadata uses an unsupported YAML construct")
        else:
            current += char
    if quote:
        raise MetadataError("metadata contains a malformed quoted string")
    items.append(current)
    return items


def _yaml_value(raw: str) -> Any:
    value = raw.strip()
    if value.startswith("["):
        if not value.endswith("]"):
            raise MetadataError("metadata contains a malformed list")
        inner = value[1:-1].strip()
        if not inner:
            return []
        parsed = []
        for item in _split_flow(inner):
            if not item.strip():
                raise MetadataError("metadata contains a malformed list")
            parsed.append(_yaml_scalar(item))
        return parsed
    return _yaml_scalar(value)


def _parse_block(lines: list[tuple[int, str]], start: int, indent: int) -> tuple[Any, int]:
    """Parse a mapping or a list of scalars at ``indent`` starting at ``start``."""
    if lines[start][1].startswith("- ") or lines[start][1] == "-":
        items: list[Any] = []
        index = start
        while index < len(lines) and lines[index][0] == indent and (lines[index][1].startswith("- ") or lines[index][1] == "-"):
            text = lines[index][1][1:].strip()
            if re.match(r"^[^\"'\[]*:\s", text + " ") and not text.startswith(("\"", "'")):
                raise MetadataError("metadata uses an unsupported YAML construct")
            items.append(_yaml_value(text))
            index += 1
        if index < len(lines) and lines[index][0] > indent:
            raise MetadataError("metadata contains malformed indentation")
        return items, index
    mapping: list[tuple[str, Any]] = []
    index = start
    while index < len(lines) and lines[index][0] == indent:
        text = lines[index][1]
        match = re.match(r"^([A-Za-z_][A-Za-z0-9_-]*)\s*:(?:\s+(.*))?$", text) or re.match(r"^([A-Za-z_][A-Za-z0-9_-]*):$", text)
        if not match:
            raise MetadataError("metadata contains malformed YAML")
        key = match.group(1)
        rest = (match.group(2) if match.lastindex and match.lastindex >= 2 else None) or ""
        index += 1
        if rest.strip():
            mapping.append((key, _yaml_value(rest)))
        elif index < len(lines) and lines[index][0] > indent:
            child, index = _parse_block(lines, index, lines[index][0])
            mapping.append((key, child))
        elif index < len(lines) and lines[index][0] == indent and lines[index][1].startswith("- "):
            child, index = _parse_block(lines, index, indent)
            mapping.append((key, child))
        else:
            mapping.append((key, None))
    if index < len(lines) and lines[index][0] > indent:
        raise MetadataError("metadata contains malformed indentation")
    return _reject_duplicates(mapping), index


def parse_yaml_subset(text: str) -> dict[str, Any]:
    """Parse the documented YAML subset (or JSON) into a mapping without evaluation."""
    stripped = text.strip()
    if stripped.startswith("{"):
        try:
            parsed = json.loads(stripped, object_pairs_hook=_reject_duplicates)
        except json.JSONDecodeError as exc:
            raise MetadataError("metadata is not valid JSON") from exc
        if not isinstance(parsed, dict):
            raise MetadataError("metadata must be a mapping")
        return parsed
    lines: list[tuple[int, str]] = []
    for raw in text.splitlines():
        if "\t" in raw[: len(raw) - len(raw.lstrip())]:
            raise MetadataError("metadata indentation must use spaces")
        if raw.strip() in ("---", "..."):
            raise MetadataError("metadata must be a single YAML document")
        content = _strip_comment(raw)
        if not content.strip():
            continue
        indent = len(content) - len(content.lstrip(" "))
        lines.append((indent, content.strip()))
    if not lines:
        return {}
    if lines[0][0] != 0:
        raise MetadataError("metadata contains malformed indentation")
    parsed, index = _parse_block(lines, 0, 0)
    if index != len(lines) or not isinstance(parsed, dict):
        raise MetadataError("metadata must be a mapping")
    return parsed


# --------------------------------------------------------------------------- metadata


def _handoff_number(name: str) -> int:
    match = HANDOFF_NAME_RE.match(name)
    if not match:
        raise ExportError("source filename must match HANDOFF-NNN.md")
    return int(match.group(1))


def _field_error(field: str, reason: str) -> MetadataError:
    return MetadataError(f"metadata field {field}: {reason}")


def _check_types(metadata: dict[str, Any], profile: str) -> None:
    unknown = set(metadata) - set(METADATA_FIELDS)
    if unknown:
        raise MetadataError(f"metadata has unknown field {sorted(unknown)[0]}")
    missing = [field for field in METADATA_FIELDS if field not in metadata]
    if missing:
        raise _field_error(missing[0], "missing")
    if not isinstance(metadata["handoff"], str) or not HANDOFF_REF_RE.match(metadata["handoff"]):
        raise _field_error("handoff", "must be HANDOFF-NNN")
    if not isinstance(metadata["project_id"], str) or not metadata["project_id"].strip():
        raise _field_error("project_id", "must be a nonempty canonical project id")
    date = metadata["date"]
    if date is not None and not isinstance(date, str):
        raise _field_error("date", "must be an ISO string or null")
    if metadata["continues"] is not None and (not isinstance(metadata["continues"], str) or not HANDOFF_REF_RE.match(metadata["continues"])):
        raise _field_error("continues", "must be HANDOFF-NNN or null")
    for field in NULLABLE_STRING_FIELDS:
        value = metadata[field]
        if value is not None and (not isinstance(value, str) or not value.strip()):
            raise _field_error(field, "must be a nonempty string or null")
    operator = metadata["operator"]
    if not isinstance(operator, dict) or set(operator) != set(OPERATOR_FIELDS):
        raise _field_error("operator", "must be a mapping with person, provider and models")
    for key in ("person", "provider"):
        if operator[key] is not None and (not isinstance(operator[key], str) or not operator[key].strip()):
            raise _field_error(f"operator.{key}", "must be a nonempty string or null")
    models = operator["models"]
    if models is None and profile == "new":
        raise _field_error("operator.models", "must be a list")
    if models is not None and (not isinstance(models, list) or not all(isinstance(m, str) and m.strip() for m in models)):
        raise _field_error("operator.models", "must be a list of nonempty strings")
    context = metadata["work_context"]
    if context is None and profile == "new":
        raise _field_error("work_context", "must be a list")
    if context is not None and (not isinstance(context, list) or not all(isinstance(c, str) and c.strip() for c in context)):
        raise _field_error("work_context", "must be a list of nonempty strings")


def _secret_values(metadata: dict[str, Any]) -> list[str]:
    found = []
    for field in METADATA_FIELDS:
        value = metadata.get(field)
        values = value if isinstance(value, list) else [value]
        if field == "operator" and isinstance(value, dict):
            values = [value.get("person"), value.get("provider"), *(value.get("models") or [])]
        if any(isinstance(item, str) and _contains_secret(item) for item in values):
            found.append(field)
    return found


def validate_metadata(
    metadata: dict[str, Any],
    *,
    profile: str,
    filename: str,
    available: set[str],
) -> None:
    """Validate approved metadata for a profile. ``available`` lists same-folder handoff filenames."""
    if profile not in PROFILES:
        raise MetadataError("unknown metadata profile")
    _check_types(metadata, profile)
    sensitive = _secret_values(metadata)
    if sensitive:
        raise _field_error(sensitive[0], "contains a credential-like value")
    number = _handoff_number(filename)
    if metadata["handoff"] != filename[:-3]:
        raise _field_error("handoff", "does not match the filename")
    date = metadata["date"]
    if profile == "new":
        if not isinstance(date, str) or not OFFSET_DATETIME_RE.match(date):
            raise _field_error("date", "must be an ISO date-time with timezone")
    elif date is not None and not (OFFSET_DATETIME_RE.match(date) or LOCAL_DATETIME_RE.match(date) or DATE_ONLY_RE.match(date)):
        raise _field_error("date", "must be ISO or null")
    if isinstance(date, str):
        try:
            if DATE_ONLY_RE.match(date):
                _dt.date.fromisoformat(date)
            else:
                _dt.datetime.fromisoformat(date.replace("Z", "+00:00"))
        except ValueError as exc:
            raise _field_error("date", "is not a valid calendar value") from exc
    reference = metadata["continues"]
    if reference is not None:
        target = int(HANDOFF_REF_RE.match(reference).group(1))
        if target == number:
            raise _field_error("continues", "references the handoff itself")
        if target > number:
            raise _field_error("continues", "references a later handoff")
        if f"{reference}.md" not in available:
            raise _field_error("continues", "references a handoff that does not exist in this folder")


def _parse_meta_envelope(text: str) -> dict[str, Any]:
    envelope = parse_yaml_subset(text)
    unknown = set(envelope) - META_ENVELOPE_KEYS
    if unknown or "metadata" not in envelope or "profile" not in envelope:
        raise MetadataError("metadata file has an invalid envelope")
    if envelope.get("meta_version") != 1:
        raise MetadataError("metadata file has an unsupported version")
    if envelope["profile"] not in PROFILES:
        raise MetadataError("metadata file has an unknown profile")
    if not isinstance(envelope["metadata"], dict):
        raise MetadataError("metadata file has an invalid envelope")
    identity = envelope.get("identity")
    if identity is not None:
        if not isinstance(identity, dict) or set(identity) - {"technical_project_uuid", "collection"}:
            raise MetadataError("metadata file has an invalid identity envelope")
        tech = identity.get("technical_project_uuid")
        if tech is not None and not (isinstance(tech, str) and _valid_uuid(tech)):
            raise MetadataError("metadata file has an invalid technical project identity")
    return envelope


def resolve_metadata(
    frontmatter: dict[str, Any] | None,
    meta: dict[str, Any] | None,
    *,
    require_complete_frontmatter: bool,
) -> tuple[dict[str, Any], str]:
    """Merge frontmatter > metadata file > null. Explicit null is a value."""
    if frontmatter is not None:
        unknown = set(frontmatter) - set(METADATA_FIELDS)
        if unknown:
            raise MetadataError(f"frontmatter has unknown field {sorted(unknown)[0]}")
        if require_complete_frontmatter:
            missing = [field for field in METADATA_FIELDS if field not in frontmatter]
            if missing:
                raise _field_error(missing[0], "missing from frontmatter")
    merged: dict[str, Any] = {}
    for field in METADATA_FIELDS:
        if frontmatter is not None and field in frontmatter:
            merged[field] = frontmatter[field]
        elif meta is not None and field in meta:
            merged[field] = meta[field]
        else:
            merged[field] = None
    origin = "+".join(name for name, present in (("frontmatter", frontmatter is not None), ("meta", meta is not None)) if present)
    return merged, origin


# --------------------------------------------------------------------------- building


def _available_handoffs(directory: Path, extra: str | None = None) -> set[str]:
    names = {path.name for path in directory.glob("HANDOFF-*.md") if HANDOFF_NAME_RE.match(path.name)}
    if extra:
        names.add(extra)
    return names


def git_work_tree(path: Path) -> Path | None:
    """Return the nearest ancestor of ``path`` (itself included) holding a ``.git`` entry."""
    current = path.expanduser().resolve()
    for candidate in (current, *current.parents):
        if (candidate / ".git").exists():
            return candidate
    return None


def relative_source_directory(directory: Path) -> str:
    """Write a handoff folder relative to the git work tree that contains it.

    POSIX form, ``.`` for the work-tree root; the folder's own name when no
    ancestor holds ``.git``. Never an absolute path.
    """
    folder = directory.expanduser().resolve()
    tree = git_work_tree(folder)
    if tree is None:
        return folder.name or "."
    relative = folder.relative_to(tree).as_posix()
    return relative or "."


def _records(
    markdown: str,
    handoff_id: uuid.UUID,
    filename: str,
    directory: str | None,
    version: int,
    spans: dict[str, dict[str, Any]] | None = None,
    line_offset: int = 0,
) -> tuple[list[dict[str, Any]], int]:
    candidates, omitted = extract_elements(markdown, failures=version >= 2)
    records: list[dict[str, Any]] = []
    occurrences: dict[tuple[str, str, str], int] = {}
    for record_type, text, section, first, last in candidates:
        key = (record_type, section, text)
        occurrence = occurrences.get(key, 0)
        occurrences[key] = occurrence + 1
        record_id = uuid.uuid5(handoff_id, f"{record_type}\0{section}\0{text}\0{occurrence}")
        provenance: dict[str, Any] = {"file": filename, "section": section}
        if version >= 2:
            provenance["directory"] = directory
        records.append({"id": str(record_id), "type": record_type, "text": text, "provenance": provenance})
        if spans is not None:
            spans[str(record_id)] = {
                "line_start": first + line_offset,
                "line_end": last + line_offset,
                "text_sha256": hashlib.sha256(text.encode("utf-8")).hexdigest(),
            }
    records.sort(key=lambda record: (record["type"], record["provenance"]["section"], record["text"], record["id"]))
    return records, omitted


def build_document(
    markdown_path: Path,
    *,
    version: int | None = None,
    profile: str | None = None,
    meta_text: str | None = None,
    project_uuid: str | None = None,
    collection: str | None = None,
    available: set[str] | None = None,
    create_identity: bool = True,
    spans: dict[str, dict[str, Any]] | None = None,
) -> tuple[dict[str, Any], int]:
    """Build a sidecar document.

    ``version`` None selects v2 when structured metadata exists and v1 otherwise.
    ``meta_text`` overrides the sibling ``.meta.yaml`` (historical candidates).
    ``create_identity`` False never creates a ``graph-project-id`` file.
    ``spans``, when given, is filled with record id -> 1-based line span in the
    whole Markdown file and text SHA-256; the returned document is unchanged.
    """
    source = markdown_path.expanduser().resolve()
    _handoff_number(source.name)
    if not source.is_file():
        raise ExportError("source Markdown handoff does not exist")
    try:
        markdown = source.read_text(encoding="utf-8")
    except (OSError, UnicodeError) as exc:
        raise ExportError("could not read the Markdown handoff as UTF-8") from exc
    front_text, body = split_frontmatter(markdown)
    frontmatter = parse_yaml_subset(front_text) if front_text is not None else None
    if frontmatter is not None and not isinstance(frontmatter, dict):
        raise MetadataError("frontmatter must be a mapping")
    meta_path = source.with_name(f"{source.stem}.meta.yaml")
    envelope = None
    if meta_text is None and meta_path.exists():
        try:
            meta_text = meta_path.read_text(encoding="utf-8")
        except (OSError, UnicodeError) as exc:
            raise MetadataError("could not read the metadata file as UTF-8") from exc
    if meta_text is not None:
        envelope = _parse_meta_envelope(meta_text)
    has_metadata = frontmatter is not None or envelope is not None
    if version is None:
        version = CURRENT_VERSION if has_metadata else 1
    if profile == "new" and frontmatter is None:
        raise MetadataError("a new handoff requires structured frontmatter")
    if version == 2 and not has_metadata:
        raise MetadataError("schema version 2 requires frontmatter or a metadata file")

    envelope_uuid = None
    if envelope and envelope.get("identity"):
        envelope_uuid = envelope["identity"].get("technical_project_uuid")
        collection = collection or envelope["identity"].get("collection")
    stored = _read_project_identity(source.parent)
    chosen = project_uuid or envelope_uuid
    if chosen is not None:
        if stored is not None and str(stored) != chosen:
            raise ExportError("the technical project identity conflicts with graph-project-id")
        project = uuid.UUID(chosen)
    elif stored is not None:
        project = stored
    elif create_identity:
        project = _project_id(source.parent)
    else:
        raise ExportError("no technical project identity is available")
    handoff_id = uuid.uuid5(project, f"handoff:{source.name}")
    session_id = uuid.uuid5(project, f"session:{source.name}")
    directory = relative_source_directory(source.parent)
    line_offset = markdown[: len(markdown) - len(body)].count("\n")
    records, omitted = _records(body, handoff_id, source.name, directory, version, spans, line_offset)
    if version == 1:
        document = {
            "schema_version": 1,
            "project": {"id": str(project)},
            "session": {"id": str(session_id)},
            "handoff": {"id": str(handoff_id), "filename": source.name},
            "records": [{**record, "provenance": {"file": record["provenance"]["file"], "section": record["provenance"]["section"]}} for record in records],
        }
        return document, omitted

    if profile is None:
        profile = "new" if frontmatter is not None else envelope["profile"]
    metadata, origin = resolve_metadata(
        frontmatter,
        envelope["metadata"] if envelope else None,
        require_complete_frontmatter=profile == "new",
    )
    validate_metadata(
        metadata,
        profile=profile,
        filename=source.name,
        available=available if available is not None else _available_handoffs(source.parent, source.name),
    )
    document = {
        "schema_version": 2,
        "project": {"id": str(project), "project_id": metadata["project_id"]},
        "session": {"id": str(session_id)},
        "handoff": {"id": str(handoff_id), "filename": source.name},
        "source": {"directory": directory, "file": source.name, "collection": collection},
        "metadata_profile": profile,
        "metadata_origin": origin,
        "metadata": metadata,
        "records": records,
    }
    return document, omitted


def canonical_json(document: dict[str, Any]) -> bytes:
    return (json.dumps(document, ensure_ascii=False, indent=2, sort_keys=True) + "\n").encode("utf-8")


def _fsync_dir(directory: Path) -> None:
    try:
        directory_fd = os.open(directory, os.O_RDONLY)
        try:
            os.fsync(directory_fd)
        finally:
            os.close(directory_fd)
    except OSError:
        pass


def _write_temp(destination: Path, payload: bytes) -> Path:
    with tempfile.NamedTemporaryFile("wb", dir=destination.parent, prefix=f".{destination.name}.", suffix=".tmp", delete=False) as temporary:
        temporary.write(payload)
        temporary.flush()
        os.fsync(temporary.fileno())
        return Path(temporary.name)


def publish_bytes(payload: bytes, destination: Path, *, create_only: bool = False, expected_sha256: str | None = None) -> str:
    """Publish atomically. create_only never replaces or truncates an existing file.

    Returns "created", "replaced" or "unchanged" (identical bytes already present).
    """
    destination.parent.mkdir(parents=True, exist_ok=True)
    temp_path: Path | None = None
    try:
        if destination.exists() and destination.read_bytes() == payload:
            return "unchanged"
        if expected_sha256 is not None:
            try:
                current = hashlib.sha256(destination.read_bytes()).hexdigest()
            except FileNotFoundError as exc:
                raise PublicationConflict("the destination to replace no longer exists") from exc
            if current != expected_sha256:
                raise PublicationConflict("the destination changed since it was reviewed")
        elif create_only and destination.exists():
            raise PublicationConflict("destination already exists")
        temp_path = _write_temp(destination, payload)
        if create_only and expected_sha256 is None:
            try:
                os.link(temp_path, destination)
            except FileExistsError as exc:
                raise PublicationConflict("destination already exists") from exc
            outcome = "created"
        else:
            existed = destination.exists()
            os.replace(temp_path, destination)
            temp_path = None
            outcome = "replaced" if existed else "created"
        _fsync_dir(destination.parent)
        return outcome
    except OSError as exc:
        raise ExportError("could not atomically publish the file") from exc
    finally:
        if temp_path is not None:
            try:
                temp_path.unlink()
            except OSError:
                pass


def publish_document(document: dict[str, Any], destination: Path, schema_path: Path | None = None, *, create_only: bool = False) -> str:
    if schema_path is not None:
        validate_schema_instance(document, load_schema(schema_path))
    else:
        validate_document(document)
    return publish_bytes(canonical_json(document), destination, create_only=create_only)


def export_handoff(markdown_path: Path, *, profile: str | None = None) -> tuple[Path, int]:
    document, omitted = build_document(markdown_path, profile=profile, version=2 if profile == "new" else None)
    output = markdown_path.with_name(f"{markdown_path.stem}.graph.json")
    publish_document(document, output)
    return output, omitted


def check_new(draft_path: Path, target_name: str | None = None) -> None:
    """Strictly validate new-save frontmatter of a draft as if it were ``target_name``."""
    draft = draft_path.expanduser().resolve()
    name = target_name or draft.name
    _handoff_number(name)
    try:
        markdown = draft.read_text(encoding="utf-8")
    except (OSError, UnicodeError) as exc:
        raise ExportError("could not read the draft handoff as UTF-8") from exc
    front_text, _ = split_frontmatter(markdown)
    if front_text is None:
        raise MetadataError("a new handoff requires structured frontmatter")
    metadata, _ = resolve_metadata(parse_yaml_subset(front_text), None, require_complete_frontmatter=True)
    validate_metadata(metadata, profile="new", filename=name, available=_available_handoffs(draft.parent, name))


# --------------------------------------------------------------------------- historical recovery


SESSION_ID_RE = re.compile(r"\b([a-z][a-z0-9]*(?:[-_][A-Za-z0-9.]+)+)\b")
PROVIDERS = {"claude": "claude", "codex": "codex", "kimi": "kimi", "gemini": "gemini", "openai": "openai", "chatgpt": "openai", "gpt": "openai"}
EMPTY_VALUES = {"", "-", "—", "–", "n/a", "nessuno", "nessuna", "none", "null"}
TZ_OFFSETS = {"CEST": "+02:00", "CET": "+01:00", "UTC": "+00:00", "GMT": "+00:00", "Z": "+00:00"}


def _header_fields(markdown: str) -> tuple[dict[str, str], dict[str, str]]:
    """Return header key/value pairs before the first level-2 heading, and plan-position body statements."""
    header: dict[str, str] = {}
    body: dict[str, str] = {}
    in_header = True
    for line in markdown.splitlines():
        if line.startswith("## "):
            in_header = False
        if in_header:
            text = line.strip()
            if text.startswith(">"):
                parts = text.lstrip(">").split("|")
            elif text.startswith("**"):
                parts = [text]
            else:
                continue
            for part in parts:
                match = re.match(r"^\s*\*{0,2}([^:*]{1,30}?)\*{0,2}\s*:\s*\*{0,2}\s*(.*?)\s*$", part)
                if match:
                    key = match.group(1).strip().casefold()
                    header.setdefault(key, match.group(2).strip())
        else:
            match = re.match(r"^\s*[-*]\s+\*\*Voce corrente\*\*\s*:\s*(.*)$", line)
            if match and "plan_entry" not in body:
                body["plan_entry"] = match.group(1)
            match = re.match(r"^\s*[-*]?\s*\*{0,2}Change\*{0,2}\s*:\s*`?([a-z0-9][a-z0-9-]+)`?\s*$", line)
            if match and "change" not in body:
                body["change"] = match.group(1)
    return header, body


def _empty(value: str | None) -> bool:
    return value is None or re.sub(r"\s*\(.*\)\s*$", "", value).strip().casefold() in EMPTY_VALUES


def _parse_date(value: str | None, diagnostics: list[str]) -> str | None:
    if _empty(value):
        return None
    match = re.match(r"^\s*(\d{4}-\d{2}-\d{2})(?:[ T](\d{2}):(\d{2})(?::(\d{2}))?)?\s*([A-Z]{1,4})?\b", value)
    if not match:
        diagnostics.append("date: header value is not a recognised ISO date")
        return None
    date, hour, minute, second, zone = match.groups()
    try:
        _dt.date.fromisoformat(date)
    except ValueError:
        diagnostics.append("date: header value is not a valid calendar date")
        return None
    if hour is None:
        return date
    clock = f"{hour}:{minute}" + (f":{second}" if second else "")
    if zone and zone in TZ_OFFSETS:
        return f"{date}T{clock}{TZ_OFFSETS[zone]}"
    diagnostics.append("date: local time without stated timezone kept without offset")
    return f"{date}T{clock}"


def _parse_operator(value: str | None) -> dict[str, Any]:
    operator: dict[str, Any] = {"person": None, "provider": None, "models": None}
    if _empty(value):
        return operator
    text = re.split(r"\s+[—–]\s+|\s+-\s+", value.replace("`", ""))[0]
    parts = [part.strip() for part in re.split(r"\s+\+\s+|\s*,\s*", text) if part.strip()]
    for part in parts:
        note_match = re.search(r"\(([^()]*)\)?", part)
        note = note_match.group(1).strip() if note_match else ""
        main_text = part[: note_match.start()].strip() if note_match else part
        words = main_text.split()
        lowered = [word.casefold() for word in words]
        provider_index = next((i for i, word in enumerate(lowered) if word in PROVIDERS), None)
        if provider_index is None:
            if operator["person"] is None and re.fullmatch(r"[A-ZÀ-Ý][\w'’-]+(?:\s+[A-ZÀ-Ý]?[\w'’-]+){1,3}", main_text):
                operator["person"] = main_text
            continue
        if operator["provider"] is None:
            operator["provider"] = PROVIDERS[lowered[provider_index]]
        model = " ".join(words[provider_index + 1:]).strip()
        if lowered[provider_index] in ("gpt", "chatgpt"):
            model = " ".join(words[provider_index:])
        if not model and note and re.match(r"^(?:[A-Z][A-Za-z]*\s+)?(?:\d|[A-Z][a-z]+\s+\d)", note):
            model = note
        if model:
            operator["models"] = (operator["models"] or []) + [model]
    return operator


def _parse_session(value: str | None) -> str | None:
    if _empty(value):
        return None
    for candidate in re.findall(r"`([^`]+)`", value) + SESSION_ID_RE.findall(value):
        candidate = candidate.strip()
        if re.fullmatch(r"[a-z][a-z0-9]*(?:[-_][A-Za-z0-9.]+)+", candidate) and re.search(r"\d", candidate):
            return candidate
    return None


def extract_historical(markdown: str, filename: str, project_id: str, names: set[str]) -> tuple[dict[str, Any], dict[str, Any], list[str]]:
    """Deterministically extract mechanical metadata. Returns (metadata, evidence, diagnostics)."""
    header, body = _header_fields(markdown)
    diagnostics: list[str] = []
    evidence: dict[str, Any] = {"handoff": "filename", "project_id": "manifest"}
    metadata: dict[str, Any] = {field: None for field in METADATA_FIELDS}
    metadata["handoff"] = filename[:-3]
    metadata["project_id"] = project_id

    date_raw = header.get("data") or header.get("date")
    metadata["date"] = _parse_date(date_raw, diagnostics)
    evidence["date"] = "header" if metadata["date"] else "none"

    continues_raw = header.get("continua da") or header.get("continues")
    if not _empty(continues_raw):
        refs = re.findall(r"HANDOFF-(\d+)(?:\.md)?", continues_raw)
        if refs:
            number = int(refs[0])
            width = max(3, len(refs[0]))
            reference = f"HANDOFF-{number:0{width}d}"
            if f"{reference}.md" not in names:
                raise MetadataError(f"{filename}: explicit continuation references a nonexistent handoff")
            metadata["continues"] = reference
            evidence["continues"] = "header"
            if len(set(refs)) > 1:
                diagnostics.append("continues: several handoffs named in header, first one kept")
        else:
            diagnostics.append("continues: header value names no handoff file")
    evidence.setdefault("continues", "none")

    session_raw = header.get("sessione") or header.get("session")
    metadata["agent_session"] = _parse_session(session_raw)
    evidence["agent_session"] = "header" if metadata["agent_session"] else "none"
    if session_raw and not metadata["agent_session"] and not _empty(session_raw):
        diagnostics.append("agent_session: header names no registry session id")
    parent = re.search(r"(?:parent|padre|sub-?agente di|lanciat[oa] da)\W+`?([a-z][a-z0-9]*(?:[-_][A-Za-z0-9.]+)+)`?", session_raw or "", re.I)
    if parent and re.search(r"\d", parent.group(1)):
        metadata["parent_session"] = parent.group(1)
        evidence["parent_session"] = "header"
    else:
        evidence["parent_session"] = "none"

    operator_raw = header.get("operatore") or header.get("autore") or header.get("operator")
    metadata["operator"] = _parse_operator(operator_raw)
    evidence["operator"] = "header" if operator_raw and not _empty(operator_raw) else "none"

    client_raw = header.get("cliente") or header.get("client")
    metadata["client"] = None if _empty(client_raw) else _normalise_text(client_raw)
    evidence["client"] = "header" if client_raw is not None else "none"

    deadline_raw = header.get("deadline") or header.get("scadenza")
    metadata["deadline"] = None if _empty(deadline_raw) else _normalise_text(deadline_raw)
    evidence["deadline"] = "header" if deadline_raw is not None else "none"

    track_raw = header.get("filone") or header.get("track")
    metadata["track"] = None if _empty(track_raw) else _normalise_text(track_raw)
    evidence["track"] = "header" if metadata["track"] else "none"

    plan_raw = header.get("voce") or header.get("plan entry") or body.get("plan_entry")
    plan = re.search(r"\bE\d{2,}\b", plan_raw or "")
    metadata["plan_entry"] = plan.group(0) if plan else None
    evidence["plan_entry"] = ("header" if header.get("voce") or header.get("plan entry") else "body") if plan else "none"

    change_raw = header.get("change") or body.get("change")
    change = re.fullmatch(r"`?([a-z0-9][a-z0-9-]+)`?", (change_raw or "").strip())
    metadata["change"] = change.group(1) if change else None
    evidence["change"] = ("header" if header.get("change") else "body") if change else "none"
    evidence["work_context"] = "none"

    for field in METADATA_FIELDS:
        value = metadata[field]
        if isinstance(value, str) and _contains_secret(value) and field not in ("handoff", "project_id"):
            metadata[field] = None
            evidence[field] = "omitted"
            diagnostics.append(f"{field}: credential-like value omitted")
    operator = metadata["operator"]
    if any(isinstance(v, str) and _contains_secret(v) for v in [operator["person"], operator["provider"], *(operator["models"] or [])]):
        metadata["operator"] = {"person": None, "provider": None, "models": None}
        evidence["operator"] = "omitted"
        diagnostics.append("operator: credential-like value omitted")
    return metadata, evidence, diagnostics


def _load_json(path: Path) -> Any:
    try:
        return json.loads(path.read_text(encoding="utf-8"), object_pairs_hook=_reject_duplicates)
    except (OSError, UnicodeError, json.JSONDecodeError) as exc:
        raise ExportError(f"could not read {path.name} as JSON") from exc


def load_classification(path: Path, inventory: dict[str, Any]) -> tuple[dict[tuple[str, str], dict[str, Any]], dict[str, Any]]:
    """Load and validate the Sonnet classification keyed by (collection, filename)."""
    data = _load_json(path)
    if not isinstance(data, dict) or not isinstance(data.get("items"), list) or not isinstance(data.get("provenance"), dict):
        raise ExportError("classification must contain provenance and items")
    expected = {(c["collection"], f["name"]) for c in inventory["collections"] for f in c["files"]}
    items: dict[tuple[str, str], dict[str, Any]] = {}
    for item in data["items"]:
        if not isinstance(item, dict):
            raise ExportError("classification item must be an object")
        key = (item.get("collection"), item.get("filename"))
        if key in items:
            raise ExportError(f"classification repeats {key[0]}/{key[1]}")
        if key not in expected:
            raise ExportError("classification names an unknown handoff")
        context, track = item.get("work_context"), item.get("track")
        if context is not None and (not isinstance(context, list) or not all(isinstance(c, str) and c.strip() for c in context)):
            raise ExportError(f"classification work_context invalid for {key[0]}/{key[1]}")
        if track is not None and (not isinstance(track, str) or not track.strip()):
            raise ExportError(f"classification track invalid for {key[0]}/{key[1]}")
        evidence = item.get("evidence", [])
        if not isinstance(evidence, list):
            raise ExportError(f"classification evidence invalid for {key[0]}/{key[1]}")
        for field, value in (("work_context", context), ("track", track)):
            if value not in (None, []) and not any(isinstance(e, dict) and e.get("field") == field and e.get("quote") for e in evidence):
                raise ExportError(f"classification {field} lacks evidence for {key[0]}/{key[1]}")
        items[key] = item
    missing = sorted(expected - set(items))
    if missing:
        raise ExportError(f"classification omits {missing[0][0]}/{missing[0][1]}")
    return items, data["provenance"]


PRIVATE_OUTPUT = Path("output") / "private"


def default_recovery_out(cwd: Path | None = None) -> Path:
    """``output/private/`` under the git work tree of ``cwd``; refused outside any work tree."""
    tree = git_work_tree(Path.cwd() if cwd is None else cwd)
    if tree is None:
        raise ExportError("--out is required outside a git work tree")
    return tree / PRIVATE_OUTPUT


def check_recovery_destination(out_dir: Path) -> None:
    """Refuse a destination inside a git work tree that git does not ignore."""
    destination = out_dir.expanduser().resolve()
    existing = destination
    while not existing.exists() and existing != existing.parent:
        existing = existing.parent
    tree = git_work_tree(existing)
    if tree is None:
        return
    try:
        result = subprocess.run(
            ["git", "-C", str(tree), "check-ignore", "-q", destination.as_posix().rstrip("/") + "/"],
            stdout=subprocess.DEVNULL,
            stderr=subprocess.DEVNULL,
            check=False,
        )
    except OSError as exc:
        raise ExportError("git is required to check the recovery destination") from exc
    if result.returncode != 0:
        raise ExportError("the recovery destination lies inside a git work tree and is not ignored by git; use output/private/ or a folder outside the repository")


def recover(manifest_path: Path, inventory_path: Path, classification_path: Path, out_dir: Path) -> dict[str, Any]:
    """Stage deterministic historical metadata and v2 sidecar candidates under ``out_dir``."""
    check_recovery_destination(out_dir)
    manifest = _load_json(manifest_path)
    inventory = _load_json(inventory_path)
    mapping = {entry["collection"]: entry for entry in manifest["collections"]}
    classification, provenance = load_classification(classification_path, inventory)
    report: dict[str, Any] = {"collections": {}, "null_counts": {field: 0 for field in METADATA_FIELDS}, "held": [], "drift": [], "continues_edges": 0}
    staged: list[tuple[Path, bytes]] = []
    for collection in inventory["collections"]:
        name = collection["collection"]
        entry = mapping.get(name)
        if entry is None:
            raise ExportError(f"manifest has no mapping for {name}")
        source_dir = Path(entry["source_directory"])
        names = {f["name"] for f in collection["files"]}
        live = {p.name for p in source_dir.glob("HANDOFF-*.md") if HANDOFF_NAME_RE.match(p.name)}
        if live != names:
            report["drift"].append({"collection": name, "added": sorted(live - names), "missing": sorted(names - live)})
        counts = {"markdown": len(names), "candidates": 0, "existing_graph": [], "diagnostics": 0}
        for record in collection["files"]:
            path = source_dir / record["name"]
            raw = path.read_bytes()
            if hashlib.sha256(raw).hexdigest() != record["sha256"]:
                report["drift"].append({"collection": name, "changed": record["name"]})
            markdown = raw.decode("utf-8")
            metadata, evidence, diagnostics = extract_historical(markdown, record["name"], entry["project_id"], names)
            item = classification[(name, record["name"])]
            metadata["work_context"] = sorted(dict.fromkeys(item["work_context"])) if item["work_context"] is not None else None
            evidence["work_context"] = "classification" if item["work_context"] is not None else "none"
            if metadata["track"] is None and item.get("track") is not None:
                metadata["track"] = item["track"]
                evidence["track"] = "classification"
            elif metadata["track"] is not None and item.get("track") not in (None, metadata["track"]):
                diagnostics.append("track: header value kept over classification")
            envelope = {
                "meta_version": 1,
                "profile": "historical",
                "identity": {"technical_project_uuid": entry["technical_project_uuid"], "collection": name},
                "metadata": metadata,
                "evidence": evidence,
                "diagnostics": diagnostics,
                "classification": {
                    "model": provenance.get("model"),
                    "session_id": item.get("session_id"),
                    "batch": item.get("batch"),
                    "verified_sha256": provenance.get("verified_sha256"),
                    "evidence": [e for e in item.get("evidence", []) if isinstance(e, dict)],
                },
            }
            meta_bytes = canonical_json(envelope)
            document, _ = build_document(
                path,
                version=2,
                profile="historical",
                meta_text=meta_bytes.decode("utf-8"),
                available=names,
                create_identity=False,
            )
            front_text, _ = split_frontmatter(markdown)
            if front_text is not None:
                diagnostics.append("frontmatter present in historical source: frontmatter values take precedence")
            validate_document(document)
            target = out_dir / name
            staged.append((target / f"{path.stem}.meta.yaml", meta_bytes))
            staged.append((target / f"{path.stem}.graph.json", canonical_json(document)))
            counts["candidates"] += 1
            counts["diagnostics"] += len(diagnostics)
            if metadata["continues"]:
                report["continues_edges"] += 1
            for field in METADATA_FIELDS:
                if metadata[field] is None:
                    report["null_counts"][field] += 1
            for suffix in (".meta.yaml", ".graph.json"):
                existing = path.with_name(path.stem + suffix)
                if existing.exists() and existing.read_bytes() != (meta_bytes if suffix == ".meta.yaml" else canonical_json(document)):
                    counts["existing_graph"].append(existing.name)
                    report["held"].append(str(existing))
        report["collections"][name] = counts
    for destination, payload in staged:
        publish_bytes(payload, destination)
    report_bytes = canonical_json(report)
    publish_bytes(report_bytes, out_dir / "recovery-report.json")
    return report


def publish_candidates(candidates: Path, manifest_path: Path, collections: list[str] | None, replacements: dict[str, str]) -> dict[str, Any]:
    """Create-only publication of staged candidates next to their source Markdown."""
    manifest = _load_json(manifest_path)
    results: dict[str, list[str]] = {"created": [], "unchanged": [], "replaced": [], "held": []}
    for entry in manifest["collections"]:
        if collections and entry["collection"] not in collections:
            continue
        source_dir = Path(entry["source_directory"])
        staged_dir = candidates / entry["collection"]
        for staged in sorted(staged_dir.glob("HANDOFF-*")):
            if not staged.name.endswith((".meta.yaml", ".graph.json")):
                continue
            payload = staged.read_bytes()
            if staged.name.endswith(".graph.json"):
                validate_document(json.loads(payload))
            else:
                _parse_meta_envelope(payload.decode("utf-8"))
            if not (source_dir / (staged.name.split(".")[0] + ".md")).is_file():
                raise ExportError(f"candidate {staged.name} has no source Markdown")
            destination = source_dir / staged.name
            expected = replacements.get(str(destination))
            try:
                outcome = publish_bytes(payload, destination, create_only=True, expected_sha256=expected)
            except PublicationConflict:
                results["held"].append(str(destination))
                continue
            results[outcome].append(str(destination))
    return results


# --------------------------------------------------------------------------- CLI


def _run(action) -> int:
    try:
        result = action()
    except PublicationConflict as exc:
        print(f"graph export conflict: {exc}", file=sys.stderr)
        return 2
    except ExportError as exc:
        print(f"graph export failed: {exc}", file=sys.stderr)
        return 1
    except Exception as exc:  # noqa: BLE001 - never echo unexpected source-bearing exceptions
        print(f"graph export failed ({type(exc).__name__})", file=sys.stderr)
        return 1
    if result is not None:
        print(result)
    return 0


def main(argv: list[str] | None = None) -> int:
    argv = list(sys.argv[1:] if argv is None else argv)
    if argv and argv[0] == "recover":
        parser = argparse.ArgumentParser(prog="graph_export.py recover", description="Stage historical metadata and v2 sidecar candidates")
        parser.add_argument("--manifest", type=Path, required=True)
        parser.add_argument("--inventory", type=Path, required=True)
        parser.add_argument("--classification", type=Path, required=True)
        parser.add_argument("--out", type=Path, help="destination (default: output/private/ of the current git work tree)")
        args = parser.parse_args(argv[1:])
        return _run(lambda: json.dumps(recover(args.manifest, args.inventory, args.classification, args.out or default_recovery_out())["collections"], sort_keys=True))
    if argv and argv[0] == "publish":
        parser = argparse.ArgumentParser(prog="graph_export.py publish", description="Create-only publication of staged candidates")
        parser.add_argument("--candidates", type=Path, required=True)
        parser.add_argument("--manifest", type=Path, required=True)
        parser.add_argument("--collection", action="append")
        parser.add_argument("--replace", action="append", default=[], help="PATH=SHA256 of a reviewed existing destination")
        args = parser.parse_args(argv[1:])
        replacements = dict(item.rsplit("=", 1) for item in args.replace)
        return _run(lambda: json.dumps(publish_candidates(args.candidates, args.manifest, args.collection, replacements), indent=1, sort_keys=True))

    parser = argparse.ArgumentParser(description="Export a Markdown handoff as a deterministic graph sidecar")
    parser.add_argument("handoff", type=Path, help="path to HANDOFF-NNN.md (or a draft with --check-only)")
    parser.add_argument("--new", action="store_true", help="strict new-save profile: frontmatter required, schema v2")
    parser.add_argument("--check-only", action="store_true", help="validate new-save frontmatter without writing")
    parser.add_argument("--as", dest="target_name", help="filename the draft will be saved as (with --check-only)")
    args = parser.parse_args(argv)
    if args.check_only:
        return _run(lambda: check_new(args.handoff, args.target_name) or "frontmatter valid")

    def export() -> str:
        output, omitted = export_handoff(args.handoff, profile="new" if args.new else None)
        if omitted:
            print(f"warning: omitted {omitted} record(s) containing credential-like content", file=sys.stderr)
        return str(output)

    return _run(export)


if __name__ == "__main__":
    raise SystemExit(main())

README.md

tile.json