CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/coding-policy

General-purpose coding policy for Baruch's AI agents

74

Quality

93%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide
SecuritybySnyk

Medium

Suggest reviewing before use

Overview
Quality
Evals
Security
Files

test_recovery.pyskills/herdr-foreman/tests/

"""Outcome tests for bounded recovery and truthful, reusable attempt history."""

import os
import sys

ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
if ROOT not in sys.path:
    sys.path.insert(0, ROOT)

import copy
import hashlib
import json
import re
import tempfile
import unittest
from pathlib import Path

from foreman import runnable
from foreman.errors import UsageError
from foreman.recovery import (
    ASSIGNMENT_DISPATCH_VERSION, DIAGNOSIS_BOUND_CEILING, DIAGNOSIS_RECORD_VERSION, RECOVERY_STORE_VERSION,
    approach_ceiling, authorize_approach, current_approach,
    abort_pre_send, active_plans, authorize_context, authorize_plan, authorize_refused_dispatch, brief_identity, checkpoint, confirmed_fix,
    diagnose, require_investigation_before_judge, require_judge_mode,
    dispatch_identity, finish_dispatch, fresh_transition, mark_sending,
    _item, migrate_store, prior_dispatch, record_refusal, record_report, refusal_move, register_task, reserve,
    task_statuses, validate_store, validate_work,
)
from foreman.state import add_assignment, empty_state, load_state_checked, save_state

AT = "2026-02-03T10:00:00+00:00"
TASK = "fixture-task"
BASE = "a" * 40
HEAD = "b" * 40
AUTH = {"source": "fixture operator message", "quote": "Approve this task and its stated bounds."}
#: The operator's request a cited checkpoint ruling answers (#400).
REQUEST = {"source": "fixture operator message", "quote": "Ask the judge to rule on this exhaustion."}
WORK = {"base_revision": BASE, "scope": "Correct parser findings", "paths": ["src/parser.py"], "findings": ["F1"]}
PROGRESS = "PROGRESS: two of the three findings closed under the prior remedy.\n"
DIRECTION = "Collect callables at the return-expression level instead of walking the call graph."
VERIFICATION = "The tuple-returning fixture reports both callables, and the existing suite stays green."
OVERRIDE = {"source": "operator message 2026-02-03", "quote": "Try the visitor rewrite; I am authorizing a fresh allowance for it."}


class RecoveryTests(unittest.TestCase):
    def setUp(self):
        self.temp = tempfile.TemporaryDirectory()
        self.addCleanup(self.temp.cleanup)
        self.root = Path(self.temp.name)
        self.state = empty_state()
        self.store = self.state["recovery"]
        self.history = self.state["assignments"]
        register_task(self.store, {"task": TASK, "base_revision": BASE, "scope": WORK["scope"],
                                  "allowed_paths": ["src/*"], "authorization": AUTH}, AT)
        self.judge_report = self.root / "judge.md"
        self.judge_report.write_text("RULING: amend — correct the remaining parser defect\nACTION: Use one canonical parser\n")
        investigation = self.root / "investigation.md"
        investigation.write_text("Reproduction: the quoted case. Cause: two parsers. Experiment: unify them.\n")
        self.investigation = str(investigation)
        self.investigation_sha = hashlib.sha256(investigation.read_bytes()).hexdigest()
        self.review = self.root / "review.md"
        self.review.write_text("Reviewed head: " + HEAD + "\nBlocking finding F1: quoted input is still accepted as a completion signal.\nVERDICT: blocking\n")
        # The assessment a change of direction rests on: what was tried, why it
        # failed, and the experiment that discriminates (#462).
        assessed = self.root / "assessed.md"
        assessed.write_text(
            "FAILED APPROACH: tracking single-return callables through the existing walker.\n"
            "ROOT CAUSE: the walker never visits a callable returned alongside another value.\n"
            "EXPERIMENT: a tuple-returning fixture reproduces the omission in one run.\n")
        self.assessed = str(assessed)

    def exhaust(self):
        for fix in (None, 1, 2, 3, 4, 5):
            add_assignment(self.state, "2026-02-03T09:00:0{}+00:00".format(fix or 0), "developer", "worker", task=TASK, fix_round=fix)

    def seed_checkpoint(self):
        self.exhaust()
        self.consult_investigator("2026-02-03T09:30:00+00:00", "2026-02-03T09:45:00+00:00")
        add_assignment(self.state, AT, "judge", "judge", task=TASK, judge_mode="diagnosis")
        return checkpoint(self.store, self.history, {
            "id": "checkpoint-5", "task": TASK, "defect": "F1 remains open", "previous_attempts": "Five attempts changed parsing and quoting",
            "progress": "Some counterexamples now pass; the quoted case remains red", "change_in_approach": "Use one canonical parser",
            "judge_report": str(self.judge_report), "requested_by": REQUEST,
        }, AT, "judge")

    def approve(self):
        # The operator overrides a remedy; a diagnosis is recorded first.
        self.seed_checkpoint()
        remedy = self.diagnosis("diag-approve", "continue", 1)
        diag = self.run_diagnosis(remedy, "judge")
        return authorize_plan(self.store, self.history, {"id": "plan-1", "task": TASK, "checkpoint": "checkpoint-5",
            "scope": WORK["scope"], "allowed_paths": ["src/*"], "additional_fixes": 2, "authorization": AUTH,
            "supersedes": diag["plan"]}, AT)

    def reservation(self, number, plan: "str | None" = "plan-1"):
        return {"id": f"fix-{number}", "task": TASK, "role": "developer", "agent": "worker", "fix_round": number,
                "fingerprint": "c" * 64, "plan": plan, "work": copy.deepcopy(WORK)}

    def finish(self, number, plan: "str | None" = "plan-1"):
        record = self.reservation(number, plan)
        reserve(self.store, record, AT)
        mark_sending(self.store, record["id"], AT, {"cleared": True})
        add_assignment(self.state, "2026-02-03T11:00:0{}+00:00".format(number), "developer", "worker", task=TASK, fix_round=number)
        finish_dispatch(self.store, record["id"], {"status": "applied", **{
            key: record[key] for key in ("task", "role", "agent", "fix_round")}}, len(self.history) - 1, AT)

    def diagnosis_report(self, remedy, bound, name="diagnosis.md", extra="", assessment=None):
        path = self.root / name
        path.write_text(
            "DIAGNOSIS: the find-rate held flat while every round closed its finding.\n"
            "REMEDY: {} — {}\n"
            "BOUND: {} — one attempt per open finding.\n"
            "ASSESSMENT: {}\n"
            "EVIDENCE: rounds 10-20 and their review bodies.\n"
            "UNVERIFIED: none\n{}".format(remedy, "the named change", bound,
                                           self.investigation if assessment is None else assessment, extra))
        return str(path)

    def judge_after_developer(self, number):
        # The diagnosis gate needs the pinned judge's assignment after the
        # latest developer attempt, and the consultation it rules on before
        # that dispatch.
        self.consult_investigator("2026-02-03T11:30:0{}+00:00".format(number), "2026-02-03T11:45:0{}+00:00".format(number))
        add_assignment(self.state, "2026-02-03T12:00:0{}+00:00".format(number), "judge", "judge", task=TASK, judge_mode="diagnosis")

    def next_checkpoint(self, name):
        # Each exhaustion records its own checkpoint; a plan's first fix is
        # always its checkpoint's round plus one.
        return checkpoint(self.store, self.history, {
            "id": name, "task": TASK, "defect": "F1 remains open", "previous_attempts": "The bounded remedy was spent",
            "progress": "The named change landed; the finding did not close",
            "change_in_approach": "Take the next remedy on the ladder"}, AT, "judge")["id"]

    def consult_investigator(self, at, assessed_at):
        """Record the consultation the judge rules on, assessed before it."""
        add_assignment(self.state, at, "investigator", "worker", task=TASK)
        self.investigator_index = len(self.history) - 1
        self.investigator_assessed_at = assessed_at
        return self.investigator_index

    def investigated(self, index=None, at=None, report=None):
        """The assessed investigator consultation #408 requires."""
        if index is None:
            index = getattr(self, "investigator_index", 0)
        if at is None:
            at = getattr(self, "investigator_assessed_at", "2026-02-03T09:45:00+00:00")
        report = self.investigation if report is None else report
        sha = (self.investigation_sha if report == self.investigation
               else hashlib.sha256(Path(report).read_bytes()).hexdigest() if Path(report).exists()
               else "d" * 64)
        return [{"task": TASK, "role": "investigator", "assignment_index": index, "at": at,
                 "report": report, "report_evidence": {"path": report, "sha256": sha},
                 "source": "report", "acceptance": [{"k": 1, "state": "met", "evidence": "reproduced"}]}]

    def run_diagnosis(self, data, judge="judge", enrolled=None, investigations=None):
        # The fixture's judge dispatch enrolls the report the request cites,
        # unless a case overrides it to prove the binding.
        if enrolled is None:
            enrolled = data.get("judge_report")
        if investigations is None:
            investigations = self.investigated()
        return diagnose(self.store, self.history, data, AT, judge, enrolled, enrolled is not None, investigations)

    def diagnosis(self, name, remedy, bound, checkpoint_id="checkpoint-5", extra="", assessment=None):
        return {"id": name, "task": TASK, "checkpoint": checkpoint_id,
                "judge_report": self.diagnosis_report(remedy, bound, name + ".md", extra, assessment),
                "scope": WORK["scope"], "allowed_paths": ["src/*"]}

    def approach_lines(self, direction=DIRECTION, verification=VERIFICATION):
        return "APPROACH: {}\nVERIFICATION: {}\n".format(direction, verification)

    def approach_diagnosis(self, name, bound, checkpoint_id="checkpoint-5", **lines):
        """A diagnosis that approves a materially different direction."""
        return self.diagnosis(name, "restructure", bound, checkpoint_id,
                              extra=self.approach_lines(**lines), assessment=self.assessed)

    def run_approach_diagnosis(self, data, report=None):
        return self.run_diagnosis(data, "judge",
                                  investigations=self.investigated(report=report or self.assessed))

    def spend(self, first, last, plan: "str | None" = None):
        """Confirm every developer attempt from `first` through `last`."""
        for number in range(first, last + 1):
            self.finish(number, plan)

    def test_a_bounded_remedy_records_the_plan_its_bound_authorizes(self):
        # coding-policy#407: the judge supplies the budget the operator used to.
        self.seed_checkpoint()
        record = self.run_diagnosis(self.diagnosis("diag-1", "continue", 2), "judge")
        self.assertEqual((record["remedy"], record["bound"], record["fix_round"]), ("continue", 2, 5))
        plan = _item_plan = next(row for row in self.store["plans"] if row["id"] == record["plan"])
        self.assertEqual((plan["first_fix"], plan["last_fix"]), (6, 7))
        self.assertEqual(plan["authorization"]["source"], record["judge_evidence"]["path"])
        self.assertIn("REMEDY: continue", plan["authorization"]["quote"])
        # The ordinary allowance machinery enforces it from here.
        validate_work(self.store, self.history, TASK, 6, plan["id"], WORK)
        with self.assertRaisesRegex(UsageError, "outside the approved task or budget"):
            validate_work(self.store, self.history, TASK, 8, plan["id"], WORK)
        self.assertEqual(self.run_diagnosis(self.diagnosis("diag-1", "continue", 2)), record)
        validate_store(self.store, self.history)

    def test_the_remedy_ladder_descends_and_stop_is_terminal(self):
        self.seed_checkpoint()
        self.run_diagnosis(self.diagnosis("diag-1", "continue", 1), "judge")
        # The bound must be spent before the next diagnosis.
        with self.assertRaisesRegex(UsageError, "unspent attempts under plan"):
            self.run_diagnosis(self.diagnosis("diag-2", "restructure", 2), "judge")
        self.finish(6, "diag-1:plan")
        self.judge_after_developer(6)
        second = self.next_checkpoint("checkpoint-6")
        # coding-policy#415: a rung is reissuable once, and only against the
        # progress the re-entry records. A remedy that produced none is not.
        with self.assertRaisesRegex(UsageError, "needs its PROGRESS line"):
            self.run_diagnosis(self.diagnosis("diag-2", "continue", 2, second), "judge")
        again = self.run_diagnosis(self.diagnosis("diag-2", "continue", 2, second, extra=PROGRESS), "judge")
        self.assertTrue(again["reissue"])
        self.finish(7, "diag-2:plan")
        self.finish(8, "diag-2:plan")
        self.judge_after_developer(8)
        third = self.next_checkpoint("checkpoint-7")
        # That rung is now spent; the reissue does not repeat.
        with self.assertRaisesRegex(UsageError, "may not sit above restructure"):
            self.run_diagnosis(self.diagnosis("diag-3", "continue", 1, third, extra=PROGRESS), "judge")
        self.assertFalse(self.run_diagnosis(self.diagnosis("diag-3", "restructure", 1, third), "judge")["reissue"])
        self.finish(9, "diag-3:plan")
        self.judge_after_developer(9)
        fourth = self.next_checkpoint("checkpoint-8")
        self.assertTrue(self.run_diagnosis(self.diagnosis("diag-4", "restructure", 1, fourth, extra=PROGRESS), "judge")["reissue"])
        self.finish(10, "diag-4:plan")
        self.judge_after_developer(10)
        fifth = self.next_checkpoint("checkpoint-9")
        with self.assertRaisesRegex(UsageError, "may not sit above stop"):
            self.run_diagnosis(self.diagnosis("diag-5", "restructure", 1, fifth, extra=PROGRESS), "judge")
        stop = self.run_diagnosis(self.diagnosis("diag-5", "stop", "none", fifth), "judge")
        self.assertEqual((stop["bound"], stop["plan"]), (None, None))
        with self.assertRaisesRegex(UsageError, "terminal"):
            self.run_diagnosis(self.diagnosis("diag-6", "stop", "none", fifth), "judge")
        self.assertEqual(task_statuses(self.store, self.history)[TASK]["status"], "diagnosed_stop")
        validate_store(self.store, self.history)

    def test_a_changed_scope_supersedes_an_unspent_remedy_and_still_descends(self):
        # coding-policy#407: Fix Loops re-enters on a changed scope or an
        # operator override, not only on an exhausted bound.
        self.seed_checkpoint()
        first = self.run_diagnosis(self.diagnosis("diag-1", "continue", 3), "judge")
        with self.assertRaisesRegex(UsageError, "unspent attempts under plan"):
            self.run_diagnosis(self.diagnosis("diag-2", "restructure", 2), "judge")
        with self.assertRaisesRegex(UsageError, "Supersedes must name"):
            self.run_diagnosis({**self.diagnosis("diag-2", "restructure", 2), "supersedes": "plan-nope"}, "judge")
        # Naming the plan is not the change: an early re-entry proves one.
        with self.assertRaisesRegex(UsageError, "needs the change it claims"):
            self.run_diagnosis({**self.diagnosis("diag-2", "restructure", 2), "supersedes": first["plan"]}, "judge")
        narrowed = {**self.diagnosis("diag-2", "restructure", 2), "supersedes": first["plan"], "allowed_paths": ["src/parser.py"]}
        # The ladder still descends: a supersession is a diagnosis like any other.
        with self.assertRaisesRegex(UsageError, "needs its PROGRESS line"):
            self.run_diagnosis({**narrowed, **self.diagnosis("diag-2", "continue", 2), "supersedes": first["plan"], "allowed_paths": ["src/parser.py"]}, "judge")
        # An operator override stands in for a changed scope.
        override = self.run_diagnosis({**self.diagnosis("diag-override", "restructure", 2),
                                       "supersedes": first["plan"], "authorization": AUTH})
        self.assertEqual(override["supersedes"], first["plan"])
        second = override
        self.assertEqual(second["supersedes"], first["plan"])
        plans = {row["id"]: row for row in self.store["plans"]}
        self.assertIn(first["plan"], plans)
        self.assertEqual(plans[second["plan"]]["supersedes"], first["plan"])
        with self.assertRaisesRegex(UsageError, "needs the change it claims"):
            self.run_diagnosis({**self.diagnosis("diag-late", "stop", "none"), "supersedes": second["plan"]})
        self.assertNotIn(first["plan"], [row["id"] for row in active_plans(self.store)])
        with self.assertRaisesRegex(UsageError, "superseded"):
            validate_work(self.store, self.history, TASK, 6, first["plan"], WORK)
        validate_work(self.store, self.history, TASK, 6, second["plan"], WORK)
        validate_store(self.store, self.history)

    def test_a_diagnosis_needs_an_exhausted_budget_a_checkpoint_and_the_pinned_judge(self):
        with self.assertRaisesRegex(UsageError, "requires id, task, checkpoint"):
            self.run_diagnosis({"id": "d", "task": TASK}, "judge")
        self.seed_checkpoint()
        for remedy, bound, message in (("continue", "0", "positive BOUND"), ("continue", "none", "positive BOUND"),
                                       ("sideways", "1", "must carry DIAGNOSIS")):
            with self.assertRaisesRegex(UsageError, message):
                self.run_diagnosis(self.diagnosis("diag-bad", remedy, bound), "judge")
        with self.assertRaisesRegex(UsageError, "pinned judge's completed assignment"):
            self.run_diagnosis(self.diagnosis("diag-1", "continue", 1), "someone-else")
        incomplete = self.root / "incomplete.md"
        incomplete.write_text("DIAGNOSIS: x\nREMEDY: continue — more\nBOUND: 2\n")
        with self.assertRaisesRegex(UsageError, "must carry DIAGNOSIS"):
            self.run_diagnosis({**self.diagnosis("diag-2", "continue", 1), "judge_report": str(incomplete)}, "judge")

    def test_a_stop_remedy_ends_implementation_on_the_task(self):
        # coding-policy#407: `stop` is terminal, so no allowance survives it.
        self.seed_checkpoint()
        first = self.run_diagnosis(self.diagnosis("diag-1", "continue", 3), "judge")
        validate_work(self.store, self.history, TASK, 6, first["plan"], WORK)
        self.run_diagnosis({**self.diagnosis("diag-2", "stop", "none"), "supersedes": first["plan"], "authorization": AUTH})
        for plan in (first["plan"], None):
            with self.assertRaisesRegex(UsageError, "terminal"):
                validate_work(self.store, self.history, TASK, 6, plan, WORK)
        with self.assertRaisesRegex(UsageError, "terminal"):
            validate_work(self.store, self.history, TASK, 3, None, None)
        # What `stop` ships still gets reviewed, tested and released.
        validate_work(self.store, self.history, TASK, None, None, None, implementation=False)
        # Only the operator overrides a ruling, and a plan over the stop is how.
        self.judge_after_developer(6)
        override = authorize_plan(self.store, self.history, {"id": "override", "task": TASK, "checkpoint": "checkpoint-5",
            "scope": WORK["scope"], "allowed_paths": ["src/*"], "additional_fixes": 1, "authorization": AUTH}, AT)
        self.assertEqual(override["first_fix"], 6)
        validate_work(self.store, self.history, TASK, 6, "override", WORK)
        with self.assertRaisesRegex(UsageError, "overridden by plan override"):
            validate_work(self.store, self.history, TASK, 6, None, WORK)
        validate_work(self.store, self.history, TASK, 6, "override", WORK, implementation=False)
        # The status follows the override, as validate_work does.
        self.assertEqual(task_statuses(self.store, self.history)[TASK]["status"], "within_authorized_budget")
        validate_store(self.store, self.history)

    def test_a_report_must_be_the_one_supervision_enrolled(self):
        self.seed_checkpoint()
        request = self.diagnosis("diag-1", "continue", 2)
        with self.assertRaisesRegex(UsageError, "not the one supervision enrolled"):
            self.run_diagnosis(request, enrolled=str(self.root / "elsewhere.md"))
        record = self.run_diagnosis(request)
        self.assertEqual(record["remedy"], "continue")

    def test_a_remedy_names_what_it_means(self):
        self.seed_checkpoint()
        bare = self.root / "bare.md"
        bare.write_text("DIAGNOSIS: flat find-rate\nREMEDY: restructure\nBOUND: 2 — one per finding\n"
                        "ASSESSMENT: " + self.investigation + "\nEVIDENCE: rounds 10-20\nUNVERIFIED: none\n")
        with self.assertRaisesRegex(UsageError, "names its remedy and what it means"):
            self.run_diagnosis({**self.diagnosis("diag-1", "continue", 2), "judge_report": str(bare)}, "judge")

    def test_a_bound_states_developer_attempts_within_a_ceiling(self):
        # coding-policy#415: the operator's budget was a human pricing the
        # spend; the replacement states its units and justifies the number.
        self.seed_checkpoint()
        with self.assertRaisesRegex(UsageError, "exceeds the 5-attempt ceiling"):
            self.run_diagnosis(self.diagnosis("diag-1", "continue", DIAGNOSIS_BOUND_CEILING + 1), "judge")
        bare = self.root / "unjustified.md"
        bare.write_text("DIAGNOSIS: flat find-rate\nREMEDY: continue — two more rounds\nBOUND: 2\n"
                        "ASSESSMENT: " + self.investigation + "\nEVIDENCE: rounds 10-20\nUNVERIFIED: none\n")
        with self.assertRaisesRegex(UsageError, "justifies the number"):
            self.run_diagnosis({**self.diagnosis("diag-1", "continue", 2), "judge_report": str(bare)}, "judge")
        record = self.run_diagnosis(self.diagnosis("diag-1", "continue", DIAGNOSIS_BOUND_CEILING), "judge")
        self.assertEqual(record["bound"], DIAGNOSIS_BOUND_CEILING)
        validate_store(self.store, self.history)

    def _report_without_assessment(self):
        path = self.root / "noassess.md"
        path.write_text("DIAGNOSIS: flat find-rate\nREMEDY: continue — two more rounds\n"
                        "BOUND: 2 — one per finding\nEVIDENCE: rounds 10-20\nUNVERIFIED: none\n")
        return str(path)

    def test_a_changed_investigator_report_refuses_the_diagnosis(self):
        # rules/stateful-artifacts.md Hints, Not Authority: the saved receipt
        # is a last-seen snapshot, so a rewritten report authorizes nothing.
        self.seed_checkpoint()
        Path(self.investigation).write_text("Rewritten after the assessment.\n")
        with self.assertRaisesRegex(UsageError, "changed since its assessment"):
            self.run_diagnosis(self.diagnosis("diag-1", "continue", 2), "judge")
        Path(self.investigation).unlink()
        with self.assertRaisesRegex(UsageError, "Cannot read evidence"):
            self.run_diagnosis(self.diagnosis("diag-2", "continue", 2), "judge")

    def test_a_diagnosis_cites_the_assessment_it_ruled_on(self):
        # coding-policy#415: the ordering gate proves an assessment exists; the
        # citation proves this diagnosis consumed it.
        self.seed_checkpoint()
        with self.assertRaisesRegex(UsageError, "must carry DIAGNOSIS"):
            self.run_diagnosis({**self.diagnosis("diag-1", "continue", 2),
                                "judge_report": self._report_without_assessment()}, "judge")
        with self.assertRaisesRegex(UsageError, "not an accepted investigator report"):
            self.run_diagnosis(self.diagnosis("diag-1", "continue", 2, assessment=str(self.root / "other.md")), "judge")
        record = self.run_diagnosis(self.diagnosis("diag-1", "continue", 2), "judge")
        self.assertEqual(record["investigator_report"]["report"], self.investigation)
        self.assertEqual(record["investigator_report"]["evidence"],
                         {"path": self.investigation, "sha256": self.investigation_sha})
        validate_store(self.store, self.history)

    def test_an_older_diagnosis_row_migrates_to_the_recorded_shape(self):
        # rules/stateful-artifacts.md: the owner upgrades, and a version-1 row
        # carried neither field, so its defaults are the facts it already held.
        self.seed_checkpoint()
        self.run_diagnosis(self.diagnosis("diag-1", "continue", 2), "judge")
        older = copy.deepcopy(self.store)
        row = older["diagnoses"][0]
        row.update(schema_version=1)
        del row["reissue"], row["investigator_report"], row["approach"], row["approach_change"]
        # Both record kinds migrate in one pass; neither short-circuits the
        # other (rules/stateful-artifacts.md Migration Policy).
        older["checkpoints"][0]["schema_version"] = 1
        older["checkpoints"][0].pop("requested_by", None)
        self.assertTrue(migrate_store(older))
        self.assertEqual(older["checkpoints"][0]["schema_version"], 2)
        self.assertEqual(older["diagnoses"][0]["schema_version"], DIAGNOSIS_RECORD_VERSION)
        self.assertIs(older["diagnoses"][0]["reissue"], False)
        self.assertIsNone(older["diagnoses"][0]["investigator_report"])
        # coding-policy#462: a pre-approach diagnosis ruled on the initial
        # direction and approved no transition away from it.
        self.assertIsNone(older["diagnoses"][0]["approach"])
        self.assertIsNone(older["diagnoses"][0]["approach_change"])
        validate_store(older, self.history)

    def test_an_older_diagnosis_carrying_newer_fields_is_refused(self):
        # A version-1 row predates both fields, so one already carrying either
        # is unowned newer data; stamping it would let the value through.
        self.seed_checkpoint()
        self.run_diagnosis(self.diagnosis("diag-1", "continue", 2), "judge")
        for field, value in (("reissue", True), ("investigator_report", None),
                             ("approach", "approach-1"), ("approach_change", None)):
            corrupt = copy.deepcopy(self.store)
            row = corrupt["diagnoses"][0]
            row.update(schema_version=1)
            del row["reissue"], row["investigator_report"], row["approach"], row["approach_change"]
            row[field] = value
            with self.assertRaisesRegex(UsageError, "newer recorded fields"):
                migrate_store(corrupt)

    def test_a_judge_seat_is_not_spent_before_the_assessment_exists(self):
        # coding-policy#408: the gate guards the dispatch, not only the record,
        # so the most expensive seat is never spent on an uninvestigated loop.
        # Inside the allowance an ordinary dispute reaches the judge freely.
        add_assignment(self.state, "2026-02-03T09:00:00+00:00", "developer", "worker", task=TASK)
        self.assertIsNone(require_investigation_before_judge(self.store, self.history, TASK, []))
        self.assertIsNone(require_investigation_before_judge(self.store, self.history, "unknown-task", []))
        self.exhaust()
        with self.assertRaisesRegex(UsageError, "consult the investigator"):
            require_investigation_before_judge(self.store, self.history, TASK, [])
        index = self.consult_investigator("2026-02-03T09:30:00+00:00", "2026-02-03T09:45:00+00:00")
        self.assertIsNone(require_investigation_before_judge(self.store, self.history, TASK, self.investigated(index=index)))

    def test_a_declared_diagnosis_on_a_stopped_task_refuses_before_dispatch(self):
        # coding-policy#425: with the mode declared, the bound #400 could not
        # enforce is enforceable — before the expensive round, not after its
        # report exists.
        self.seed_checkpoint()
        self.run_diagnosis(self.diagnosis("diag-stop", "stop", None), "judge")
        with self.assertRaisesRegex(UsageError, "diagnosed `stop`"):
            require_investigation_before_judge(self.store, self.history, TASK, self.investigated(),
                                               mode="diagnosis")
        # An adjudication on the same task is untouched: `stop` ends
        # implementation and the ladder, never a contested verdict's ruling.
        self.assertIsNone(require_investigation_before_judge(self.store, self.history, TASK,
                                                            self.investigated(), mode="adjudication"))
        # The operator's plan over the stop lifts the refusal.
        authorize_plan(self.store, self.history, {"id": "over-stop", "task": TASK, "checkpoint": "checkpoint-5",
            "scope": WORK["scope"], "allowed_paths": ["src/*"], "additional_fixes": 2, "authorization": AUTH}, AT)
        self.assertIsNone(require_investigation_before_judge(self.store, self.history, TASK,
                                                            self.investigated(), mode="diagnosis"))

    def test_an_adjudication_carries_no_assessment_requirement(self):
        # The assessment gate belongs to diagnosis; an adjudication rules on a
        # contested verdict and needs none.
        self.exhaust()
        with self.assertRaisesRegex(UsageError, "consult the investigator"):
            require_investigation_before_judge(self.store, self.history, TASK, [], mode="diagnosis")
        self.assertIsNone(require_investigation_before_judge(self.store, self.history, TASK, [],
                                                             mode="adjudication"))

    def test_an_unknown_mode_is_refused_rather_than_defaulted(self):
        self.seed_checkpoint()
        with self.assertRaisesRegex(UsageError, "declares its mode"):
            require_investigation_before_judge(self.store, self.history, TASK, self.investigated(),
                                               mode="whatever")
        with self.assertRaisesRegex(UsageError, "declares what it is for"):
            require_judge_mode(None)
        self.assertEqual(require_judge_mode("diagnosis"), "diagnosis")

    def test_a_stopped_task_still_reaches_the_judge_for_an_adjudication(self):
        # coding-policy#400: a pre-dispatch refusal was considered for a
        # `stop`ped task and dropped. `stop` ends implementation and the
        # diagnosis ladder, never adjudication — this gate sees no judge mode,
        # so refusing here would refuse a contested verdict's ruling too.
        self.seed_checkpoint()
        self.run_diagnosis(self.diagnosis("diag-stop", "stop", None), "judge")
        self.assertIsNone(require_investigation_before_judge(self.store, self.history, TASK, self.investigated()))

    def test_a_diagnosis_rules_on_a_prepared_causal_assessment(self):
        # coding-policy#408: the investigator's profile is written for repeated
        # unsuccessful fixes, and the judge is the more expensive seat.
        self.seed_checkpoint()
        seeded, seeded_at = self.investigator_index, self.investigator_assessed_at
        request = self.diagnosis("diag-1", "continue", 2)
        with self.assertRaisesRegex(UsageError, "prepared causal assessment"):
            self.run_diagnosis(request, investigations=[])
        # Another task's assessment, and one predating the latest attempt, are
        # not this loop's evidence.
        with self.assertRaisesRegex(UsageError, "prepared causal assessment"):
            self.run_diagnosis(request, investigations=[{"task": "another", "role": "investigator", "assignment_index": 0,
                                                                           "source": "report", "acceptance": []}])
        with self.assertRaisesRegex(UsageError, "prepared causal assessment"):
            self.run_diagnosis(request, investigations=[{"task": TASK, "role": "architect", "assignment_index": 0,
                                                                           "source": "report", "acceptance": []}])
        with self.assertRaisesRegex(UsageError, "prepared causal assessment"):
            self.run_diagnosis(request, investigations=self.investigated(index=0))
        # A consultation delivered after the judge dispatch is not what it read.
        late = self.consult_investigator("2026-02-03T20:00:00+00:00", "2026-02-03T20:30:00+00:00")
        with self.assertRaisesRegex(UsageError, "assessed before the judge dispatch"):
            self.run_diagnosis(request, investigations=self.investigated(index=late))
        # Dispatched early, assessed late: the judge still did not read it.
        with self.assertRaisesRegex(UsageError, "assessed before the judge dispatch"):
            self.run_diagnosis(request, investigations=self.investigated(index=seeded, at="2026-02-03T23:00:00+00:00"))
        # An unmet criterion, or a migrated foreman assessment, is no accepted
        # investigation (#625).
        accepted = self.investigated(index=seeded, at=seeded_at)
        for change in ({"acceptance": [{"k": 1, "state": "unmet", "evidence": "no reproduction"}]},
                       {"source": "foreman_assessment", "acceptance": None},
                       {"verdict": "blocking"}):
            with self.subTest(change=change), self.assertRaisesRegex(UsageError, "prepared causal assessment"):
                self.run_diagnosis(request, investigations=[{**accepted[0], **change}])
        self.assertEqual(self.run_diagnosis(request, investigations=accepted)["remedy"], "continue")

    def test_an_adjudication_report_is_not_a_diagnosis(self):
        # coding-policy#407: RULING and ACTION belong to adjudication; a mixed
        # report could carry a blocked ruling and still grant attempts.
        self.seed_checkpoint()
        mixed = self.root / "mixed.md"
        mixed.write_text("RULING: blocked — which boundary ships?\nACTION: ask the operator\n"
                         "DIAGNOSIS: flat find-rate\nREMEDY: continue — two more rounds\n"
                         "BOUND: 2 — one per finding\nASSESSMENT: " + self.investigation
                         + "\nEVIDENCE: rounds 1-5\nUNVERIFIED: none\n")
        with self.assertRaisesRegex(UsageError, "adjudication's RULING or ACTION"):
            self.run_diagnosis({**self.diagnosis("diag-1", "continue", 2), "judge_report": str(mixed)})
        self.assertEqual(self.store["diagnoses"], [])
        self.assertEqual([row for row in self.store["plans"]], [])

    def test_corrupt_diagnoses_refuse_the_ledger(self):
        self.seed_checkpoint()
        self.run_diagnosis(self.diagnosis("diag-1", "continue", 2), "judge")
        for mutate in (
            lambda row: row.update(remedy="sideways"),
            lambda row: row.update(bound=None),
            lambda row: row.update(plan=None),
            lambda row: row.update(judge_evidence="not-a-receipt"),
            lambda row: row.update(reissue=True),
            lambda row: row.update(investigator_report={"schema_version": 2, "report": "relative.md",
                                                        "evidence": {"path": "relative.md", "sha256": "d" * 64}}),
            lambda row: row.update(investigator_report={"schema_version": 2, "report": "/tmp/other.md",
                                                        "evidence": row["investigator_report"]["evidence"]}),
        ):
            corrupt = copy.deepcopy(self.store)
            mutate(corrupt["diagnoses"][0])
            with self.assertRaises(UsageError):
                validate_store(corrupt, self.history)
        corrupt = copy.deepcopy(self.store)
        corrupt["diagnoses"].append({**corrupt["diagnoses"][0], "id": "diag-back", "remedy": "continue"})
        with self.assertRaisesRegex(UsageError, "must move down the remedy ladder"):
            validate_store(corrupt, self.history)
        for mutate in (lambda row: row.update(fix_round=4),
                       lambda row: row.update(base_revision="f" * 40)):
            corrupt = copy.deepcopy(self.store)
            mutate(corrupt["diagnoses"][0])
            with self.assertRaisesRegex(UsageError, "another task, base or fix round"):
                validate_store(corrupt, self.history)

    def test_an_evidenced_change_of_approach_starts_a_fresh_allowance(self):
        # coding-policy#462: the bound exists to stop repeated attempts at a
        # failing approach. A different direction is not that repetition, so it
        # gets its own allowance while the cumulative history stays intact.
        self.seed_checkpoint()
        record = self.run_approach_diagnosis(self.approach_diagnosis("diag-1", 5))
        self.assertEqual(record["approach_change"], "diag-1:approach")
        # The allowance is the approach's, so no extra-correction plan is minted.
        self.assertIsNone(record["plan"])
        self.assertIsNone(record["approach"])
        approach = self.store["approaches"][-1]
        self.assertEqual(current_approach(self.store, TASK), approach)
        self.assertEqual((approach["from_fix"], approach["allowance"], approach["origin"], approach["direction"]),
                         (5, 5, "diagnosis", DIRECTION))
        self.assertEqual(approach_ceiling(self.store, TASK), 10)
        status = task_statuses(self.store, self.history)[TASK]
        self.assertEqual((status["status"], status["confirmed_fixes"], status["approach_attempts"],
                          status["approach_allowance"], status["remaining_fixes"]),
                         ("within_authorized_budget", 5, 0, 5, 5))
        # Cumulative attempt 6 is approach attempt 1: no plan, no operator prompt.
        self.assertIsNone(validate_work(self.store, self.history, TASK, 6, None, None))
        with self.assertRaisesRegex(UsageError, "cannot relabel a fix inside the current approach"):
            validate_work(self.store, self.history, TASK, 6, "diag-1:plan", WORK)
        self.finish(6, None)
        # Review feedback returns to the developer inside the fresh allowance.
        self.assertIsNone(validate_work(self.store, self.history, TASK, 7, None, None))
        self.spend(7, 10)
        self.assertEqual(task_statuses(self.store, self.history)[TASK]["approach_attempts"], 5)
        with self.assertRaisesRegex(UsageError, "correction allowance is exhausted"):
            validate_work(self.store, self.history, TASK, 11, None, None)
        validate_store(self.store, self.history)

    def test_a_diagnosis_cannot_cite_another_tasks_approach(self):
        # coding-policy#468: validation confirmed the reference resolved to SOME
        # approach, never that it belonged to the citing task. `current_diagnoses`
        # filters the ladder by task, so a hand-edited cross-task reference
        # dropped the row out of its own ladder and the per-approach `stop` and
        # rung checks skipped it entirely.
        other = "another-task"
        register_task(self.store, {"task": other, "base_revision": BASE, "scope": WORK["scope"],
                                   "allowed_paths": ["src/*"], "authorization": AUTH}, AT)
        self.seed_checkpoint()
        self.run_approach_diagnosis(self.approach_diagnosis("diag-1", 5))
        self.spend(6, 10)
        self.judge_after_developer(10)
        second = self.next_checkpoint("checkpoint-6")
        later = self.run_diagnosis(self.diagnosis("diag-2", "continue", 1, second), "judge")
        self.assertEqual(later["approach"], "diag-1:approach")
        validate_store(self.store, self.history)
        # The second task's own checkpoint and approach, each internally
        # consistent, so the only thing left to object to is which task the
        # citing diagnosis belongs to.
        origin_checkpoint = _item(self.store["checkpoints"], self.store["approaches"][-1]["checkpoint"], "checkpoint")
        self.store["checkpoints"].append(dict(origin_checkpoint, id="foreign-checkpoint", task=other))
        foreign = dict(self.store["approaches"][-1], id="foreign-approach", task=other,
                       checkpoint="foreign-checkpoint", origin="operator", diagnosis=None,
                       judge_evidence=None, investigator_report=None,
                       authorization=OVERRIDE, supersedes=None)
        self.store["approaches"].append(foreign)
        row = next(item for item in self.store["diagnoses"] if item["id"] == "diag-2")
        row["approach"] = "foreign-approach"
        with self.assertRaisesRegex(UsageError, "belongs to task " + other):
            validate_store(self.store, self.history)
        row["approach"] = "diag-1:approach"
        validate_store(self.store, self.history)

    def test_a_diagnosis_cannot_approve_another_tasks_approach(self):
        # coding-policy#497: the cross-task check covers `approach_change` as
        # well as `approach`. A diagnosed approach cites its diagnosis back, so
        # that back-reference already rejects most cross-task edits; the
        # diagnosis-side check is the only guard once the task's own approach
        # carries operator provenance and cites no diagnosis at all.
        other = "another-task"
        register_task(self.store, {"task": other, "base_revision": BASE, "scope": WORK["scope"],
                                   "allowed_paths": ["src/*"], "authorization": AUTH}, AT)
        self.seed_checkpoint()
        self.run_approach_diagnosis(self.approach_diagnosis("diag-1", 5))
        own = self.store["approaches"][-1]
        own.update(origin="operator", diagnosis=None, judge_evidence=None,
                   investigator_report=None, authorization=OVERRIDE)
        validate_store(self.store, self.history)
        origin_checkpoint = _item(self.store["checkpoints"], own["checkpoint"], "checkpoint")
        self.store["checkpoints"].append(dict(origin_checkpoint, id="foreign-checkpoint", task=other))
        self.store["approaches"].append(dict(own, id="foreign-approach", task=other,
                                             checkpoint="foreign-checkpoint"))
        validate_store(self.store, self.history)
        row = next(item for item in self.store["diagnoses"] if item["id"] == "diag-1")
        row["approach_change"] = "foreign-approach"
        with self.assertRaisesRegex(UsageError, "cites approach foreign-approach, which belongs to task " + other):
            validate_store(self.store, self.history)
        row["approach_change"] = own["id"]
        validate_store(self.store, self.history)

    def test_a_new_approach_starts_its_ladder_rather_than_inheriting_one(self):
        # An earlier approach's exhausted rungs never force a newly approved
        # direction down a lifetime ladder to permanent stop (#462).
        self.seed_checkpoint()
        self.run_diagnosis(self.diagnosis("diag-1", "continue", 1), "judge")
        self.finish(6, "diag-1:plan")
        self.judge_after_developer(6)
        second = self.next_checkpoint("checkpoint-6")
        self.run_approach_diagnosis(self.approach_diagnosis("diag-2", 4, second))
        self.spend(7, 10)
        self.judge_after_developer(10)
        third = self.next_checkpoint("checkpoint-7")
        # `continue` is available again although the task's lifetime already
        # spent that rung under the previous approach.
        fresh = self.run_diagnosis(self.diagnosis("diag-3", "continue", 1, third), "judge")
        self.assertEqual((fresh["remedy"], fresh["reissue"], fresh["approach"]), ("continue", False, "diag-2:approach"))
        self.assertEqual(fresh["fix_round"], 10)
        validate_store(self.store, self.history)

    def test_a_reset_needs_its_investigator_and_judge_evidence(self):
        self.seed_checkpoint()
        # The assessment must explain the failed approach, its cause and the
        # discriminating experiment; a consultation alone resets nothing.
        with self.assertRaisesRegex(UsageError, "FAILED APPROACH"):
            self.run_diagnosis(self.diagnosis("diag-1", "restructure", 3, extra=self.approach_lines()), "judge")
        # Both judge lines are required; one alone is an incomplete transition.
        for extra in ("APPROACH: {}\n".format(DIRECTION), "VERIFICATION: {}\n".format(VERIFICATION)):
            with self.assertRaisesRegex(UsageError, "carries both APPROACH"):
                self.run_approach_diagnosis({**self.approach_diagnosis("diag-1", 3), "judge_report":
                    self.diagnosis_report("restructure", 3, "partial.md", extra, self.assessed)})
        # `stop` ends implementation; it approves no direction.
        with self.assertRaisesRegex(UsageError, "approves no new direction"):
            self.run_approach_diagnosis({**self.approach_diagnosis("diag-1", 3), "judge_report":
                self.diagnosis_report("stop", "none", "stopped.md", self.approach_lines(), self.assessed)})
        self.assertEqual(self.store["approaches"], [])

    def test_the_same_direction_and_a_replayed_transition_grant_no_second_allowance(self):
        self.seed_checkpoint()
        first = self.run_approach_diagnosis(self.approach_diagnosis("diag-1", 3))
        # A retry of the recorded transition returns it; it never mints another.
        self.assertEqual(self.run_approach_diagnosis(self.approach_diagnosis("diag-1", 3)), first)
        self.assertEqual(len(self.store["approaches"]), 1)
        self.spend(6, 8)
        self.judge_after_developer(8)
        second = self.next_checkpoint("checkpoint-8")
        # Another attempt at the same direction spends the existing allowance.
        with self.assertRaisesRegex(UsageError, "already recorded this direction"):
            self.run_approach_diagnosis(self.approach_diagnosis("diag-2", 3, second))
        with self.assertRaisesRegex(UsageError, "already recorded this direction"):
            authorize_approach(self.store, self.history, {
                "id": "approach-2", "task": TASK, "checkpoint": second, "direction": DIRECTION.upper(),
                "verification": VERIFICATION, "allowance": 2, "authorization": OVERRIDE}, AT)
        self.assertEqual(len(self.store["approaches"]), 1)
        validate_store(self.store, self.history)

    def test_an_overridden_stop_is_recordable_plannable_and_diagnosable(self):
        # coding-policy#462: `authorize_plan` needed a diagnosis at the current
        # count while `diagnose` refused one on a stopped ladder, so an
        # authorized continuation had nowhere to land.
        self.seed_checkpoint()
        self.run_diagnosis(self.diagnosis("diag-1", "stop", "none"), "judge")
        authorize_plan(self.store, self.history, {"id": "plan-1", "task": TASK, "checkpoint": "checkpoint-5",
            "scope": WORK["scope"], "allowed_paths": ["src/*"], "additional_fixes": 1, "authorization": AUTH}, AT)
        self.finish(6, "plan-1")
        self.judge_after_developer(6)
        second = self.next_checkpoint("checkpoint-6")
        with self.assertRaisesRegex(UsageError, "terminal"):
            self.run_diagnosis(self.diagnosis("diag-2", "continue", 1, second), "judge")
        with self.assertRaisesRegex(UsageError, "no diagnosis at fix round 6"):
            authorize_plan(self.store, self.history, {"id": "plan-2", "task": TASK, "checkpoint": second,
                "scope": WORK["scope"], "allowed_paths": ["src/*"], "additional_fixes": 1, "authorization": OVERRIDE}, AT)
        approach = authorize_approach(self.store, self.history, {
            "id": "approach-1", "task": TASK, "checkpoint": second, "direction": DIRECTION,
            "verification": VERIFICATION, "allowance": 3, "authorization": OVERRIDE}, AT)
        self.assertEqual((approach["from_fix"], approach["allowance"], approach["origin"]), (6, 3, "operator"))
        status = task_statuses(self.store, self.history)[TASK]
        self.assertEqual((status["status"], status["confirmed_fixes"], status["approach_attempts"], status["approach"]),
                         ("within_authorized_budget", 6, 0, "approach-1"))
        self.assertIsNone(validate_work(self.store, self.history, TASK, 7, None, None))
        self.spend(7, 9)
        self.judge_after_developer(9)
        third = self.next_checkpoint("checkpoint-9")
        # The diagnosis path reopens with the new approach's own fresh ladder.
        self.assertIsNone(require_investigation_before_judge(
            self.store, self.history, TASK, self.investigated(), "diagnosis"))
        later = self.run_diagnosis(self.diagnosis("diag-2", "continue", 1, third), "judge")
        self.assertEqual((later["remedy"], later["approach"]), ("continue", "approach-1"))
        validate_store(self.store, self.history)

    def test_an_operator_approach_needs_a_current_exhaustion_and_its_own_authority(self):
        self.seed_checkpoint()
        grant = {"id": "approach-1", "task": TASK, "checkpoint": "checkpoint-5", "direction": DIRECTION,
                 "verification": VERIFICATION, "allowance": 2, "authorization": OVERRIDE}
        with self.assertRaisesRegex(UsageError, "requires id, task, checkpoint"):
            authorize_approach(self.store, self.history, {**grant, "extra": 1}, AT)
        with self.assertRaisesRegex(UsageError, "exceeds the 5-attempt ceiling"):
            authorize_approach(self.store, self.history, {**grant, "allowance": 6}, AT)
        # With no plan holding attempts there is nothing to retire.
        with self.assertRaisesRegex(UsageError, "must name this task's current unexhausted plan"):
            authorize_approach(self.store, self.history, {**grant, "supersedes": "plan-1"}, AT)
        reservation = self.reservation(6, None)
        reserve(self.store, reservation, AT)
        with self.assertRaisesRegex(UsageError, "dispatch outcome is still unknown"):
            authorize_approach(self.store, self.history, grant, AT)
        abort_pre_send(self.store, reservation["id"], AT, "reconciled before approval")
        first = authorize_approach(self.store, self.history, grant, AT)
        self.assertEqual(authorize_approach(self.store, self.history, grant, AT), first)
        with self.assertRaisesRegex(UsageError, "already records a different direction"):
            authorize_approach(self.store, self.history, {**grant, "allowance": 3}, AT)
        self.spend(6, 7)
        self.judge_after_developer(7)
        second = self.next_checkpoint("checkpoint-7")
        # Old authority never carries a new transition.
        with self.assertRaisesRegex(UsageError, "already approved an earlier approach"):
            authorize_approach(self.store, self.history, {**grant, "id": "approach-2", "checkpoint": second,
                "direction": "Rewrite the walker as a visitor over return expressions."}, AT)
        fresh = authorize_approach(self.store, self.history, {**grant, "id": "approach-2", "checkpoint": second,
            "direction": "Rewrite the walker as a visitor over return expressions.",
            "authorization": {"source": "operator message 2026-02-04", "quote": "Take the visitor rewrite next."}}, AT)
        self.assertEqual(fresh["from_fix"], 7)
        # An unexhausted approach continues; it does not start another.
        with self.assertRaisesRegex(UsageError, "allowance is not exhausted"):
            authorize_approach(self.store, self.history, {**grant, "id": "approach-3", "checkpoint": second,
                "direction": "Parse the module twice and diff the callable sets.",
                "authorization": {"source": "operator message 2026-02-05", "quote": "And try the double parse."}}, AT)
        validate_store(self.store, self.history)

    def test_a_new_approach_retires_the_plan_the_old_one_bought(self):
        # A plan's fix range is cumulative and outlives the direction it was
        # bought for: left active, a 5-attempt plan authorized fixes 7-10 under
        # an approach whose own allowance was 1 (policy reviewer on #467).
        self.seed_checkpoint()
        diagnosed = self.run_diagnosis(self.diagnosis("diag-1", "continue", 5), "judge")
        grant = {"id": "approach-1", "task": TASK, "checkpoint": "checkpoint-5", "direction": DIRECTION,
                 "verification": VERIFICATION, "allowance": 1, "authorization": OVERRIDE}
        with self.assertRaisesRegex(UsageError, "still holds unspent attempts"):
            authorize_approach(self.store, self.history, grant, AT)
        # Naming something other than the plan that holds the attempts retires
        # nothing, so the same refusal stands.
        with self.assertRaisesRegex(UsageError, "still holds unspent attempts"):
            authorize_approach(self.store, self.history, {**grant, "supersedes": "no-such-plan"}, AT)
        approach = authorize_approach(self.store, self.history, {**grant, "supersedes": diagnosed["plan"]}, AT)
        self.assertEqual(approach["supersedes"], diagnosed["plan"])
        self.assertEqual(active_plans(self.store), [])
        self.finish(6, None)
        # The retired plan's remaining range no longer buys an attempt.
        with self.assertRaisesRegex(UsageError, "correction allowance is exhausted"):
            validate_work(self.store, self.history, TASK, 7, None, None)
        with self.assertRaisesRegex(UsageError, "was superseded"):
            validate_work(self.store, self.history, TASK, 7, diagnosed["plan"], WORK)
        status = task_statuses(self.store, self.history)[TASK]
        self.assertEqual((status["status"], status["plan"], status["remaining_fixes"],
                          status["approach_attempts"], status["approach_allowance"]),
                         ("checkpoint_required", None, 0, 1, 1))
        validate_store(self.store, self.history)

    def test_a_ledger_without_approaches_reads_as_its_own_initial_approach(self):
        # Existing history migrates with no transition invented and no attempt
        # renumbered (rules/stateful-artifacts.md Migration Policy).
        self.seed_checkpoint()
        self.assertEqual(self.store["approaches"], [])
        self.assertIsNone(current_approach(self.store, TASK))
        status = task_statuses(self.store, self.history)[TASK]
        self.assertEqual((status["approach"], status["approach_attempts"], status["approach_allowance"],
                          status["confirmed_fixes"]), (None, 5, 5, 5))
        self.assertEqual(approach_ceiling(self.store, TASK), 5)

    def test_a_diagnosis_plan_is_matched_on_every_field_the_remedy_derives(self):
        # coding-policy#412: the check compared task, bound and authorization
        # source alone, so a same-task plan carrying another scope or path set
        # could be attached and `validate_work` would then enforce ITS budget
        # and scope against this diagnosis.
        self.seed_checkpoint()
        self.run_diagnosis(self.diagnosis("diag-1", "continue", 2), "judge")
        for field, value in (("scope", "Something else entirely"),
                             ("allowed_paths", ["docs/*"])):
            with self.subTest(field=field):
                corrupt = copy.deepcopy(self.store)
                plan = next(row for row in corrupt["plans"] if row["id"] == corrupt["diagnoses"][0]["plan"])
                plan[field] = value
                with self.assertRaisesRegex(UsageError, "does not match the remedy that authorized it"):
                    validate_store(corrupt, self.history)
        # The rest are caught by the plan's own guards; the ledger is refused
        # either way, and no substitution reaches `validate_work`.
        for field, value in (("checkpoint", "another-checkpoint"),
                             ("base_revision", "f" * 40),
                             ("first_fix", 9),
                             ("last_fix", 99),
                             ("supersedes", "some-other-plan")):
            with self.subTest(field=field):
                corrupt = copy.deepcopy(self.store)
                plan = next(row for row in corrupt["plans"] if row["id"] == corrupt["diagnoses"][0]["plan"])
                plan[field] = value
                with self.assertRaises(UsageError):
                    validate_store(corrupt, self.history)

    def test_two_corrections_share_one_approval_and_the_first_outside_budget_refuses(self):
        plan = self.approve()
        self.assertEqual((plan["first_fix"], plan["last_fix"]), (6, 7))
        validate_work(self.store, self.history, TASK, 6, "plan-1", WORK)
        self.finish(6)
        with self.assertRaisesRegex(UsageError, "actual blocking review"):
            validate_work(self.store, self.history, TASK, 7, "plan-1", WORK)
        result = record_report(self.store, {"dispatch": "fix-6", "head_revision": HEAD, "verdict": "blocking",
            "review_mode": "full", "reviewer": "independent-reviewer", "report": str(self.review), "changed_paths": ["src/parser.py"]}, AT)
        self.assertEqual(result["evidence"]["sha256"], hashlib.sha256(self.review.read_bytes()).hexdigest())
        validate_work(self.store, self.history, TASK, 7, "plan-1", WORK)
        self.finish(7)
        self.assertEqual([row["id"] for row in active_plans(self.store)], ["plan-1"])
        self.assertEqual(confirmed_fix(self.history, TASK), 7)
        with self.assertRaisesRegex(UsageError, "outside the approved task or budget"):
            validate_work(self.store, self.history, TASK, 8, "plan-1", WORK)
        path = self.root / "state.json"
        save_state(path, self.state)
        restored, usable = load_state_checked(path)
        self.assertTrue(usable)
        self.assertEqual(restored, self.state)

    def test_ordinary_sixth_wrong_task_scope_base_and_paths_are_refused(self):
        self.approve()
        for task, plan, work in (
            (TASK, None, WORK), ("other-task", "plan-1", WORK),
            (TASK, "plan-1", {**WORK, "scope": "Other work"}),
            (TASK, "plan-1", {**WORK, "base_revision": HEAD}),
            (TASK, "plan-1", {**WORK, "paths": ["secrets/config.py"]}),
        ):
            with self.subTest(task=task, work=work), self.assertRaises(UsageError):
                validate_work(self.store, self.history, task, 6, plan, work)

    def test_checkpoint_pauses_implementation_even_when_an_audit_runs(self):
        self.seed_checkpoint()
        add_assignment(self.state, AT, "reviewer", "auditor", task=TASK)
        status = task_statuses(self.store, self.history)[TASK]
        self.assertEqual((status["status"], status["paused_work"]), ("awaiting_diagnosis", "implementation"))

    def test_a_cited_judge_report_still_needs_the_pinned_judge_after_latest_development(self):
        self.exhaust()
        with self.assertRaisesRegex(UsageError, "pinned judge"):
            checkpoint(self.store, self.history, {"id": "cp", "task": TASK, "defect": "F1", "previous_attempts": "Five fixes",
                "progress": "Still blocked", "change_in_approach": "Reassess", "judge_report": str(self.judge_report), "requested_by": REQUEST}, AT, "judge")

    def test_an_exhausted_allowance_awaits_the_diagnosis_without_a_judge_citation(self):
        # The allowance boundary is a budget decision only the operator makes.
        # Requiring a ruling first spent 16 of one fleet's 30 lifetime judge
        # dispatches on one task (jbaruch/coding-policy#396).
        self.exhaust()
        record = checkpoint(self.store, self.history, {"id": "cp", "task": TASK, "defect": "F1",
            "previous_attempts": "Five fixes", "progress": "Still blocked",
            "change_in_approach": "Reassess"}, AT, "judge")
        self.assertNotIn("judge_evidence", record)
        self.assertNotIn("judge_agent", record)
        self.assertEqual(record["fix_round"], 5)
        status = task_statuses(self.store, self.history)[TASK]
        self.assertEqual((status["status"], status["paused_work"]), ("awaiting_diagnosis", "implementation"))

    def test_that_checkpoint_carries_the_operator_bounded_approval_through(self):
        self.exhaust()
        self.consult_investigator("2026-02-03T09:30:00+00:00", "2026-02-03T09:45:00+00:00")
        add_assignment(self.state, AT, "judge", "judge", task=TASK, judge_mode="diagnosis")
        checkpoint(self.store, self.history, {"id": "cp", "task": TASK, "defect": "F1",
            "previous_attempts": "Five fixes", "progress": "Still blocked",
            "change_in_approach": "Reassess"}, AT, "judge")
        # The operator overrides a remedy; without one there is nothing to override.
        with self.assertRaisesRegex(UsageError, "no diagnosis at fix round 5"):
            authorize_plan(self.store, self.history, {"id": "plan-1", "task": TASK, "checkpoint": "cp",
                "scope": WORK["scope"], "allowed_paths": ["src/*"], "additional_fixes": 2, "authorization": AUTH}, AT)
        diag = self.run_diagnosis(self.diagnosis("diag-cp", "continue", 1, "cp"))
        self.assertEqual(diag["fix_round"], 5)
        plan = authorize_plan(self.store, self.history, {"id": "plan-1", "task": TASK, "checkpoint": "cp",
            "scope": WORK["scope"], "allowed_paths": ["src/*"], "additional_fixes": 2, "authorization": AUTH,
            "supersedes": diag["plan"]}, AT)
        self.assertEqual((plan["first_fix"], plan["last_fix"]), (6, 7))
        validate_work(self.store, self.history, TASK, 6, "plan-1", WORK)

    def test_an_unexhausted_budget_still_refuses_the_checkpoint(self):
        with self.assertRaisesRegex(UsageError, "not exhausted"):
            checkpoint(self.store, self.history, {"id": "cp", "task": TASK, "defect": "F1",
                "previous_attempts": "None", "progress": "None",
                "change_in_approach": "Reassess"}, AT, "judge")

    def test_an_exhausted_budget_names_the_diagnosis_not_an_operator_plan(self):
        with self.assertRaisesRegex(UsageError, re.escape("`{}`".format(runnable.command("diagnose")))):
            validate_work(self.store, self.history, TASK, 6, None, WORK)

    def replay_data(self):
        return {"id": "checkpoint-5", "task": TASK, "defect": "F1 remains open",
                "previous_attempts": "Five attempts changed parsing and quoting",
                "progress": "Some counterexamples now pass; the quoted case remains red",
                "change_in_approach": "Use one canonical parser", "judge_report": str(self.judge_report), "requested_by": REQUEST}

    def test_a_checkpoint_written_before_the_bump_migrates_with_its_evidence(self):
        prior = self.seed_checkpoint()
        evidence = copy.deepcopy(prior["judge_evidence"])
        prior["schema_version"] = 1
        prior.pop("requested_by", None)
        self.assertTrue(migrate_store(self.store))
        self.assertEqual(prior["schema_version"], 2)
        self.assertNotIn("requested_by", prior)
        self.assertEqual((prior["judge_agent"], prior["judge_evidence"]), ("judge", evidence))
        validate_store(self.store, self.history)
        self.assertFalse(migrate_store(self.store))

    def test_a_rejected_document_is_left_exactly_as_it_was_found(self):
        # coding-policy#400: the nested row upgrades ran before the enclosing
        # document's own preflight, so a store the migration then refused had
        # already been stamped in memory.
        prior = self.seed_checkpoint()
        prior["schema_version"] = 1
        prior.pop("requested_by", None)
        rejected = copy.deepcopy(self.store)
        rejected["schema_version"] = 5
        # Unowned newer data for a version-5 document: the preflight refuses it.
        before = copy.deepcopy(rejected)
        with self.assertRaisesRegex(UsageError, "unowned newer"):
            migrate_store(rejected)
        self.assertEqual(rejected, before)
        self.assertEqual(rejected["checkpoints"][0]["schema_version"], 1)

    def test_an_older_checkpoint_without_its_required_ruling_refuses_to_migrate(self):
        prior = self.seed_checkpoint()
        prior["schema_version"] = 1
        prior.pop("requested_by", None)
        del prior["judge_evidence"]
        with self.assertRaisesRegex(UsageError, "missing the ruling evidence"):
            migrate_store(self.store)

    def test_replaying_a_legacy_checkpoint_with_its_original_payload_returns_it(self):
        # coding-policy#400: the receipt is required of NEW records. An older
        # row was written before the field existed, so re-running its original
        # request is an already-processed one, not a missing receipt
        # (rules/file-hygiene.md Idempotency).
        prior = self.seed_checkpoint()
        prior["schema_version"] = 2
        prior.pop("requested_by", None)
        legacy = {key: value for key, value in self.replay_data().items() if key != "requested_by"}
        replay = checkpoint(self.store, self.history, legacy, AT, "judge")
        self.assertIs(replay, prior)
        self.assertNotIn("requested_by", replay)
        self.assertEqual(len(self.store["checkpoints"]), 1)
        validate_store(self.store, self.history)

    def test_replaying_a_migrated_checkpoint_returns_it(self):
        # The row outlives the writer's version: re-running an already-recorded
        # checkpoint must return it, not read the version as changed evidence.
        prior = self.seed_checkpoint()
        prior["schema_version"] = 1
        prior.pop("requested_by", None)
        migrate_store(self.store)
        replay = checkpoint(self.store, self.history, self.replay_data(), AT, "judge")
        self.assertIs(replay, prior)
        self.assertEqual(len(self.store["checkpoints"]), 1)
        validate_store(self.store, self.history)

    def test_reusing_a_checkpoint_identity_for_other_evidence_still_refuses(self):
        self.seed_checkpoint()
        with self.assertRaisesRegex(UsageError, "already describes different evidence"):
            checkpoint(self.store, self.history, {"id": "checkpoint-5", "task": TASK, "defect": "Another defect",
                "previous_attempts": "Five attempts changed parsing and quoting",
                "progress": "Some counterexamples now pass; the quoted case remains red",
                "change_in_approach": "Use one canonical parser", "judge_report": str(self.judge_report), "requested_by": REQUEST}, AT, "judge")

    def test_a_second_cited_ruling_for_the_task_is_refused(self):
        # The bound is per task: a re-granted budget exhausting again must not
        # buy another ruling (jbaruch/coding-policy#396).
        self.seed_checkpoint()
        for fix in (6, 7):
            add_assignment(self.state, "2026-02-03T12:00:0{}+00:00".format(fix), "developer", "worker", task=TASK, fix_round=fix)
        with self.assertRaisesRegex(UsageError, "at most one per task"):
            checkpoint(self.store, self.history, {"id": "checkpoint-7", "task": TASK, "defect": "F1 still open",
                "previous_attempts": "Seven attempts", "progress": "Unchanged",
                "change_in_approach": "Rewrite the parser", "judge_report": str(self.judge_report), "requested_by": REQUEST}, AT, "judge")

    def test_a_later_checkpoint_without_a_ruling_still_records(self):
        self.seed_checkpoint()
        for fix in (6, 7):
            add_assignment(self.state, "2026-02-03T12:00:0{}+00:00".format(fix), "developer", "worker", task=TASK, fix_round=fix)
        record = checkpoint(self.store, self.history, {"id": "checkpoint-7", "task": TASK, "defect": "F1 still open",
            "previous_attempts": "Seven attempts", "progress": "Unchanged",
            "change_in_approach": "Rewrite the parser"}, AT, "judge")
        self.assertEqual(record["fix_round"], 7)
        self.assertNotIn("judge_evidence", record)

    def test_a_partial_ruling_trio_on_a_checkpoint_is_refused(self):
        prior = self.seed_checkpoint()
        del prior["judge_evidence"]
        with self.assertRaises(UsageError):
            validate_store(self.store, self.history)

    def test_a_cited_ruling_records_the_operator_request_that_asked_for_it(self):
        # coding-policy#400: the ledger called the ruling "operator-requested"
        # while recording nothing of the request.
        self.exhaust()
        self.consult_investigator("2026-02-03T09:30:00+00:00", "2026-02-03T09:45:00+00:00")
        add_assignment(self.state, AT, "judge", "judge", task=TASK, judge_mode="diagnosis")
        base = {"id": "cp", "task": TASK, "defect": "F1 remains open",
                "previous_attempts": "Five attempts", "progress": "Partly",
                "change_in_approach": "One parser"}
        with self.assertRaisesRegex(UsageError, "the operator's to request"):
            checkpoint(self.store, self.history, {**base, "judge_report": str(self.judge_report)}, AT, "judge")
        for bad in ({"source": "operator"}, {"source": "operator", "quote": ""}, "operator said so"):
            with self.subTest(receipt=bad):
                with self.assertRaises(UsageError):
                    checkpoint(self.store, self.history,
                               {**base, "judge_report": str(self.judge_report), "requested_by": bad}, AT, "judge")
        # The receipt without a ruling has nothing to authorize.
        with self.assertRaisesRegex(UsageError, "record the checkpoint without it"):
            checkpoint(self.store, self.history, {**base, "requested_by": REQUEST}, AT, "judge")
        record = checkpoint(self.store, self.history,
                            {**base, "judge_report": str(self.judge_report), "requested_by": REQUEST}, AT, "judge")
        self.assertEqual(record["requested_by"], REQUEST)
        validate_store(self.store, self.history)

    def test_a_ledger_citing_two_rulings_for_one_task_is_refused(self):
        # coding-policy#400: `checkpoint` refuses the second as it writes it;
        # the read boundary checked each row alone and accepted both.
        self.seed_checkpoint()
        corrupt = copy.deepcopy(self.store)
        first = corrupt["checkpoints"][0]
        corrupt["checkpoints"].append({**first, "id": "checkpoint-5b"})
        with self.assertRaisesRegex(UsageError, "more than one operator-requested ruling"):
            validate_store(corrupt, self.history)

    def test_legacy_citations_predating_the_bound_still_read(self):
        # coding-policy#436: the bound arrived with version 3. Reading it over
        # version-2 rows made ledgers written before it exist unreadable, which
        # blocked new work on four tasks and 20 rows of real history.
        self.seed_checkpoint()
        legacy = copy.deepcopy(self.store)
        for index, row in enumerate(list(legacy["checkpoints"])):
            row["schema_version"] = 2
            row.pop("requested_by", None)
            legacy["checkpoints"].append({**copy.deepcopy(row), "id": "legacy-{}".format(index)})
        self.assertEqual(len(legacy["checkpoints"]), 2)
        validate_store(legacy, self.history)
        # A version-3 row still carries the bound: one per task, no more.
        mixed = copy.deepcopy(legacy)
        current = {**copy.deepcopy(self.store["checkpoints"][0]), "id": "current-a"}
        mixed["checkpoints"].append(current)
        validate_store(mixed, self.history)
        mixed["checkpoints"].append({**copy.deepcopy(current), "id": "current-b"})
        with self.assertRaisesRegex(UsageError, "more than one operator-requested ruling"):
            validate_store(mixed, self.history)

    def test_an_unknown_checkpoint_field_is_refused(self):
        self.exhaust()
        with self.assertRaisesRegex(UsageError, "optional judge_report"):
            checkpoint(self.store, self.history, {"id": "cp", "task": TASK, "defect": "F1",
                "previous_attempts": "Five fixes", "progress": "Still blocked",
                "change_in_approach": "Reassess", "operator_note": "approve"}, AT, "judge")

    def test_retry_returns_the_same_dispatch_and_never_consumes_twice(self):
        self.approve()
        self.finish(6)
        previous = copy.deepcopy(self.state)
        replay = prior_dispatch(self.store, "fix-6", "c" * 64)
        assert replay is not None
        self.assertEqual(replay["status"], "applied")
        self.assertEqual(self.state, previous)
        self.assertEqual(confirmed_fix(self.history, TASK), 6)
        with self.assertRaisesRegex(UsageError, "different inputs"):
            prior_dispatch(self.store, "fix-6", "d" * 64)

    def test_interrupted_send_holds_its_slot_and_a_proven_pre_send_failure_retries(self):
        self.approve()
        record = self.reservation(6)
        reserve(self.store, record, AT)
        abort_pre_send(self.store, record["id"], AT, "clear refused before a brief was sent")
        replay = prior_dispatch(self.store, record["id"], record["fingerprint"])
        assert replay is not None
        self.assertEqual(replay["status"], "not_sent")
        reserve(self.store, record, AT)
        mark_sending(self.store, record["id"], AT, {"cleared": True})
        abort_pre_send(self.store, record["id"], AT, "connection lost after send began")
        with self.assertRaisesRegex(UsageError, "uncertain outcome"):
            prior_dispatch(self.store, record["id"], record["fingerprint"])
        with self.assertRaisesRegex(UsageError, "unresolved dispatch"):
            reserve(self.store, {**record, "id": "another-id"}, AT)
        self.assertEqual(len(self.store["dispatches"]), 1)
        self.assertEqual(confirmed_fix(self.history, TASK), 5)

    def test_a_seated_dispatch_needs_the_migrated_store(self):
        # A seat in the dispatch role is what store version 10 added. Appending
        # one to an older store leaves it carrying a row its version never
        # wrote, and the next load refuses the whole ledger, not the row (#434).
        record = {**self.reservation(1), "role": "reviewer#api", "fix_round": None,
                  "plan": None, "work": None}
        older = copy.deepcopy(self.store)
        older["schema_version"] = 9
        before = copy.deepcopy(older)
        with self.assertRaisesRegex(UsageError, "needs the owner-migrated recovery store"):
            reserve(older, record, AT)
        self.assertEqual(older, before)
        reserve(self.store, record, AT)
        self.assertEqual(self.store["dispatches"][-1]["role"], "reviewer#api")

    def test_release_clear_and_explicit_null_recovery_preserve_original_rows(self):
        add_assignment(self.state, AT, "developer", "worker", task=TASK)
        original = copy.deepcopy(self.history)
        permission = authorize_context(self.store, self.history, {"task": TASK, "assignment_index": 0,
            "reason": "Old dispatch missed its native session ID", "authorization": AUTH, "evidence": str(self.review)}, AT, None)
        self.assertEqual(permission["basis"], "operator_authorized_fresh_handoff")
        self.assertEqual(self.history, original)
        transition = fresh_transition(self.store, self.history, TASK, 1)
        assert transition is not None
        self.assertEqual(transition["reason"], "authorized_context_recovery")
        add_assignment(self.state, "2026-02-03T10:00:01+00:00", "release", "worker", task=TASK, cleared=True, clear_reason="automatic")
        transition = fresh_transition(self.store, self.history, TASK, 1)
        assert transition is not None
        self.assertEqual(transition["release_assignment"], 1)
        self.assertEqual(self.history[0], original[0])

    def test_corrupt_bounds_or_unrecorded_sixth_fix_leave_state_unusable(self):
        self.approve()
        self.finish(6)
        corrupt = copy.deepcopy(self.store)
        corrupt["plans"][0]["last_fix"] = 999
        with self.assertRaises(UsageError):
            validate_store(corrupt, self.history)

    def test_changed_operator_decision_supersedes_bounds_without_erasing_approval(self):
        original = copy.deepcopy(self.approve())
        data = {"id": "plan-2", "task": TASK, "checkpoint": "checkpoint-5", "scope": "A changed correction approach",
                "allowed_paths": ["src/parser.py"], "additional_fixes": 1, "authorization": AUTH}
        with self.assertRaisesRegex(UsageError, "supersedes"):
            authorize_plan(self.store, self.history, data, AT)
        changed = authorize_plan(self.store, self.history, {**data, "supersedes": "plan-1"}, AT)
        self.assertEqual(next(row for row in self.store["plans"] if row["id"] == "plan-1"), original)
        self.assertEqual((changed["first_fix"], changed["last_fix"]), (6, 6))
        with self.assertRaisesRegex(UsageError, "superseded"):
            validate_work(self.store, self.history, TASK, 6, "plan-1", WORK)
        validate_work(self.store, self.history, TASK, 6, "plan-2", {**WORK, "scope": data["scope"]})
        validate_store(self.store, self.history)

    def test_review_receipts_preserve_independence_full_approval_and_current_artifact(self):
        self.approve()
        self.finish(6)
        data = {"dispatch": "fix-6", "head_revision": HEAD, "verdict": "blocking", "review_mode": "full",
                "reviewer": "independent-reviewer", "report": str(self.review), "changed_paths": ["src/parser.py"]}
        for change in ({"reviewer": "worker"}, {"verdict": "approved", "review_mode": "scoped"},
                       {"head_revision": "c" * 40}, {"changed_paths": ["outside/scope.py"]}):
            with self.subTest(change=change), self.assertRaises(UsageError):
                record_report(self.store, {**data, **change}, AT)
        # The report's own VERDICT line decides: a differing verdict and a
        # contract gap both refuse (#625).
        with self.assertRaisesRegex(UsageError, "VERDICT: blocking"):
            record_report(self.store, {**data, "verdict": "approved"}, AT)
        original = self.review.read_text()
        for text in (original.replace("VERDICT: blocking\n", ""), original + "VERDICT: blocking\n"):
            self.review.write_text(text)
            with self.subTest(text=text), self.assertRaisesRegex(UsageError, "contract"):
                record_report(self.store, data, AT)
        self.review.write_text(original)
        record_report(self.store, data, AT)
        self.review.write_text("This report was replaced after it was recorded.\n")
        with self.assertRaisesRegex(UsageError, "artifact changed"):
            validate_work(self.store, self.history, TASK, 7, "plan-1", WORK)

    def test_future_or_corrupt_recovery_records_preserve_disk_and_refuse_history(self):
        self.approve()
        self.finish(6)
        variants = []
        future = copy.deepcopy(self.state)
        future["recovery"]["plans"][0]["schema_version"] = 2
        variants.append(future)
        invalid_result = copy.deepcopy(self.state)
        invalid_result["recovery"]["dispatches"][0]["result"] = None
        variants.append(invalid_result)
        invalid_version = copy.deepcopy(self.state)
        invalid_version["recovery"]["schema_version"] = True
        variants.append(invalid_version)
        path = self.root / "invalid.json"
        for variant in variants:
            with self.subTest(variant=variant):
                save_state(path, variant)
                before = path.read_bytes()
                warnings = []
                restored, usable = load_state_checked(path, warn=warnings.append)
                self.assertFalse(usable)
                self.assertEqual(restored["assignments"], [])
                self.assertEqual(path.read_bytes(), before)
                self.assertTrue(warnings)
        corrupt = copy.deepcopy(self.store)
        corrupt["dispatches"] = []
        with self.assertRaisesRegex(UsageError, "lacks its owner-managed"):
            validate_store(corrupt, self.history)

    def test_brief_identity_changes_with_content_and_explicit_reuse_is_detectable(self):
        common, brief = self.root / "COMMON.md", self.root / "developer.md"
        common.write_text("Common\n")
        brief.write_text("First attempt\n")
        paths_by_role = {"common": str(common), "developer": str(brief)}
        first = dispatch_identity(TASK, "developer", "worker", 1, paths_by_role, "try-1")
        brief.write_text("Changed attempt\n")
        second = dispatch_identity(TASK, "developer", "worker", 1, paths_by_role, "try-1")
        self.assertEqual(first[0], second[0])
        self.assertNotEqual(first[1], second[1])

    def test_assignment_scoped_dispatch_and_result_use_version_four(self):
        record = {"id": "scoped-1", "task": TASK, "role": "reviewer", "agent": "reviewer-a1",
                  "worker_kind": "codex", "fix_round": None, "fingerprint": "d" * 64,
                  "plan": None, "work": None, "reviewer_scope": "verification"}
        reserved = reserve(self.store, record, AT)
        self.assertEqual(reserved["schema_version"], ASSIGNMENT_DISPATCH_VERSION)
        mark_sending(self.store, record["id"], AT, {"cleared": True})
        add_assignment(self.state, AT, "reviewer", "reviewer-a1", task=TASK,
                       reviewer_scope="verification")
        finish_dispatch(self.store, record["id"], {
            "task": TASK, "role": "reviewer", "agent": "reviewer-a1",
            "worker_kind": "codex", "assignment_scoped": True,
            "fix_round": None, "reviewer_scope": "verification", "status": "applied",
        }, len(self.history) - 1, AT)
        self.assertEqual(reserved["result"]["schema_version"], ASSIGNMENT_DISPATCH_VERSION)
        validate_store(self.store, self.history)

    def test_store_version_fourteen_migrates_only_without_scoped_fields(self):
        clean = copy.deepcopy(self.store)
        clean["schema_version"] = 14
        self.assertTrue(migrate_store(clean))
        self.assertEqual(clean["schema_version"], RECOVERY_STORE_VERSION)

        scoped = copy.deepcopy(self.store)
        scoped["schema_version"] = 14
        scoped["dispatches"].append({
            "schema_version": ASSIGNMENT_DISPATCH_VERSION, "at": AT,
            "id": "scoped", "task": TASK, "role": "reviewer", "agent": "reviewer-a1",
            "worker_kind": "codex", "fix_round": None, "fingerprint": "e" * 64,
            "plan": None, "work": None, "status": "reserved", "result": None, "report": None,
        })
        with self.assertRaisesRegex(UsageError, "assignment-scoped"):
            migrate_store(scoped)

    REPORT = "/reports/tester.md"
    BRIEF = "brief-identity-tester"

    def dispatch_tester(self, number, agent, identity: "str | None" = BRIEF):
        record = {"id": "tester-{}-{}".format(number, agent), "task": TASK, "role": "tester", "agent": agent, "fix_round": None,
                  "fingerprint": ("%02d" % number) * 32, "plan": None, "work": None, "brief_identity": identity,
                  "provider": agent.rsplit("-", 1)[0]}
        reserve(self.store, record, AT)
        mark_sending(self.store, record["id"], AT, {"cleared": True})
        add_assignment(self.state, "2026-02-03T12:00:0{}+00:00".format(number), "tester", agent, task=TASK)
        finish_dispatch(self.store, record["id"], {"status": "applied", **{
            key: record[key] for key in ("task", "role", "agent", "fix_round")}}, len(self.history) - 1, AT)
        return record["id"]

    def refusal_receipt(self, agent, name="refusal.json", **overrides):
        path = self.root / name
        path.write_text(json.dumps({"agent": agent, "state": "idle", "report_path": self.REPORT,
                                    "found": False, "elapsed_seconds": 12, "reason": "terminal_provider_refusal", **overrides}))
        return str(path)

    def test_refusal_is_recorded_once_against_its_applied_dispatch(self):
        # coding-policy#399: wait-report's exit 5 went to stdout and nowhere
        # else, so nothing could tell a first refusal from a second.
        first = self.dispatch_tester(1, "codex-a")
        receipt = self.refusal_receipt("codex-a")
        with self.assertRaisesRegex(UsageError, "requires dispatch and receipt"):
            record_refusal(self.store, {"dispatch": first}, AT, "codex", self.REPORT)
        with self.assertRaisesRegex(UsageError, "exit-5 output"):
            record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a", "wrong.json", reason="marker_unconfirmed")}, AT, "codex", self.REPORT)
        with self.assertRaisesRegex(UsageError, "exit-5 output"):
            record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("someone-else", "other.json")}, AT, "codex", self.REPORT)
        with self.assertRaisesRegex(UsageError, "not wait-report JSON"):
            record_refusal(self.store, {"dispatch": first, "receipt": str(self.review)}, AT, "codex", self.REPORT)
        result = record_refusal(self.store, {"dispatch": first, "receipt": receipt}, AT, "codex", self.REPORT)
        self.assertEqual((result["provider"], result["reason"]), ("codex", "terminal_provider_refusal"))
        self.assertEqual(result["evidence"]["sha256"], hashlib.sha256(Path(receipt).read_bytes()).hexdigest())
        self.assertEqual(record_refusal(self.store, {"dispatch": first, "receipt": receipt}, AT, "codex", self.REPORT), result)
        with self.assertRaisesRegex(UsageError, "different refusal receipt"):
            record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a", "later.json", elapsed_seconds=99)}, AT, "codex", self.REPORT)
        self.assertEqual([row["kind"] for row in self.store["events"] if row["kind"] == "provider_refusal_recorded"], ["provider_refusal_recorded"])
        pending = {"id": "tester-pending", "task": TASK, "role": "tester", "agent": "claude-a", "fix_round": None,
                   "fingerprint": "ab" * 32, "plan": None, "work": None}
        reserve(self.store, pending, AT)
        with self.assertRaisesRegex(UsageError, "confirmed applied dispatch"):
            record_refusal(self.store, {"dispatch": "tester-pending", "receipt": self.refusal_receipt("claude-a", "pending.json")}, AT, "claude", self.REPORT)
        abort_pre_send(self.store, "tester-pending", AT, "fixture")
        validate_store(self.store, self.history)

    def test_refusal_receipt_must_bind_to_the_dispatch_report_and_full_exit_five_shape(self):
        first = self.dispatch_tester(1, "codex-a")
        with self.assertRaisesRegex(UsageError, "no supervision enrollment"):
            record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a")}, AT, "codex", None)
        with self.assertRaisesRegex(UsageError, "names report /reports/other.md"):
            record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a", "other.json", report_path="/reports/other.md")}, AT, "codex", self.REPORT)
        with self.assertRaisesRegex(UsageError, "complete exit-5 output"):
            record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a", "working.json", state="working")}, AT, "codex", self.REPORT)
        with self.assertRaisesRegex(UsageError, "complete exit-5 output"):
            record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a", "extra.json", extra=True)}, AT, "codex", self.REPORT)
        result = record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a")}, AT, "codex", self.REPORT)
        self.assertEqual(result["report_path"], self.REPORT)
        validate_store(self.store, self.history)

    def test_a_malformed_receipt_is_a_usage_error_not_a_crash(self):
        # coding-policy#403: `agent` and `state` reached set membership
        # unvalidated, so a list or object raised TypeError out of the CLI.
        first = self.dispatch_tester(1, "codex-a")
        for name, override in (("list-agent", {"agent": ["codex-a"]}), ("dict-state", {"state": {"idle": True}}),
                               ("int-reason", {"reason": 5}), ("list-report", {"report_path": [self.REPORT]}),
                               ("negative", {"elapsed_seconds": -1}), ("float", {"elapsed_seconds": 1.5})):
            path = self.root / (name + ".json")
            path.write_text(json.dumps({"agent": "codex-a", "state": "idle", "report_path": self.REPORT, "found": False,
                                        "elapsed_seconds": 12, "reason": "terminal_provider_refusal", **override}))
            with self.assertRaises(UsageError):
                record_refusal(self.store, {"dispatch": first, "receipt": str(path)}, AT, "codex", self.REPORT)
        self.assertIsNone(self.store["dispatches"][0].get("refusal"))

    def test_a_not_sent_retry_refreshes_the_provider_the_fingerprint_misses(self):
        # coding-policy#403: reserve reuses a not_sent row, so a config change
        # before the retry would otherwise leave the original provider on it.
        record = {"id": "tester-retry", "task": TASK, "role": "tester", "agent": "codex-a", "fix_round": None,
                  "fingerprint": "cd" * 32, "plan": None, "work": None, "brief_identity": self.BRIEF, "provider": "codex"}
        reserve(self.store, record, AT)
        abort_pre_send(self.store, record["id"], AT, "fixture")
        reserve(self.store, {**record, "provider": "claude", "brief_identity": "brief-identity-refreshed"}, AT)
        saved = next(row for row in self.store["dispatches"] if row["id"] == record["id"])
        self.assertEqual((saved["provider"], saved["brief_identity"]), ("claude", "brief-identity-refreshed"))
        # A stale move naming the old provider would fail the ledger on the
        # next refusal, so the retry drops one the new record does not carry.
        abort_pre_send(self.store, record["id"], AT, "fixture")
        self.store["dispatches"][-1]["refusal_move"] = {"schema_version": 1, "from": "gone", "from_provider": "codex", "provider": "claude"}
        reserve(self.store, {**record, "provider": "grok"}, AT)
        saved = next(row for row in self.store["dispatches"] if row["id"] == record["id"])
        self.assertNotIn("refusal_move", saved)
        self.assertEqual(saved["provider"], "grok")
        validate_store(self.store, self.history)

    def test_the_dispatchs_recorded_provider_outranks_the_current_config(self):
        # coding-policy#403: a config edit between the send and the record
        # must not re-attribute the refusal to the new kind.
        first = self.dispatch_tester(1, "codex-a")
        result = record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a")}, AT, "claude", self.REPORT)
        self.assertEqual(result["provider"], "codex")
        legacy = self.dispatch_tester(2, "grok-a")
        del self.store["dispatches"][-1]["provider"]
        self.assertEqual(record_refusal(self.store, {"dispatch": legacy, "receipt": self.refusal_receipt("grok-a", "legacy.json")}, AT, "grok", self.REPORT)["provider"], "grok")
        validate_store(self.store, self.history)

    def test_a_move_never_reuses_a_report_path_the_chain_burned(self):
        # coding-policy#403: two attempts enrolled against one file would let
        # their evidence collide.
        first = self.dispatch_tester(1, "codex-a")
        record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a")}, AT, "codex", self.REPORT)
        with self.assertRaisesRegex(UsageError, "already carries the refusal"):
            refusal_move(self.store, TASK, "tester", None, "claude", self.BRIEF, self.REPORT)
        self.assertIsNotNone(refusal_move(self.store, TASK, "tester", None, "claude", self.BRIEF, "/reports/tester-2.md"))
        # An alias of the burned path is the same file (#403).
        with self.assertRaisesRegex(UsageError, "already carries the refusal"):
            refusal_move(self.store, TASK, "tester", None, "claude", self.BRIEF, "/reports/sub/../tester.md")
        self.assertIsNotNone(refusal_move(self.store, TASK, "tester", None, "claude", self.BRIEF))

    def test_a_reworded_brief_is_not_a_move(self):
        first = self.dispatch_tester(1, "codex-a")
        record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a")}, AT, "codex", self.REPORT)
        with self.assertRaisesRegex(UsageError, "reworded brief is not a move"):
            refusal_move(self.store, TASK, "tester", None, "claude", "brief-identity-reworded")
        self.assertIsNotNone(refusal_move(self.store, TASK, "tester", None, "claude", self.BRIEF))
        validate_store(self.store, self.history)

    def test_a_refused_dispatch_without_brief_identity_cannot_be_moved(self):
        legacy = self.dispatch_tester(1, "grok-a", identity=None)
        del self.store["dispatches"][-1]["brief_identity"]
        record_refusal(self.store, {"dispatch": legacy, "receipt": self.refusal_receipt("grok-a", "legacy.json")}, AT, "grok", self.REPORT)
        with self.assertRaisesRegex(UsageError, "predates brief identity"):
            refusal_move(self.store, TASK, "tester", None, "claude", self.BRIEF)
        validate_store(self.store, self.history)

    def test_brief_identity_masks_only_the_report_path(self):
        common = self.root / "COMMON.md"
        common.write_text("# common\n")
        brief = self.root / "tester.md"
        paths = {"common": str(common), "tester": str(brief)}
        brief.write_text("Write `/reports/a.md` covering the plan.\nREPORT: /reports/a.md\n")
        first = brief_identity(paths, "tester", "/reports/a.md")
        brief.write_text("Write `/reports/b.md` covering the plan.\nREPORT: /reports/b.md\n")
        self.assertEqual(brief_identity(paths, "tester", "/reports/b.md"), first)
        self.assertNotEqual(brief_identity(paths, "tester", None), first)
        brief.write_text("Write `/reports/b.md` covering the plan.\nSkip the security checks.\nREPORT: /reports/b.md\n")
        self.assertNotEqual(brief_identity(paths, "tester", "/reports/b.md"), first)
        with self.assertRaisesRegex(UsageError, "Cannot read brief"):
            brief_identity({**paths, "tester": str(self.root / "missing.md")}, "tester", None)

    def test_one_refusal_permits_one_move_and_the_second_stops_the_line(self):
        self.assertIsNone(refusal_move(self.store, TASK, "tester", None, "codex", self.BRIEF))
        first = self.dispatch_tester(1, "codex-a")
        record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a")}, AT, "codex", self.REPORT)
        with self.assertRaisesRegex(UsageError, "same provider"):
            refusal_move(self.store, TASK, "tester", None, "codex", self.BRIEF)
        self.assertIsNone(refusal_move(self.store, TASK, "reviewer", None, "codex", self.BRIEF))
        self.assertIsNone(refusal_move(self.store, TASK, "tester", 3, "codex", self.BRIEF))
        self.assertIsNone(refusal_move(self.store, "another-task", "tester", None, "codex", self.BRIEF))
        move = refusal_move(self.store, TASK, "tester", None, "claude", self.BRIEF)
        self.assertEqual(move, {"schema_version": 1, "from": first, "from_provider": "codex", "provider": "claude"})
        second = self.dispatch_tester(2, "claude-a")
        self.store["dispatches"][-1]["refusal_move"] = move
        validate_store(self.store, self.history)
        record_refusal(self.store, {"dispatch": second, "receipt": self.refusal_receipt("claude-a", "second.json")}, AT, "claude", self.REPORT)
        with self.assertRaisesRegex(UsageError, "refused by 2 providers") as caught:
            refusal_move(self.store, TASK, "tester", None, "grok", self.BRIEF)
        self.assertEqual(caught.exception.details["refusals"], [first, second])
        with self.assertRaisesRegex(UsageError, "refused by 2 providers"):
            refusal_move(self.store, TASK, "tester", None, "codex", self.BRIEF)
        validate_store(self.store, self.history)

    def test_a_second_move_waits_for_the_first_moves_outcome(self):
        first = self.dispatch_tester(1, "codex-a")
        record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a")}, AT, "codex", self.REPORT)
        move = refusal_move(self.store, TASK, "tester", None, "claude", self.BRIEF)
        pending = {"id": "tester-move", "task": TASK, "role": "tester", "agent": "claude-a", "fix_round": None,
                   "fingerprint": "ab" * 32, "plan": None, "work": None, "refusal_move": move, "brief_identity": self.BRIEF}
        reserve(self.store, pending, AT)
        with self.assertRaisesRegex(UsageError, "already moved to provider claude"):
            refusal_move(self.store, TASK, "tester", None, "grok", self.BRIEF)
        abort_pre_send(self.store, "tester-move", AT, "fixture")
        retry = refusal_move(self.store, TASK, "tester", None, "grok", self.BRIEF)
        assert retry is not None
        self.assertEqual(retry["from"], first)
        second = self.dispatch_tester(2, "claude-a")
        self.store["dispatches"][-1]["refusal_move"] = move
        with self.assertRaisesRegex(UsageError, "already moved to provider claude") as caught:
            refusal_move(self.store, TASK, "tester", None, "grok", self.BRIEF)
        self.assertEqual(caught.exception.details["moves"], [second])
        record_refusal(self.store, {"dispatch": second, "receipt": self.refusal_receipt("claude-a", "second.json")}, AT, "claude", self.REPORT)
        with self.assertRaisesRegex(UsageError, "refused by 2 providers"):
            refusal_move(self.store, TASK, "tester", None, "grok", self.BRIEF)
        validate_store(self.store, self.history)

    def test_store_migration_stamps_a_clean_older_store_and_refuses_unowned_fields(self):
        first = self.dispatch_tester(1, "codex-a")
        identified = copy.deepcopy(self.store)
        identified["schema_version"] = 5
        del identified["refusal_authorizations"]
        with self.assertRaisesRegex(UsageError, "unowned newer refusal"):
            migrate_store(identified)
        old = copy.deepcopy(self.store)
        old["schema_version"] = 5
        del old["refusal_authorizations"]
        del old["diagnoses"]
        del old["legacy_ruling_recoveries"]
        del old["approaches"]
        for row in old["dispatches"]:
            del row["brief_identity"]
            del row["provider"]
        before = copy.deepcopy(old)
        self.assertTrue(migrate_store(old))
        self.assertEqual(old["schema_version"], RECOVERY_STORE_VERSION)
        self.assertEqual(old.pop("refusal_authorizations"), [])
        self.assertEqual(old.pop("diagnoses"), [])
        self.assertEqual(old.pop("legacy_ruling_recoveries"), [])
        self.assertEqual(old.pop("approaches"), [])
        self.assertEqual({key: value for key, value in old.items() if key != "schema_version"},
                         {key: value for key, value in before.items() if key != "schema_version"})
        old["refusal_authorizations"] = []
        old["diagnoses"] = []
        old["legacy_ruling_recoveries"] = []
        old["approaches"] = []
        validate_store(old, self.history)
        self.assertFalse(migrate_store(old))
        stale = copy.deepcopy(self.store)
        stale["schema_version"] = 5
        for row in stale["dispatches"]:
            del row["brief_identity"]
            del row["provider"]
        with self.assertRaisesRegex(UsageError, "unowned newer records"):
            migrate_store(stale)
        # Version 6 owned brief_identity but not provider: a v6 store carrying
        # one is unowned newer data, a clean one stamps to 7 (#403).
        six = copy.deepcopy(self.store)
        six["schema_version"] = 6
        del six["diagnoses"]
        del six["legacy_ruling_recoveries"]
        with self.assertRaisesRegex(UsageError, "unowned newer refusal"):
            migrate_store(six)
        six = copy.deepcopy(self.store)
        six["schema_version"] = 6
        del six["diagnoses"]
        del six["legacy_ruling_recoveries"]
        del six["approaches"]
        for row in six["dispatches"]:
            del row["provider"]
        self.assertTrue(migrate_store(six))
        self.assertEqual(six["schema_version"], RECOVERY_STORE_VERSION)
        self.assertEqual(six["diagnoses"], [])
        validate_store(six, self.history)
        record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a")}, AT, "codex", self.REPORT)
        for version in (4, 5):
            stale = copy.deepcopy(self.store)
            stale["schema_version"] = version
            del stale["refusal_authorizations"]
            del stale["diagnoses"]
            for row in stale["dispatches"]:
                del row["brief_identity"]
                del row["provider"]
            with self.assertRaisesRegex(UsageError, "unowned newer refusal"):
                migrate_store(stale)
        with self.assertRaisesRegex(UsageError, "Unsupported recovery schema"):
            validate_store({**copy.deepcopy(self.store), "schema_version": 5}, self.history)

    def test_operator_authorization_permits_one_dispatch_after_the_stop(self):
        grant = {"id": "auth-1", "task": TASK, "role": "tester", "fix_round": None, "provider": "codex", "brief": "revised",
                 "decision": "Run the revised tester brief on codex.", "authorization": AUTH}
        with self.assertRaisesRegex(UsageError, "refusals from 0 provider"):
            authorize_refused_dispatch(self.store, grant, AT)
        with self.assertRaisesRegex(UsageError, "requires id, task, role"):
            authorize_refused_dispatch(self.store, {**grant, "extra": 1}, AT)
        first = self.dispatch_tester(1, "codex-a")
        record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a")}, AT, "codex", self.REPORT)
        # One refusal is a move, not an operator decision (Copilot on #402).
        with self.assertRaisesRegex(UsageError, "refusals from 1 provider"):
            authorize_refused_dispatch(self.store, grant, AT)
        second = self.dispatch_tester(2, "claude-a")
        self.store["dispatches"][-1]["refusal_move"] = {"schema_version": 1, "from": first, "from_provider": "codex", "provider": "claude"}
        record_refusal(self.store, {"dispatch": second, "receipt": self.refusal_receipt("claude-a", "second.json")}, AT, "claude", self.REPORT)
        with self.assertRaisesRegex(UsageError, "authorize-refused-dispatch"):
            refusal_move(self.store, TASK, "tester", None, "grok", self.BRIEF)
        saved = authorize_refused_dispatch(self.store, grant, AT)
        self.assertEqual(authorize_refused_dispatch(self.store, grant, AT), saved)
        with self.assertRaisesRegex(UsageError, "different decision"):
            authorize_refused_dispatch(self.store, {**grant, "decision": "Something else."}, AT)
        with self.assertRaisesRegex(UsageError, "unchanged or revised"):
            authorize_refused_dispatch(self.store, {**grant, "id": "auth-bad", "brief": "anything"}, AT)
        # The grant replaces the stop with its own scope: the approved provider and, here, a revised brief.
        with self.assertRaisesRegex(UsageError, "approves provider codex"):
            refusal_move(self.store, TASK, "tester", None, "grok", "brief-identity-reworded")
        move = refusal_move(self.store, TASK, "tester", None, "codex", "brief-identity-reworded")
        self.assertEqual(move, {"schema_version": 1, "from": second, "from_provider": "claude", "provider": "codex", "authorization": "auth-1"})
        third = self.dispatch_tester(3, "codex-a", identity="brief-identity-reworded")
        self.store["dispatches"][-1]["refusal_move"] = move
        validate_store(self.store, self.history)
        # The grant is consumed; the stop holds again until another decision.
        with self.assertRaisesRegex(UsageError, "refused by 2 providers"):
            refusal_move(self.store, TASK, "tester", None, "grok", self.BRIEF)
        record_refusal(self.store, {"dispatch": third, "receipt": self.refusal_receipt("codex-a", "third.json")}, AT, "codex", self.REPORT)
        with self.assertRaisesRegex(UsageError, "refused by 2 providers"):
            refusal_move(self.store, TASK, "tester", None, "grok", self.BRIEF)
        corrupt = copy.deepcopy(self.store)
        corrupt["dispatches"][1]["refusal_move"]["authorization"] = "auth-1"
        with self.assertRaisesRegex(UsageError, "reuses a consumed authorization"):
            validate_store(corrupt, self.history)
        for mutate in (
            lambda grant: grant.update(role="reviewer"),
            lambda grant: grant.update(provider="grok"),
            lambda grant: grant.update(brief="unchanged"),
        ):
            corrupt = copy.deepcopy(self.store)
            mutate(corrupt["refusal_authorizations"][0])
            with self.assertRaisesRegex(UsageError, "exceeds its authorization"):
                validate_store(corrupt, self.history)
        corrupt = copy.deepcopy(self.store)
        corrupt["dispatches"][-1]["refusal"]["provider"] = "grok"
        with self.assertRaisesRegex(UsageError, "other than the one it moved to"):
            validate_store(corrupt, self.history)
        corrupt = copy.deepcopy(self.store)
        corrupt["refusal_authorizations"].append({**saved, "id": "auth-early", "role": "reviewer"})
        with self.assertRaisesRegex(UsageError, "precedes the independent refusals"):
            validate_store(corrupt, self.history)
        corrupt = copy.deepcopy(self.store)
        corrupt["dispatches"][-1]["refusal_move"]["authorization"] = "auth-missing"
        with self.assertRaises(UsageError):
            validate_store(corrupt, self.history)
        self.assertEqual(third, "tester-3-codex-a")

    def test_an_unchanged_brief_authorization_holds_the_brief_fixed(self):
        first = self.dispatch_tester(1, "codex-a")
        record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a")}, AT, "codex", self.REPORT)
        second = self.dispatch_tester(2, "claude-a")
        self.store["dispatches"][-1]["refusal_move"] = {"schema_version": 1, "from": first, "from_provider": "codex", "provider": "claude"}
        record_refusal(self.store, {"dispatch": second, "receipt": self.refusal_receipt("claude-a", "second.json")}, AT, "claude", self.REPORT)
        authorize_refused_dispatch(self.store, {"id": "auth-grok", "task": TASK, "role": "tester", "fix_round": None, "provider": "grok",
                                                "brief": "unchanged", "decision": "Send the same brief to grok.", "authorization": AUTH}, AT)
        with self.assertRaisesRegex(UsageError, "approves the refused brief unchanged"):
            refusal_move(self.store, TASK, "tester", None, "grok", "brief-identity-reworded")
        move = refusal_move(self.store, TASK, "tester", None, "grok", self.BRIEF)
        assert move is not None
        self.assertEqual((move["provider"], move["authorization"]), ("grok", "auth-grok"))
        self.dispatch_tester(3, "grok-a")
        self.store["dispatches"][-1]["refusal_move"] = move
        validate_store(self.store, self.history)

    def test_refusal_receipt_accepts_the_enrolled_pane_id_as_the_worker_name(self):
        first = self.dispatch_tester(1, "codex-a")
        receipt = self.refusal_receipt("w3:p1", "pane.json")
        with self.assertRaisesRegex(UsageError, "complete exit-5 output"):
            record_refusal(self.store, {"dispatch": first, "receipt": receipt}, AT, "codex", self.REPORT)
        with self.assertRaisesRegex(UsageError, "complete exit-5 output"):
            record_refusal(self.store, {"dispatch": first, "receipt": receipt}, AT, "codex", self.REPORT, aliases=(None,))
        self.assertEqual(record_refusal(self.store, {"dispatch": first, "receipt": receipt}, AT, "codex", self.REPORT, aliases=("w3:p1",))["provider"], "codex")

    def test_corrupt_refusal_records_refuse_the_ledger(self):
        first = self.dispatch_tester(1, "codex-a")
        record_refusal(self.store, {"dispatch": first, "receipt": self.refusal_receipt("codex-a")}, AT, "codex", self.REPORT)
        second = self.dispatch_tester(2, "claude-a")
        self.store["dispatches"][-1]["refusal_move"] = refusal_move(self.store, TASK, "tester", None, "claude", self.BRIEF)
        validate_store(self.store, self.history)
        for mutate in (
            lambda row: row["refusal"].update(reason="something_else"),
            lambda row: row["refusal"].update(evidence="not-a-receipt"),
            lambda row: row["refusal"].update(receipt="/elsewhere/refusal.json"),
            lambda row: row["refusal"].pop("report_path"),
            lambda row: row["refusal"].pop("provider"),
            lambda row: row.update(status="not_sent", result=None),
        ):
            corrupt = copy.deepcopy(self.store)
            mutate(next(row for row in corrupt["dispatches"] if row["id"] == first))
            with self.assertRaises(UsageError):
                validate_store(corrupt, self.history)
        for mutate in (
            lambda row: row["refusal_move"].update({"from": second}),
            lambda row: row["refusal_move"].update(provider="codex"),
            lambda row: row["refusal_move"].update(from_provider="claude"),
            lambda row: row["refusal_move"].update(schema_version=2),
            lambda row: row["refusal_move"].pop("from"),
            lambda row: row.update(brief_identity="brief-identity-reworded"),
        ):
            corrupt = copy.deepcopy(self.store)
            mutate(next(row for row in corrupt["dispatches"] if row["id"] == second))
            with self.assertRaises(UsageError):
                validate_store(corrupt, self.history)
        # A refused source must precede its move: a self-reference and a later row both refuse.
        record_refusal(self.store, {"dispatch": second, "receipt": self.refusal_receipt("claude-a", "second.json")}, AT, "claude", self.REPORT)
        corrupt = copy.deepcopy(self.store)
        corrupt["dispatches"][-1]["refusal_move"] = {"schema_version": 1, "from": second, "from_provider": "claude", "provider": "grok"}
        with self.assertRaises(UsageError):
            validate_store(corrupt, self.history)
        corrupt = copy.deepcopy(self.store)
        corrupt["dispatches"][0]["refusal_move"] = {"schema_version": 1, "from": second, "from_provider": "claude", "provider": "codex"}
        with self.assertRaises(UsageError):
            validate_store(corrupt, self.history)



class JudgeModeReservationTests(unittest.TestCase):
    """A judge dispatch carries its mode from reservation onward (#495)."""

    def setUp(self):
        self.store = empty_state()["recovery"]

    def judge(self, **extra):
        return {"id": "judge-1", "task": TASK, "role": "judge", "agent": "claude", "fix_round": None,
                "fingerprint": "d" * 64, "plan": None, "work": None, **extra}

    def test_a_new_judge_reservation_without_a_mode_is_refused(self):
        for record in (self.judge(), self.judge(judge_mode=None)):
            with self.subTest(record=record):
                with self.assertRaisesRegex(UsageError, "declares its mode"):
                    reserve(self.store, record, AT)
                self.assertEqual(self.store["dispatches"], [])

    def test_a_judge_reservation_with_its_mode_is_a_version_three_dispatch(self):
        item = reserve(self.store, self.judge(judge_mode="diagnosis"), AT)
        self.assertEqual((item["schema_version"], item["judge_mode"]), (3, "diagnosis"))

    def test_a_stored_mode_less_judge_row_still_reads(self):
        # Rows recorded before the mode existed stay legacy history, never refused.
        self.store["dispatches"].append({**self.judge(), "at": AT, "schema_version": 1,
                                         "status": "sending", "result": None, "report": None,
                                         "context_before_send": {"cleared": True}})
        validate_store(self.store, [])

    def test_sending_refuses_a_context_naming_another_mode(self):
        reserve(self.store, self.judge(judge_mode="diagnosis"), AT)
        for context in ({"cleared": True}, {"cleared": True, "judge_mode": "adjudication"}):
            with self.subTest(context=context):
                with self.assertRaisesRegex(UsageError, "different judge mode"):
                    mark_sending(self.store, "judge-1", AT, context)
                self.assertEqual(self.store["dispatches"][0]["status"], "reserved")
        mark_sending(self.store, "judge-1", AT, {"cleared": True, "judge_mode": "diagnosis"})
        self.assertEqual(self.store["dispatches"][0]["status"], "sending")

    def test_a_pending_judge_context_is_checked_against_its_row(self):
        reserve(self.store, self.judge(judge_mode="diagnosis"), AT)
        mark_sending(self.store, "judge-1", AT, {"cleared": True, "judge_mode": "diagnosis"})
        validate_store(self.store, [])
        for mode in ("adjudication", None):
            with self.subTest(mode=mode):
                corrupt = copy.deepcopy(self.store)
                corrupt["dispatches"][0]["context_before_send"]["judge_mode"] = mode
                with self.assertRaisesRegex(UsageError, "different judge mode"):
                    validate_store(corrupt, [])

    def test_a_sent_judge_without_its_pre_send_context_is_refused(self):
        # Reconcile reads the mode from this context; losing it recovers `unknown`.
        reserve(self.store, self.judge(judge_mode="diagnosis"), AT)
        validate_store(self.store, [])
        mark_sending(self.store, "judge-1", AT, {"cleared": True, "judge_mode": "diagnosis"})
        for status in ("sending", "sent_but_not_started", "applied"):
            for broken in (None, "diagnosis", ["diagnosis"]):
                with self.subTest(status=status, context=broken):
                    corrupt = copy.deepcopy(self.store)
                    corrupt["dispatches"][0]["status"] = status
                    if broken is None:
                        del corrupt["dispatches"][0]["context_before_send"]
                    else:
                        corrupt["dispatches"][0]["context_before_send"] = broken
                    with self.assertRaisesRegex(UsageError, "pre-send context"):
                        validate_store(corrupt, [])

    def test_a_present_context_is_an_object_on_every_status(self):
        for record, status in ((self.judge(judge_mode="diagnosis"), "reserved"),
                               ({**self.judge(), "id": "review-1", "role": "reviewer"}, "not_sent")):
            for broken in (None, "diagnosis", ["diagnosis"]):
                with self.subTest(role=record["role"], status=status, context=broken):
                    store = empty_state()["recovery"]
                    reserve(store, record, AT)
                    store["dispatches"][0]["status"] = status
                    store["dispatches"][0]["context_before_send"] = broken
                    with self.assertRaisesRegex(UsageError, "pre-send context is not an object"):
                        validate_store(store, [])

    def test_a_stored_mode_less_judge_row_is_never_resent_without_a_mode(self):
        # The row stays readable, but a retry would send an undeclared judge.
        self.store["dispatches"].append({**self.judge(), "at": AT, "schema_version": 1,
                                         "status": "not_sent", "result": None, "report": None})
        validate_store(self.store, [])
        with self.assertRaisesRegex(UsageError, "declares its mode"):
            reserve(self.store, self.judge(), AT)
        self.assertEqual(self.store["dispatches"][0]["status"], "not_sent")

    def test_sending_refuses_a_context_that_is_not_an_object(self):
        reserve(self.store, self.judge(judge_mode="diagnosis"), AT)
        for context in (None, "diagnosis", ["diagnosis"]):
            with self.subTest(context=context):
                with self.assertRaisesRegex(UsageError, "pre-send context is an object"):
                    mark_sending(self.store, "judge-1", AT, context)
                self.assertEqual(self.store["dispatches"][0]["status"], "reserved")

    def test_a_non_judge_context_carries_no_mode_key_at_all(self):
        reserve(self.store, {**self.judge(), "id": "review-1", "role": "reviewer"}, AT)
        with self.assertRaisesRegex(UsageError, "different judge mode"):
            mark_sending(self.store, "review-1", AT, {"cleared": True, "judge_mode": None})
        mark_sending(self.store, "review-1", AT, {"cleared": True})
        corrupt = copy.deepcopy(self.store)
        corrupt["dispatches"][0]["context_before_send"]["judge_mode"] = None
        with self.assertRaisesRegex(UsageError, "different judge mode"):
            validate_store(corrupt, [])

    def test_a_non_judge_context_carrying_a_mode_is_refused(self):
        reserve(self.store, {**self.judge(), "id": "review-1", "role": "reviewer"}, AT)
        mark_sending(self.store, "review-1", AT, {"cleared": True})
        corrupt = copy.deepcopy(self.store)
        corrupt["dispatches"][0]["context_before_send"]["judge_mode"] = "diagnosis"
        with self.assertRaisesRegex(UsageError, "different judge mode"):
            validate_store(corrupt, [])


if __name__ == "__main__":
    unittest.main()

skills

herdr-foreman

tests

__init__.py

fakes.py

test_assign.py

test_attention.py

test_billing.py

test_bounded_run.sh

test_capabilities.py

test_capability_routing.py

test_chronology.py

test_churn.py

test_classify.sh

test_claude_native.py

test_cli.py

test_compose_briefs.sh

test_composer.py

test_composition.py

test_config.py

test_continuity_cli.py

test_cost_report.py

test_diagnostics.py

test_engagement.py

test_entrypoints.py

test_foreman_launcher.sh

test_foreman_queue.py

test_foreman_reset.py

test_foreman_seat.py

test_foreman_tier_check.py

test_freeze.py

test_herdr.py

test_historical.py

test_home.py

test_label_workspaces.sh

test_launch.py

test_legacy_recovery.py

test_lifecycle.py

test_load_set.py

test_measure.py

test_members.py

test_memory.py

test_minimum_adequate.py

test_oracle.py

test_parsers.py

test_partition.py

test_planner.py

test_probe.py

test_provision_worktree.sh

test_prune_remote_branches.sh

test_prune_report_caches.py

test_prune_result.py

test_prune_worktrees.sh

test_recovery_cli.py

test_recovery.py

test_renderable.py

test_report_contract.py

test_report_delivery.py

test_report_gates.py

test_report_verdict.py

test_reset_input_hook.py

test_resolve_gates.sh

test_resolve_policy_paths.py

test_restoration.py

test_retrospective_runtime.py

test_retrospective.py

test_review_package.py

test_role_clear.py

test_roster.sh

test_round_preflight.sh

test_runnable.py

test_scoring.py

test_script_dir_newline.sh

test_seat_holds.py

test_selection.py

test_skill_invocations.sh

test_slice_scope_parity.py

test_specialist_cli.py

test_specialist_delivery.py

test_specialist_recovery.py

test_specialist_retention.py

test_stale_grok_delivery.py

test_start_judge_worker.py

test_state.py

test_successors.py

test_supervision_cli.py

test_supervision_diagnostics.py

test_supervision_gate.py

test_supervision_replay.py

test_supervision.py

test_sweep_worktrees.sh

test_tier_integration.py

test_tiers.py

test_triggers.py

test_typesafe_client.py

test_verdict_gates.py

test_verify_authority.sh

test_wait_report.sh

tier_fixture.py

bounded-run.sh

compose-briefs.sh

config.example.json

foreman-tier-check.py

foreman.sh

label-workspaces.sh

provision-worktree.sh

prune-remote-branches.sh

prune-report-caches.py

prune-worktrees.sh

resolve-gates.sh

resolve-policy-paths.sh

review-package.sh

roster.sh

round-preflight.sh

SKILL.md

start-judge-worker.sh

state-schema.md

sweep-worktrees.sh

verify-authority.sh

wait-report.sh

README.md

tile.json