CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/coding-policy

General-purpose coding policy for Baruch's AI agents

74

Quality

93%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide
SecuritybySnyk

Medium

Suggest reviewing before use

Overview
Quality
Evals
Security
Files

test_capabilities.pyskills/herdr-foreman/tests/

"""The model-capability table: its cadence, its refresh, and what it refuses."""

import io
import json
from contextlib import redirect_stderr
from datetime import datetime, timedelta, timezone
import os as _os
import sys as _sys
import tempfile
import unittest
from unittest import mock
from pathlib import Path

_ROOT = _os.path.dirname(_os.path.dirname(_os.path.abspath(__file__)))
if _ROOT not in _sys.path:
    _sys.path.insert(0, _ROOT)

from foreman import capabilities
from foreman.cli import main
from foreman.errors import StateError, UsageError

#: A fixed past reference; every other instant and date is derived from it
#: (rules/testing-standards.md Determinism).
REFERENCE = datetime(2026, 1, 5, tzinfo=timezone.utc)


def instant(days=0, seconds=0):
    return (REFERENCE + timedelta(days=days, seconds=seconds)).isoformat()


def dated(days):
    return (REFERENCE + timedelta(days=days)).date().isoformat()


AT = instant()
LATER = instant(days=8)


def found(document, model, effort, capability):
    """The entry, or a failure naming the row the table does not hold."""
    row = capabilities.lookup(document, model, effort, capability)
    assert row is not None, "no entry for {} / {} / {}".format(model, effort, capability)
    return row


def entry(**overrides):
    row = {"model": "opus-5", "effort": "high", "capability": "independent-defect-detection",
           "verdict": "adequate",
           "source": {"kind": "benchmark", "ref": "SWE-bench Verified", "dated": dated(-20)}}
    row.update(overrides)
    return row


class CapabilityTableTest(unittest.TestCase):
    def setUp(self):
        self.tmp = tempfile.TemporaryDirectory()
        self.addCleanup(self.tmp.cleanup)
        self.state = Path(self.tmp.name) / "state.json"

    # -- reading -----------------------------------------------------------

    def test_an_absent_table_reads_empty_and_creates_no_file(self):
        self.assertEqual(capabilities.load(self.state), capabilities.empty())
        self.assertFalse(capabilities.storage_path(self.state).exists())

    def test_a_malformed_table_names_its_repair(self):
        capabilities.storage_path(self.state).write_text("{broken", encoding="utf-8")
        with self.assertRaisesRegex(UsageError, "not valid JSON"):
            capabilities.load(self.state)

    def test_a_newer_gate_table_refuses_and_is_never_overwritten(self):
        # The documented gate-store exception preserves unknown negative gates;
        # neither a reader nor a lagging writer may discard or clobber them.
        target = capabilities.storage_path(self.state)
        original = json.dumps({"schema_version": 99, "refreshed_at": None, "entries": [], "future": 1})
        target.write_text(original, encoding="utf-8")
        with self.assertRaisesRegex(UsageError, "newer than this build"):
            capabilities.load(self.state)
        with self.assertRaisesRegex(UsageError, "Update the coding-policy plugin before recording"):
            capabilities.record(self.state, {"entries": [entry()]}, AT)
        self.assertEqual(target.read_text(encoding="utf-8"), original)

    def test_a_dangling_table_link_is_refused_never_read_as_missing(self):
        target = capabilities.storage_path(self.state)
        missing = Path(self.tmp.name) / "moved" / "capabilities.json"
        target.symlink_to(missing)
        with self.assertRaisesRegex(UsageError, "is a symlink"):
            capabilities.load(self.state)
        with self.assertRaisesRegex(UsageError, "is a symlink"):
            capabilities.record(self.state, {"entries": [entry()]}, AT)
        self.assertTrue(target.is_symlink())
        self.assertEqual(_os.readlink(target), str(missing))
        self.assertFalse(missing.exists())

    def test_a_link_swapped_in_after_the_probe_is_never_read_through(self):
        # The probe sees no link (as if the swap landed just after it); the
        # read must still refuse to follow the link now at the table's path.
        real = Path(self.tmp.name) / "elsewhere.json"
        capabilities.record(real, {"entries": [entry()]}, AT)
        target = capabilities.storage_path(self.state)
        target.symlink_to(capabilities.storage_path(real))
        genuine = _os.lstat

        def probe_misses_the_swap(candidate, *args, **kwargs):
            if str(candidate) == str(target):
                raise FileNotFoundError(candidate)
            return genuine(candidate, *args, **kwargs)

        with mock.patch.object(capabilities.os, "lstat", probe_misses_the_swap):
            with self.assertRaisesRegex(UsageError, "is a symlink"):
                capabilities.load(self.state)
        self.assertTrue(target.is_symlink())

    def test_a_live_table_link_is_refused_and_its_target_left_as_found(self):
        real = Path(self.tmp.name) / "elsewhere.json"
        capabilities.record(real, {"entries": [entry()]}, AT)
        original = capabilities.storage_path(real).read_bytes()
        target = capabilities.storage_path(self.state)
        target.symlink_to(capabilities.storage_path(real))
        with self.assertRaisesRegex(UsageError, "is a symlink"):
            capabilities.load(self.state)
        with self.assertRaisesRegex(UsageError, "is a symlink"):
            capabilities.record(self.state, {"entries": [entry(verdict="unknown")]}, LATER)
        self.assertTrue(target.is_symlink())
        self.assertEqual(capabilities.storage_path(real).read_bytes(), original)

    @unittest.skipIf(_os.geteuid() == 0, "root searches any directory")
    def test_a_table_it_cannot_inspect_is_refused_with_its_repair(self):
        locked = Path(self.tmp.name) / "locked"
        locked.mkdir()
        state = locked / "state.json"
        _os.chmod(locked, 0)
        self.addCleanup(_os.chmod, locked, 0o755)
        with self.assertRaisesRegex(UsageError, "Cannot inspect the capability table"):
            capabilities.load(state)
        # The writer meets its lock first; either way a structured refusal.
        with self.assertRaises((UsageError, StateError)):
            capabilities.record(state, {"entries": [entry()]}, AT)

    def test_a_report_never_supplies_what_the_writer_stamps(self):
        for extra in ({"recorded_at": AT}, {"schema_version": 1}):
            with self.subTest(extra=extra), self.assertRaisesRegex(UsageError, "carries exactly"):
                capabilities.record(self.state, {"entries": [entry(**extra)]}, AT)

    def test_an_older_or_malformed_envelope_refuses_rather_than_guessing(self):
        target = capabilities.storage_path(self.state)
        for label, document, message in (
            ("older version", {"schema_version": 0, "refreshed_at": None, "entries": []},
             "Unsupported capability-table schema"),
            ("no refreshed_at", {"schema_version": 1, "entries": []}, "carries exactly"),
            ("stray key", {"schema_version": 1, "refreshed_at": None, "entries": [], "x": 1}, "carries exactly"),
            ("bad refreshed_at", {"schema_version": 1, "refreshed_at": "yesterday", "entries": []}, "timestamp"),
        ):
            with self.subTest(label=label):
                target.write_text(json.dumps(document), encoding="utf-8")
                with self.assertRaisesRegex(UsageError, message):
                    capabilities.load(self.state)

    # -- cadence -----------------------------------------------------------

    def test_a_fleet_that_dispatched_nothing_is_never_due(self):
        # The table is read at dispatch time, so a week with no rounds is a week
        # where a stale table is never consulted (#481).
        result = capabilities.cadence(capabilities.empty(), AT)
        self.assertEqual((result["due"], result["reason"]), (False, "no_recorded_work"))

    def test_recorded_work_with_no_table_is_due_immediately(self):
        result = capabilities.cadence(capabilities.empty(), AT, existing_work=True)
        self.assertEqual((result["due"], result["reason"]), (True, "never_refreshed"))

    def test_the_interval_is_a_week(self):
        capabilities.record(self.state, {"entries": [entry()]}, AT)
        document = capabilities.load(self.state)
        for when, due in ((instant(days=1), False),
                          (instant(days=7, seconds=-1), False),
                          (instant(days=7), True),
                          (LATER, True)):
            with self.subTest(when=when):
                self.assertEqual(capabilities.cadence(document, when)["due"], due)

    def test_a_checkpoint_before_the_saved_refresh_is_refused(self):
        capabilities.record(self.state, {"entries": [entry()]}, LATER)
        with self.assertRaisesRegex(UsageError, "precedes the saved refresh"):
            capabilities.cadence(capabilities.load(self.state), AT)

    # -- recording ---------------------------------------------------------

    def test_a_refresh_stamps_and_sorts_what_it_records(self):
        capabilities.record(self.state, {"entries": [
            entry(model="sonnet-5"), entry(model="haiku-4.5", verdict="unknown")]}, AT)
        document = capabilities.load(self.state)
        self.assertEqual([row["model"] for row in document["entries"]], ["haiku-4.5", "sonnet-5"])
        self.assertEqual(document["refreshed_at"], AT)
        self.assertTrue(all(row["recorded_at"] == AT for row in document["entries"]))

    def test_a_partial_refresh_leaves_every_other_row_untouched(self):
        # One report about two models must not retire the rest of the table.
        capabilities.record(self.state, {"entries": [
            entry(), entry(model="haiku-4.5", effort="low", verdict="inadequate",
                           source={"kind": "project", "ref": "coding-policy#324", "dated": dated(-69)})]}, AT)
        capabilities.record(self.state, {"entries": [entry(
            source={"kind": "evaluation", "ref": "third-party eval", "dated": dated(-1)})]}, LATER)
        document = capabilities.load(self.state)
        self.assertEqual(len(document["entries"]), 2)
        kept = found(document, "haiku-4.5", "low", "independent-defect-detection")
        replaced = found(document, "opus-5", "high", "independent-defect-detection")
        self.assertEqual(kept["source"]["ref"], "coding-policy#324")
        self.assertEqual(replaced["source"]["ref"], "third-party eval")

    def test_an_empty_report_refreshes_nothing(self):
        with self.assertRaisesRegex(UsageError, "at least one entry"):
            capabilities.record(self.state, {"entries": []}, AT)

    def test_one_refresh_cannot_record_a_key_twice(self):
        with self.assertRaisesRegex(UsageError, "twice"):
            capabilities.record(self.state, {"entries": [entry(), entry(verdict="unknown")]}, AT)

    # -- the source hierarchy ----------------------------------------------

    def test_a_vendor_claim_cannot_support_an_adequate_verdict(self):
        # A vendor's claim about its own model would route real work on
        # marketing; the hierarchy exists to keep that out (#480).
        with self.assertRaisesRegex(UsageError, "routes real work on marketing"):
            capabilities.record(self.state, {"entries": [entry(
                source={"kind": "vendor", "ref": "launch blog", "dated": dated(-1)})]}, AT)

    def test_a_vendor_claim_still_records_what_it_can_support(self):
        for verdict in ("inadequate", "unknown"):
            with self.subTest(verdict=verdict):
                capabilities.record(self.state, {"entries": [entry(
                    verdict=verdict,
                    source={"kind": "vendor", "ref": "deprecation notice", "dated": dated(-1)})]}, AT)

    def test_this_projects_own_result_supports_a_verdict(self):
        # A model that failed here is evidence, cited like any other source.
        capabilities.record(self.state, {"entries": [entry(
            model="weak-1", verdict="inadequate",
            source={"kind": "project", "ref": "coding-policy#324", "dated": dated(-69)})]}, AT)
        row = found(capabilities.load(self.state), "weak-1", "high", "independent-defect-detection")
        self.assertEqual(row["verdict"], "inadequate")

    # -- entry shape -------------------------------------------------------

    def test_an_entry_carries_exactly_its_fields(self):
        for broken in ({"model": "m"}, {**entry(), "extra": 1}):
            with self.subTest(broken=broken):
                with self.assertRaisesRegex(UsageError, "carries exactly"):
                    capabilities.record(self.state, {"entries": [broken]}, AT)

    def test_a_verdict_comes_from_the_fixed_set(self):
        with self.assertRaisesRegex(UsageError, "one of"):
            capabilities.record(self.state, {"entries": [entry(verdict="probably fine")]}, AT)

    def test_a_source_is_dated_so_a_stale_reading_is_visible(self):
        for dated in ("2026-9-1", "September 2026", "", "2026-09-01T00:00:00Z"):
            with self.subTest(dated=dated):
                with self.assertRaisesRegex(UsageError, "dated YYYY-MM-DD"):
                    capabilities.record(self.state, {"entries": [entry(
                        source={"kind": "benchmark", "ref": "x", "dated": dated})]}, AT)

    def test_the_cadence_check_refuses_an_unreadable_ledger(self):
        # An unreadable ledger is not an empty one: "no recorded work" would
        # wrongly report the table as not due.
        self.state.write_text("{broken", encoding="utf-8")
        out, err = io.StringIO(), io.StringIO()
        code = main(["capability-check", "--state", str(self.state), "--now", AT], stdout=out, stderr=err)
        self.assertEqual((code, out.getvalue()), (1, ""))
        self.assertIn("cannot be read", err.getvalue())
        self.assertEqual(self.state.read_text(encoding="utf-8"), "{broken")

    def test_a_source_date_is_a_real_calendar_day(self):
        for impossible in ("2026-02-31", "2026-99-99"):
            with self.subTest(dated=impossible), self.assertRaisesRegex(UsageError, "not a calendar date"):
                capabilities.record(self.state, {"entries": [entry(
                    source={"kind": "benchmark", "ref": "x", "dated": impossible})]}, AT)

    def test_a_source_names_where_it_was_read(self):
        with self.assertRaisesRegex(UsageError, "names where it was read"):
            capabilities.record(self.state, {"entries": [entry(
                source={"kind": "benchmark", "ref": "  ", "dated": dated(-20)})]}, AT)

    def test_a_record_carries_entries_alone(self):
        with self.assertRaisesRegex(UsageError, "carries `entries` alone"):
            capabilities.record(self.state, {"entries": [entry()], "refreshed_at": AT}, AT)

    def test_lookup_returns_none_for_a_row_the_table_does_not_hold(self):
        capabilities.record(self.state, {"entries": [entry()]}, AT)
        self.assertIsNone(capabilities.lookup(capabilities.load(self.state), "opus-5", "low", "rebase"))


if __name__ == "__main__":
    unittest.main()

skills

herdr-foreman

tests

__init__.py

fakes.py

test_assign.py

test_attention.py

test_billing.py

test_bounded_run.sh

test_capabilities.py

test_capability_routing.py

test_chronology.py

test_churn.py

test_classify.sh

test_claude_native.py

test_cli.py

test_compose_briefs.sh

test_composer.py

test_composition.py

test_config.py

test_continuity_cli.py

test_cost_report.py

test_diagnostics.py

test_engagement.py

test_entrypoints.py

test_foreman_launcher.sh

test_foreman_queue.py

test_foreman_reset.py

test_foreman_seat.py

test_foreman_tier_check.py

test_freeze.py

test_herdr.py

test_historical.py

test_home.py

test_label_workspaces.sh

test_launch.py

test_legacy_recovery.py

test_lifecycle.py

test_load_set.py

test_measure.py

test_members.py

test_memory.py

test_minimum_adequate.py

test_oracle.py

test_parsers.py

test_partition.py

test_planner.py

test_probe.py

test_provision_worktree.sh

test_prune_remote_branches.sh

test_prune_report_caches.py

test_prune_result.py

test_prune_worktrees.sh

test_recovery_cli.py

test_recovery.py

test_renderable.py

test_report_contract.py

test_report_delivery.py

test_report_gates.py

test_report_verdict.py

test_reset_input_hook.py

test_resolve_gates.sh

test_resolve_policy_paths.py

test_restoration.py

test_retrospective_runtime.py

test_retrospective.py

test_review_package.py

test_role_clear.py

test_roster.sh

test_round_preflight.sh

test_runnable.py

test_scoring.py

test_script_dir_newline.sh

test_seat_holds.py

test_selection.py

test_skill_invocations.sh

test_slice_scope_parity.py

test_specialist_cli.py

test_specialist_delivery.py

test_specialist_recovery.py

test_specialist_retention.py

test_stale_grok_delivery.py

test_start_judge_worker.py

test_state.py

test_successors.py

test_supervision_cli.py

test_supervision_diagnostics.py

test_supervision_gate.py

test_supervision_replay.py

test_supervision.py

test_sweep_worktrees.sh

test_tier_integration.py

test_tiers.py

test_triggers.py

test_typesafe_client.py

test_verdict_gates.py

test_verify_authority.sh

test_wait_report.sh

tier_fixture.py

bounded-run.sh

compose-briefs.sh

config.example.json

foreman-tier-check.py

foreman.sh

label-workspaces.sh

provision-worktree.sh

prune-remote-branches.sh

prune-report-caches.py

prune-worktrees.sh

resolve-gates.sh

resolve-policy-paths.sh

review-package.sh

roster.sh

round-preflight.sh

SKILL.md

start-judge-worker.sh

state-schema.md

sweep-worktrees.sh

verify-authority.sh

wait-report.sh

README.md

tile.json