CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/coding-policy

General-purpose coding policy for Baruch's AI agents

73

Quality

91%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide

SecuritybySnyk

Low

Low-risk findings worth noting

Overview
Quality
Evals
Security
Files

test_capabilities.pyskills/herdr-foreman/tests/

"""The model-capability table: its cadence, its refresh, and what it refuses."""

import io
import json
from contextlib import redirect_stderr
from datetime import datetime, timedelta, timezone
import os as _os
import sys as _sys
import tempfile
import unittest
from unittest import mock
from pathlib import Path

_ROOT = _os.path.dirname(_os.path.dirname(_os.path.abspath(__file__)))
if _ROOT not in _sys.path:
    _sys.path.insert(0, _ROOT)

from foreman import capabilities
from foreman.cli import main
from foreman.errors import StateError, UsageError

#: A fixed past reference; every other instant and date is derived from it
#: (rules/testing-standards.md Determinism).
REFERENCE = datetime(2026, 1, 5, tzinfo=timezone.utc)


def instant(days=0, seconds=0):
    return (REFERENCE + timedelta(days=days, seconds=seconds)).isoformat()


def dated(days):
    return (REFERENCE + timedelta(days=days)).date().isoformat()


AT = instant()
LATER = instant(days=8)


def found(document, model, effort, capability):
    """The entry, or a failure naming the row the table does not hold."""
    row = capabilities.lookup(document, model, effort, capability)
    assert row is not None, "no entry for {} / {} / {}".format(model, effort, capability)
    return row


def entry(**overrides):
    row = {"model": "opus-5", "effort": "high", "capability": "independent-defect-detection",
           "verdict": "adequate",
           "source": {"kind": "benchmark", "ref": "SWE-bench Verified", "dated": dated(-20)}}
    row.update(overrides)
    return row


class CapabilityTableTest(unittest.TestCase):
    def setUp(self):
        self.tmp = tempfile.TemporaryDirectory()
        self.addCleanup(self.tmp.cleanup)
        self.state = Path(self.tmp.name) / "state.json"

    # -- reading -----------------------------------------------------------

    def test_an_absent_table_reads_empty_and_creates_no_file(self):
        self.assertEqual(capabilities.load(self.state), capabilities.empty())
        self.assertFalse(capabilities.storage_path(self.state).exists())

    def test_a_malformed_table_names_its_repair(self):
        capabilities.storage_path(self.state).write_text("{broken", encoding="utf-8")
        with self.assertRaisesRegex(UsageError, "not valid JSON"):
            capabilities.load(self.state)

    def test_a_newer_table_reads_as_no_prior_state_and_is_never_overwritten(self):
        # rules/stateful-artifacts.md: a lagging reader treats a newer record as
        # no usable prior state; a lagging writer must not clobber it.
        target = capabilities.storage_path(self.state)
        original = json.dumps({"schema_version": 99, "refreshed_at": None, "entries": [], "future": 1})
        target.write_text(original, encoding="utf-8")
        err = io.StringIO()
        with redirect_stderr(err):
            self.assertEqual(capabilities.load(self.state), capabilities.empty())
        self.assertIn("newer than this build", err.getvalue())
        with self.assertRaisesRegex(UsageError, "Update the coding-policy plugin before recording"):
            capabilities.record(self.state, {"entries": [entry()]}, AT)
        self.assertEqual(target.read_text(encoding="utf-8"), original)

    def test_a_dangling_table_link_is_refused_never_read_as_missing(self):
        target = capabilities.storage_path(self.state)
        missing = Path(self.tmp.name) / "moved" / "capabilities.json"
        target.symlink_to(missing)
        with self.assertRaisesRegex(UsageError, "is a symlink"):
            capabilities.load(self.state)
        with self.assertRaisesRegex(UsageError, "is a symlink"):
            capabilities.record(self.state, {"entries": [entry()]}, AT)
        self.assertTrue(target.is_symlink())
        self.assertEqual(_os.readlink(target), str(missing))
        self.assertFalse(missing.exists())

    def test_a_link_swapped_in_after_the_probe_is_never_read_through(self):
        # The probe sees no link (as if the swap landed just after it); the
        # read must still refuse to follow the link now at the table's path.
        real = Path(self.tmp.name) / "elsewhere.json"
        capabilities.record(real, {"entries": [entry()]}, AT)
        target = capabilities.storage_path(self.state)
        target.symlink_to(capabilities.storage_path(real))
        genuine = _os.lstat

        def probe_misses_the_swap(candidate, *args, **kwargs):
            if str(candidate) == str(target):
                raise FileNotFoundError(candidate)
            return genuine(candidate, *args, **kwargs)

        with mock.patch.object(capabilities.os, "lstat", probe_misses_the_swap):
            with self.assertRaisesRegex(UsageError, "is a symlink"):
                capabilities.load(self.state)
        self.assertTrue(target.is_symlink())

    def test_a_live_table_link_is_refused_and_its_target_left_as_found(self):
        real = Path(self.tmp.name) / "elsewhere.json"
        capabilities.record(real, {"entries": [entry()]}, AT)
        original = capabilities.storage_path(real).read_bytes()
        target = capabilities.storage_path(self.state)
        target.symlink_to(capabilities.storage_path(real))
        with self.assertRaisesRegex(UsageError, "is a symlink"):
            capabilities.load(self.state)
        with self.assertRaisesRegex(UsageError, "is a symlink"):
            capabilities.record(self.state, {"entries": [entry(verdict="unknown")]}, LATER)
        self.assertTrue(target.is_symlink())
        self.assertEqual(capabilities.storage_path(real).read_bytes(), original)

    @unittest.skipIf(_os.geteuid() == 0, "root searches any directory")
    def test_a_table_it_cannot_inspect_is_refused_with_its_repair(self):
        locked = Path(self.tmp.name) / "locked"
        locked.mkdir()
        state = locked / "state.json"
        _os.chmod(locked, 0)
        self.addCleanup(_os.chmod, locked, 0o755)
        with self.assertRaisesRegex(UsageError, "Cannot inspect the capability table"):
            capabilities.load(state)
        # The writer meets its lock first; either way a structured refusal.
        with self.assertRaises((UsageError, StateError)):
            capabilities.record(state, {"entries": [entry()]}, AT)

    def test_a_report_never_supplies_what_the_writer_stamps(self):
        for extra in ({"recorded_at": AT}, {"schema_version": 1}):
            with self.subTest(extra=extra), self.assertRaisesRegex(UsageError, "carries exactly"):
                capabilities.record(self.state, {"entries": [entry(**extra)]}, AT)

    def test_an_older_or_malformed_envelope_refuses_rather_than_guessing(self):
        target = capabilities.storage_path(self.state)
        for label, document, message in (
            ("older version", {"schema_version": 0, "refreshed_at": None, "entries": []},
             "Unsupported capability-table schema"),
            ("no refreshed_at", {"schema_version": 1, "entries": []}, "carries exactly"),
            ("stray key", {"schema_version": 1, "refreshed_at": None, "entries": [], "x": 1}, "carries exactly"),
            ("bad refreshed_at", {"schema_version": 1, "refreshed_at": "yesterday", "entries": []}, "timestamp"),
        ):
            with self.subTest(label=label):
                target.write_text(json.dumps(document), encoding="utf-8")
                with self.assertRaisesRegex(UsageError, message):
                    capabilities.load(self.state)

    # -- cadence -----------------------------------------------------------

    def test_a_fleet_that_dispatched_nothing_is_never_due(self):
        # The table is read at dispatch time, so a week with no rounds is a week
        # where a stale table is never consulted (#481).
        result = capabilities.cadence(capabilities.empty(), AT)
        self.assertEqual((result["due"], result["reason"]), (False, "no_recorded_work"))

    def test_recorded_work_with_no_table_is_due_immediately(self):
        result = capabilities.cadence(capabilities.empty(), AT, existing_work=True)
        self.assertEqual((result["due"], result["reason"]), (True, "never_refreshed"))

    def test_the_interval_is_a_week(self):
        capabilities.record(self.state, {"entries": [entry()]}, AT)
        document = capabilities.load(self.state)
        for when, due in ((instant(days=1), False),
                          (instant(days=7, seconds=-1), False),
                          (instant(days=7), True),
                          (LATER, True)):
            with self.subTest(when=when):
                self.assertEqual(capabilities.cadence(document, when)["due"], due)

    def test_a_checkpoint_before_the_saved_refresh_is_refused(self):
        capabilities.record(self.state, {"entries": [entry()]}, LATER)
        with self.assertRaisesRegex(UsageError, "precedes the saved refresh"):
            capabilities.cadence(capabilities.load(self.state), AT)

    # -- recording ---------------------------------------------------------

    def test_a_refresh_stamps_and_sorts_what_it_records(self):
        capabilities.record(self.state, {"entries": [
            entry(model="sonnet-5"), entry(model="haiku-4.5", verdict="unknown")]}, AT)
        document = capabilities.load(self.state)
        self.assertEqual([row["model"] for row in document["entries"]], ["haiku-4.5", "sonnet-5"])
        self.assertEqual(document["refreshed_at"], AT)
        self.assertTrue(all(row["recorded_at"] == AT for row in document["entries"]))

    def test_a_partial_refresh_leaves_every_other_row_untouched(self):
        # One report about two models must not retire the rest of the table.
        capabilities.record(self.state, {"entries": [
            entry(), entry(model="haiku-4.5", effort="low", verdict="inadequate",
                           source={"kind": "project", "ref": "coding-policy#324", "dated": dated(-69)})]}, AT)
        capabilities.record(self.state, {"entries": [entry(
            source={"kind": "evaluation", "ref": "third-party eval", "dated": dated(-1)})]}, LATER)
        document = capabilities.load(self.state)
        self.assertEqual(len(document["entries"]), 2)
        kept = found(document, "haiku-4.5", "low", "independent-defect-detection")
        replaced = found(document, "opus-5", "high", "independent-defect-detection")
        self.assertEqual(kept["source"]["ref"], "coding-policy#324")
        self.assertEqual(replaced["source"]["ref"], "third-party eval")

    def test_an_empty_report_refreshes_nothing(self):
        with self.assertRaisesRegex(UsageError, "at least one entry"):
            capabilities.record(self.state, {"entries": []}, AT)

    def test_one_refresh_cannot_record_a_key_twice(self):
        with self.assertRaisesRegex(UsageError, "twice"):
            capabilities.record(self.state, {"entries": [entry(), entry(verdict="unknown")]}, AT)

    # -- the source hierarchy ----------------------------------------------

    def test_a_vendor_claim_cannot_support_an_adequate_verdict(self):
        # A vendor's claim about its own model would route real work on
        # marketing; the hierarchy exists to keep that out (#480).
        with self.assertRaisesRegex(UsageError, "routes real work on marketing"):
            capabilities.record(self.state, {"entries": [entry(
                source={"kind": "vendor", "ref": "launch blog", "dated": dated(-1)})]}, AT)

    def test_a_vendor_claim_still_records_what_it_can_support(self):
        for verdict in ("inadequate", "unknown"):
            with self.subTest(verdict=verdict):
                capabilities.record(self.state, {"entries": [entry(
                    verdict=verdict,
                    source={"kind": "vendor", "ref": "deprecation notice", "dated": dated(-1)})]}, AT)

    def test_this_projects_own_result_supports_a_verdict(self):
        # A model that failed here is evidence, cited like any other source.
        capabilities.record(self.state, {"entries": [entry(
            model="weak-1", verdict="inadequate",
            source={"kind": "project", "ref": "coding-policy#324", "dated": dated(-69)})]}, AT)
        row = found(capabilities.load(self.state), "weak-1", "high", "independent-defect-detection")
        self.assertEqual(row["verdict"], "inadequate")

    # -- entry shape -------------------------------------------------------

    def test_an_entry_carries_exactly_its_fields(self):
        for broken in ({"model": "m"}, {**entry(), "extra": 1}):
            with self.subTest(broken=broken):
                with self.assertRaisesRegex(UsageError, "carries exactly"):
                    capabilities.record(self.state, {"entries": [broken]}, AT)

    def test_a_verdict_comes_from_the_fixed_set(self):
        with self.assertRaisesRegex(UsageError, "one of"):
            capabilities.record(self.state, {"entries": [entry(verdict="probably fine")]}, AT)

    def test_a_source_is_dated_so_a_stale_reading_is_visible(self):
        for dated in ("2026-9-1", "September 2026", "", "2026-09-01T00:00:00Z"):
            with self.subTest(dated=dated):
                with self.assertRaisesRegex(UsageError, "dated YYYY-MM-DD"):
                    capabilities.record(self.state, {"entries": [entry(
                        source={"kind": "benchmark", "ref": "x", "dated": dated})]}, AT)

    def test_the_cadence_check_refuses_an_unreadable_ledger(self):
        # An unreadable ledger is not an empty one: "no recorded work" would
        # wrongly report the table as not due.
        self.state.write_text("{broken", encoding="utf-8")
        out, err = io.StringIO(), io.StringIO()
        code = main(["capability-check", "--state", str(self.state), "--now", AT], stdout=out, stderr=err)
        self.assertEqual((code, out.getvalue()), (1, ""))
        self.assertIn("cannot be read", err.getvalue())
        self.assertEqual(self.state.read_text(encoding="utf-8"), "{broken")

    def test_a_source_date_is_a_real_calendar_day(self):
        for impossible in ("2026-02-31", "2026-99-99"):
            with self.subTest(dated=impossible), self.assertRaisesRegex(UsageError, "not a calendar date"):
                capabilities.record(self.state, {"entries": [entry(
                    source={"kind": "benchmark", "ref": "x", "dated": impossible})]}, AT)

    def test_a_source_names_where_it_was_read(self):
        with self.assertRaisesRegex(UsageError, "names where it was read"):
            capabilities.record(self.state, {"entries": [entry(
                source={"kind": "benchmark", "ref": "  ", "dated": dated(-20)})]}, AT)

    def test_a_record_carries_entries_alone(self):
        with self.assertRaisesRegex(UsageError, "carries `entries` alone"):
            capabilities.record(self.state, {"entries": [entry()], "refreshed_at": AT}, AT)

    def test_lookup_returns_none_for_a_row_the_table_does_not_hold(self):
        capabilities.record(self.state, {"entries": [entry()]}, AT)
        self.assertIsNone(capabilities.lookup(capabilities.load(self.state), "opus-5", "low", "rebase"))


if __name__ == "__main__":
    unittest.main()

skills

herdr-foreman

tests

__init__.py

fakes.py

test_assign.py

test_attention.py

test_billing.py

test_bounded_run.sh

test_capabilities.py

test_capability_routing.py

test_chronology.py

test_churn.py

test_classify.sh

test_claude_native.py

test_cli.py

test_compose_briefs.sh

test_composer.py

test_composition.py

test_config.py

test_continuity_cli.py

test_cost_report.py

test_diagnostics.py

test_engagement.py

test_entrypoints.py

test_foreman_launcher.sh

test_foreman_queue.py

test_foreman_reset.py

test_foreman_seat.py

test_foreman_tier_check.py

test_freeze.py

test_herdr.py

test_historical.py

test_home.py

test_label_workspaces.sh

test_launch.py

test_legacy_recovery.py

test_load_set.py

test_measure.py

test_members.py

test_memory.py

test_oracle.py

test_parsers.py

test_partition.py

test_planner.py

test_probe.py

test_provision_worktree.sh

test_prune_remote_branches.sh

test_prune_report_caches.py

test_prune_result.py

test_prune_worktrees.sh

test_recovery_cli.py

test_recovery.py

test_renderable.py

test_report_contract.py

test_report_delivery.py

test_report_gates.py

test_report_verdict.py

test_resolve_gates.sh

test_resolve_policy_paths.py

test_restoration.py

test_retrospective_runtime.py

test_retrospective.py

test_review_package.py

test_role_clear.py

test_roster.sh

test_round_preflight.sh

test_runnable.py

test_scoring.py

test_script_dir_newline.sh

test_seat_holds.py

test_selection.py

test_skill_invocations.sh

test_slice_scope_parity.py

test_specialist_cli.py

test_specialist_delivery.py

test_specialist_recovery.py

test_specialist_retention.py

test_stale_grok_delivery.py

test_start_judge_worker.py

test_state.py

test_supervision_cli.py

test_supervision_diagnostics.py

test_supervision_gate.py

test_supervision_replay.py

test_supervision.py

test_sweep_worktrees.sh

test_tier_integration.py

test_tiers.py

test_triggers.py

test_typesafe_client.py

test_verdict_gates.py

test_verify_authority.sh

test_wait_report.sh

tier_fixture.py

bounded-run.sh

compose-briefs.sh

config.example.json

foreman-tier-check.py

foreman.sh

label-workspaces.sh

provision-worktree.sh

prune-remote-branches.sh

prune-report-caches.py

prune-worktrees.sh

resolve-gates.sh

resolve-policy-paths.sh

review-package.sh

roster.sh

round-preflight.sh

SKILL.md

start-judge-worker.sh

state-schema.md

sweep-worktrees.sh

verify-authority.sh

wait-report.sh

README.md

tile.json