CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/coding-policy

General-purpose coding policy for Baruch's AI agents

73

Quality

91%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide

SecuritybySnyk

Low

Low-risk findings worth noting

Overview
Quality
Evals
Security
Files

test_scoring.pyskills/herdr-foreman/tests/

"""Scoring labels against recorded verdicts, and calibrating Jev's bands from held-out labels."""

import hashlib
import io
import json
import sys
import tempfile
import unittest
from contextlib import redirect_stdout
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "classify"))

import report_verdict  # noqa: E402 -- the classify dir is on sys.path only from here
import scoring  # noqa: E402
from foreman import report_gates  # noqa: E402

CHANGED = report_verdict.questions()["changed"]
QUESTION = report_verdict.question_hash()

IDS = ("names_open_item", "open_items_accepted", "open_items_out_of_scope", "concludes_nothing_blocks")


def jev(recorded, p_open, p_disposed=0.01, p_concludes=0.1):
    p = dict(zip(IDS, (p_open, p_disposed, p_disposed, p_concludes)))
    return {"agent": "jev", "model": report_gates.JEV_MODEL, "recorded": recorded, "source": "corpus",
            "question": QUESTION, "recorded_at": CHANGED + "T12:00:00+00:00",
            "verdict": "blocking", "answers": {qid: {"answer": report_gates.band(v), "p_yes": v} for qid, v in p.items()}}


class ScoreTest(unittest.TestCase):
    def test_per_question_truth_comes_from_the_recorded_verdict_or_the_fixture(self):
        rows = [jev("blocking", 0.99), jev("approved", 0.1, p_concludes=0.95),
                {**jev("approved", 0.1, p_concludes=0.95), "source": "fixture",
                 "expected_answers": {"names_open_item": "no", "concludes_nothing_blocks": "yes"}}]
        for row in rows:
            row["verdict"] = "blocking" if row["recorded"] == "blocking" else "approved"
        report = scoring.score(rows, 0, "jev", "pinned", {"since": "2026-09-27", "changed": "2026-09-27", "held_out": True})
        self.assertEqual((report["scored"], report["accuracy"]), (2, 1.0))
        # The blocking row fixes the open-item answers; the approved row, with
        # neither disposal answered yes, fixes its no-open-item path.
        self.assertEqual(report["per_question"]["names_open_item"]["determined"], 2)
        self.assertEqual(report["per_question"]["concludes_nothing_blocks"],
                         {"determined": 1, "agree": 1, "unclear": 0, "accuracy": 1.0})
        self.assertEqual(report["fixtures"]["per_question"]["concludes_nothing_blocks"]["accuracy"], 1.0)
        self.assertEqual(report["fixtures"]["flipped"], [])

    def test_an_approval_through_a_disposal_fixes_no_answer(self):
        row = jev("approved", 0.99, p_disposed=0.9)
        row["verdict"] = "approved"
        table = scoring.per_question([row])
        self.assertEqual([table[qid]["determined"] for qid in IDS], [0, 0, 0, 0])

    def test_a_wrong_conclusion_on_an_approval_is_scored_wrong(self):
        row = jev("approved", 0.1, p_concludes=0.1)
        row["verdict"] = "insufficient_evidence"
        self.assertEqual(scoring.per_question([row])["concludes_nothing_blocks"],
                         {"determined": 1, "agree": 0, "unclear": 0, "accuracy": 0.0})


class CalibrateTest(unittest.TestCase):
    def test_too_few_held_out_labels_calibrate_nothing(self):
        with self.assertRaises(SystemExit) as caught:
            scoring.calibrate([jev("blocking", 0.99)] * (scoring.MIN_CALIBRATION_REPORTS - 1))
        self.assertEqual(caught.exception.code, 2)

    def test_the_block_band_never_gates_a_recorded_approval(self):
        rows = ([jev("blocking", 0.995)] * 15 + [jev("blocking", 0.9)] * 5
                + [jev("approved", 0.93, p_concludes=0.9)] * 3 + [jev("approved", 0.05, p_concludes=0.9)] * 12)
        result = scoring.calibrate(rows)
        self.assertEqual(result["counts"]["labels"], 35)
        proposed = result["proposed"]
        self.assertEqual(proposed["block"]["gated_approved"], 0)
        self.assertEqual(proposed["block"]["gated_blocking"], 15)
        self.assertLessEqual(proposed["reread"]["gated_approved"], scoring.MAX_FALSE_REREAD_RATE * 15)
        self.assertEqual(proposed["annotation"]["missed_blockers"], 0)

    def test_llm_and_fixture_labels_are_not_calibration_data(self):
        rows = ([{**jev("blocking", 0.99), "agent": "claude"}] * 40 + [{**jev("blocking", 0.99), "source": "fixture"}] * 40
                + [{**jev("blocking", 0.99), "model": "jev-other"}] * 40)
        with self.assertRaises(SystemExit):
            scoring.calibrate(rows)

    def test_labels_that_are_not_held_out_are_excluded(self):
        good = [jev("blocking", 0.995)] * 20 + [jev("approved", 0.05, p_concludes=0.9)] * 10
        stale = [{**jev("blocking", 0.99), "recorded_at": "2026-01-01T00:00:00+00:00"}] * 40
        other = [{**jev("blocking", 0.99), "question": "0" * 64}] * 40
        undated = [{**jev("blocking", 0.99), "recorded_at": None}] * 40
        with self.assertRaises(SystemExit):
            scoring.calibrate(good[:-1] + stale + other + undated)
        result = scoring.calibrate(good + stale + other + undated)
        self.assertEqual((result["counts"]["labels"], result["counts"]["excluded"]), (30, 120))

    def test_calibrate_prints_a_proposal_and_writes_nothing(self):
        rows = [jev("blocking", 0.995)] * 20 + [jev("approved", 0.05, p_concludes=0.9)] * 10
        owner = Path(report_gates.__file__).read_bytes()
        with tempfile.TemporaryDirectory(prefix="calibrate-") as root:
            results = Path(root) / "results.json"
            results.write_text(json.dumps(rows))
            before = sorted(path.name for path in Path(root).iterdir())
            out = io.StringIO()
            with redirect_stdout(out):
                self.assertEqual(scoring.main(["calibrate", str(results)]), 0)
            self.assertEqual(sorted(path.name for path in Path(root).iterdir()), before)
        proposal = json.loads(out.getvalue())
        self.assertEqual(set(proposal["proposed"]), {"annotation", "block", "reread"})
        self.assertEqual(hashlib.sha256(Path(report_gates.__file__).read_bytes()).digest(),
                         hashlib.sha256(owner).digest())


if __name__ == "__main__":
    sys.exit(0 if unittest.main(exit=False).result.wasSuccessful() else 1)

skills

herdr-foreman

tests

__init__.py

fakes.py

test_assign.py

test_attention.py

test_billing.py

test_bounded_run.sh

test_capabilities.py

test_capability_routing.py

test_chronology.py

test_churn.py

test_classify.sh

test_claude_native.py

test_cli.py

test_compose_briefs.sh

test_composer.py

test_composition.py

test_config.py

test_continuity_cli.py

test_cost_report.py

test_diagnostics.py

test_engagement.py

test_entrypoints.py

test_foreman_launcher.sh

test_foreman_queue.py

test_foreman_reset.py

test_foreman_seat.py

test_foreman_tier_check.py

test_freeze.py

test_herdr.py

test_historical.py

test_home.py

test_label_workspaces.sh

test_launch.py

test_legacy_recovery.py

test_load_set.py

test_measure.py

test_members.py

test_memory.py

test_oracle.py

test_parsers.py

test_partition.py

test_planner.py

test_probe.py

test_provision_worktree.sh

test_prune_remote_branches.sh

test_prune_report_caches.py

test_prune_result.py

test_prune_worktrees.sh

test_recovery_cli.py

test_recovery.py

test_renderable.py

test_report_contract.py

test_report_delivery.py

test_report_gates.py

test_report_verdict.py

test_resolve_gates.sh

test_resolve_policy_paths.py

test_restoration.py

test_retrospective_runtime.py

test_retrospective.py

test_review_package.py

test_role_clear.py

test_roster.sh

test_round_preflight.sh

test_runnable.py

test_scoring.py

test_script_dir_newline.sh

test_seat_holds.py

test_selection.py

test_skill_invocations.sh

test_slice_scope_parity.py

test_specialist_cli.py

test_specialist_delivery.py

test_specialist_recovery.py

test_specialist_retention.py

test_stale_grok_delivery.py

test_start_judge_worker.py

test_state.py

test_supervision_cli.py

test_supervision_diagnostics.py

test_supervision_gate.py

test_supervision_replay.py

test_supervision.py

test_sweep_worktrees.sh

test_tier_integration.py

test_tiers.py

test_triggers.py

test_typesafe_client.py

test_verdict_gates.py

test_verify_authority.sh

test_wait_report.sh

tier_fixture.py

bounded-run.sh

compose-briefs.sh

config.example.json

foreman-tier-check.py

foreman.sh

label-workspaces.sh

provision-worktree.sh

prune-remote-branches.sh

prune-report-caches.py

prune-worktrees.sh

resolve-gates.sh

resolve-policy-paths.sh

review-package.sh

roster.sh

round-preflight.sh

SKILL.md

start-judge-worker.sh

state-schema.md

sweep-worktrees.sh

verify-authority.sh

wait-report.sh

README.md

tile.json