CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/coding-policy

General-purpose coding policy for Baruch's AI agents

74

Quality

93%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide
SecuritybySnyk

Medium

Suggest reviewing before use

Overview
Quality
Evals
Security
Files

test_scoring.pyskills/herdr-foreman/tests/

"""Scoring labels against recorded verdicts, and calibrating Jev's bands from held-out labels."""

import hashlib
import io
import json
import sys
import tempfile
import unittest
from contextlib import redirect_stdout
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "classify"))

import report_verdict  # noqa: E402 -- the classify dir is on sys.path only from here
import scoring  # noqa: E402
from foreman import report_gates  # noqa: E402

CHANGED = report_verdict.questions()["changed"]
QUESTION = report_verdict.question_hash()

IDS = ("names_open_item", "open_items_accepted", "open_items_out_of_scope", "concludes_nothing_blocks")


def jev(recorded, p_open, p_disposed=0.01, p_concludes=0.1):
    p = dict(zip(IDS, (p_open, p_disposed, p_disposed, p_concludes)))
    return {"agent": "jev", "model": report_gates.JEV_MODEL, "recorded": recorded, "source": "corpus",
            "question": QUESTION, "recorded_at": CHANGED + "T12:00:00+00:00",
            "verdict": "blocking", "answers": {qid: {"answer": report_gates.band(v), "p_yes": v} for qid, v in p.items()}}


class ScoreTest(unittest.TestCase):
    def test_per_question_truth_comes_from_the_recorded_verdict_or_the_fixture(self):
        rows = [jev("blocking", 0.99), jev("approved", 0.1, p_concludes=0.95),
                {**jev("approved", 0.1, p_concludes=0.95), "source": "fixture",
                 "expected_answers": {"names_open_item": "no", "concludes_nothing_blocks": "yes"}}]
        for row in rows:
            row["verdict"] = "blocking" if row["recorded"] == "blocking" else "approved"
        report = scoring.score(rows, 0, "jev", "pinned", {"since": "2026-09-27", "changed": "2026-09-27", "held_out": True})
        self.assertEqual((report["scored"], report["accuracy"]), (2, 1.0))
        # The blocking row fixes the open-item answers; the approved row, with
        # neither disposal answered yes, fixes its no-open-item path.
        self.assertEqual(report["per_question"]["names_open_item"]["determined"], 2)
        self.assertEqual(report["per_question"]["concludes_nothing_blocks"],
                         {"determined": 1, "agree": 1, "unclear": 0, "accuracy": 1.0})
        self.assertEqual(report["fixtures"]["per_question"]["concludes_nothing_blocks"]["accuracy"], 1.0)
        self.assertEqual(report["fixtures"]["flipped"], [])

    def test_an_approval_through_a_disposal_fixes_no_answer(self):
        row = jev("approved", 0.99, p_disposed=0.9)
        row["verdict"] = "approved"
        table = scoring.per_question([row])
        self.assertEqual([table[qid]["determined"] for qid in IDS], [0, 0, 0, 0])

    def test_a_wrong_conclusion_on_an_approval_is_scored_wrong(self):
        row = jev("approved", 0.1, p_concludes=0.1)
        row["verdict"] = "insufficient_evidence"
        self.assertEqual(scoring.per_question([row])["concludes_nothing_blocks"],
                         {"determined": 1, "agree": 0, "unclear": 0, "accuracy": 0.0})


class CalibrateTest(unittest.TestCase):
    def test_too_few_held_out_labels_calibrate_nothing(self):
        with self.assertRaises(SystemExit) as caught:
            scoring.calibrate([jev("blocking", 0.99)] * (scoring.MIN_CALIBRATION_REPORTS - 1))
        self.assertEqual(caught.exception.code, 2)

    def test_the_block_band_never_gates_a_recorded_approval(self):
        rows = ([jev("blocking", 0.995)] * 15 + [jev("blocking", 0.9)] * 5
                + [jev("approved", 0.93, p_concludes=0.9)] * 3 + [jev("approved", 0.05, p_concludes=0.9)] * 12)
        result = scoring.calibrate(rows)
        self.assertEqual(result["counts"]["labels"], 35)
        proposed = result["proposed"]
        self.assertEqual(proposed["block"]["gated_approved"], 0)
        self.assertEqual(proposed["block"]["gated_blocking"], 15)
        self.assertLessEqual(proposed["reread"]["gated_approved"], scoring.MAX_FALSE_REREAD_RATE * 15)
        self.assertEqual(proposed["annotation"]["missed_blockers"], 0)

    def test_llm_and_fixture_labels_are_not_calibration_data(self):
        rows = ([{**jev("blocking", 0.99), "agent": "claude"}] * 40 + [{**jev("blocking", 0.99), "source": "fixture"}] * 40
                + [{**jev("blocking", 0.99), "model": "jev-other"}] * 40)
        with self.assertRaises(SystemExit):
            scoring.calibrate(rows)

    def test_labels_that_are_not_held_out_are_excluded(self):
        good = [jev("blocking", 0.995)] * 20 + [jev("approved", 0.05, p_concludes=0.9)] * 10
        stale = [{**jev("blocking", 0.99), "recorded_at": "2026-01-01T00:00:00+00:00"}] * 40
        other = [{**jev("blocking", 0.99), "question": "0" * 64}] * 40
        undated = [{**jev("blocking", 0.99), "recorded_at": None}] * 40
        with self.assertRaises(SystemExit):
            scoring.calibrate(good[:-1] + stale + other + undated)
        result = scoring.calibrate(good + stale + other + undated)
        self.assertEqual((result["counts"]["labels"], result["counts"]["excluded"]), (30, 120))

    def test_calibrate_prints_a_proposal_and_writes_nothing(self):
        rows = [jev("blocking", 0.995)] * 20 + [jev("approved", 0.05, p_concludes=0.9)] * 10
        owner = Path(report_gates.__file__).read_bytes()
        with tempfile.TemporaryDirectory(prefix="calibrate-") as root:
            results = Path(root) / "results.json"
            results.write_text(json.dumps(rows))
            before = sorted(path.name for path in Path(root).iterdir())
            out = io.StringIO()
            with redirect_stdout(out):
                self.assertEqual(scoring.main(["calibrate", str(results)]), 0)
            self.assertEqual(sorted(path.name for path in Path(root).iterdir()), before)
        proposal = json.loads(out.getvalue())
        self.assertEqual(set(proposal["proposed"]), {"annotation", "block", "reread"})
        self.assertEqual(hashlib.sha256(Path(report_gates.__file__).read_bytes()).digest(),
                         hashlib.sha256(owner).digest())


if __name__ == "__main__":
    sys.exit(0 if unittest.main(exit=False).result.wasSuccessful() else 1)

skills

herdr-foreman

tests

__init__.py

fakes.py

test_assign.py

test_attention.py

test_billing.py

test_bounded_run.sh

test_capabilities.py

test_capability_routing.py

test_chronology.py

test_churn.py

test_classify.sh

test_claude_native.py

test_cli.py

test_compose_briefs.sh

test_composer.py

test_composition.py

test_config.py

test_continuity_cli.py

test_cost_report.py

test_diagnostics.py

test_engagement.py

test_entrypoints.py

test_foreman_launcher.sh

test_foreman_queue.py

test_foreman_reset.py

test_foreman_seat.py

test_foreman_tier_check.py

test_freeze.py

test_herdr.py

test_historical.py

test_home.py

test_label_workspaces.sh

test_launch.py

test_legacy_recovery.py

test_lifecycle.py

test_load_set.py

test_measure.py

test_members.py

test_memory.py

test_minimum_adequate.py

test_oracle.py

test_parsers.py

test_partition.py

test_planner.py

test_probe.py

test_provision_worktree.sh

test_prune_remote_branches.sh

test_prune_report_caches.py

test_prune_result.py

test_prune_worktrees.sh

test_recovery_cli.py

test_recovery.py

test_renderable.py

test_report_contract.py

test_report_delivery.py

test_report_gates.py

test_report_verdict.py

test_reset_input_hook.py

test_resolve_gates.sh

test_resolve_policy_paths.py

test_restoration.py

test_retrospective_runtime.py

test_retrospective.py

test_review_package.py

test_role_clear.py

test_roster.sh

test_round_preflight.sh

test_runnable.py

test_scoring.py

test_script_dir_newline.sh

test_seat_holds.py

test_selection.py

test_skill_invocations.sh

test_slice_scope_parity.py

test_specialist_cli.py

test_specialist_delivery.py

test_specialist_recovery.py

test_specialist_retention.py

test_stale_grok_delivery.py

test_start_judge_worker.py

test_state.py

test_successors.py

test_supervision_cli.py

test_supervision_diagnostics.py

test_supervision_gate.py

test_supervision_replay.py

test_supervision.py

test_sweep_worktrees.sh

test_tier_integration.py

test_tiers.py

test_triggers.py

test_typesafe_client.py

test_verdict_gates.py

test_verify_authority.sh

test_wait_report.sh

tier_fixture.py

bounded-run.sh

compose-briefs.sh

config.example.json

foreman-tier-check.py

foreman.sh

label-workspaces.sh

provision-worktree.sh

prune-remote-branches.sh

prune-report-caches.py

prune-worktrees.sh

resolve-gates.sh

resolve-policy-paths.sh

review-package.sh

roster.sh

round-preflight.sh

SKILL.md

start-judge-worker.sh

state-schema.md

sweep-worktrees.sh

verify-authority.sh

wait-report.sh

README.md

tile.json