CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/coding-policy

General-purpose coding policy for Baruch's AI agents

73

Quality

91%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide

SecuritybySnyk

Low

Low-risk findings worth noting

Overview
Quality
Evals
Security
Files

test_selection.pyskills/herdr-foreman/tests/

"""Every plan records why each assignment got its model and effort (#602)."""

import copy
import io
import json
import sys
import unittest
from pathlib import Path
from typing import Any

sys.path.insert(0, str(Path(__file__).resolve().parents[1]))

from foreman import capabilities
from foreman.planner import PLAN_SCHEMA_VERSION
from foreman.tiers import BUILD_FAILED_GATES, PRESSURE_HEADROOM_PCT, REVIEW_ROW_FIX_ROUND, escalation_conditions
from tests.test_capability_routing import entry, table
from tests.test_cli import CliCase, CONFIG
from tests.tier_fixture import AT


def fired(result):
    return {row["field"]: row["fired"] for row in result["conditions"]}


class EscalationConditionsTest(unittest.TestCase):
    def tier(self, round_type, kind="claude", tier_row=None, de_escalated=False):
        return {"round": round_type, "tier_row": tier_row or round_type, "kind": kind, "de_escalated": de_escalated}

    def test_a_build_names_every_condition_and_which_fired(self):
        result = escalation_conditions("developer", self.tier("build"),
                                       {"failed_gates": BUILD_FAILED_GATES, "risk_flags": ["a", "b"]}, None)
        self.assertEqual(fired(result), {"failed_gates": True, "fix_round": False, "risk_flags": True,
                                         "input_bytes": False, "prior_high_miss": False})
        gates = result["conditions"][0]
        self.assertEqual((gates["value"], gates["effect"]), (BUILD_FAILED_GATES, {"tier_row": "review"}))
        self.assertEqual(result["conditions"][2]["effect"], {"model": "top", "effort": "xhigh"})
        self.assertEqual(result["pressure"], {"declines_at_or_below_pct": PRESSURE_HEADROOM_PCT, "de_escalated": False})

    def test_a_late_fix_round_fires_the_review_row(self):
        result = escalation_conditions("developer", self.tier("fix"), {}, REVIEW_ROW_FIX_ROUND)
        self.assertTrue(fired(result)["fix_round"])
        self.assertNotIn("failed_gates", fired(result))

    def test_a_kind_without_a_higher_effort_escalates_the_model_alone(self):
        result = escalation_conditions("developer", self.tier("build", kind="grok"), {}, None)
        self.assertEqual(result["conditions"][-1]["effect"], {"model": "top"})

    def test_a_judgment_round_is_never_declined_under_pressure(self):
        result = escalation_conditions("reviewer", self.tier("review"), {}, None)
        self.assertIsNone(result["pressure"]["declines_at_or_below_pct"])

    def test_a_consultation_names_the_evidence_that_moves_it(self):
        result = escalation_conditions("investigator", self.tier("reconciliation"), {"diagnosis_input": True}, None)
        consult = [row for row in result["conditions"] if row["effect"] == {"round": "reconciliation"}]
        self.assertEqual([(row["field"], row["fired"]) for row in consult],
                         [("diagnosis_input", True), ("prior_high_miss", False)])

    def test_the_pinned_judge_escalates_on_nothing(self):
        result = escalation_conditions("judge", {"round": "judge", "tier_row": "judge"}, {"prior_high_miss": True}, 9)
        self.assertEqual(result, {"conditions": [], "pressure": {"declines_at_or_below_pct": None, "de_escalated": False}})


class PlanSelectionTest(CliCase):
    def setUp(self):
        super().setUp()
        self.settings = copy.deepcopy(CONFIG)
        self.settings["schema_version"] = 2
        self.settings["agents"] = self.settings["agents"][:1]
        self.settings["agents"][0]["tiers"] = {
            "build": {"model": "opus-5", "effort": "high", "multiplier": 3.0},
            "fix": {"model": "sonnet-5", "effort": "high", "multiplier": 1.0},
        }
        self.config.write_text(json.dumps(self.settings), encoding="utf-8")

    def record(self, *entries):
        capabilities.storage_path(self.state).write_text(json.dumps(table(*entries)), encoding="utf-8")

    def plan(self, *extra):
        self.out, self.err = io.StringIO(), io.StringIO()
        rc, output, error = self.run_cli(["plan", *self.base(), "--roles", "developer",
                                          "--snapshot", str(self.snapshot), "--now", AT, *extra])
        self.assertEqual(rc, 0, error)
        document: Any = json.loads(output)
        return document

    def test_the_plan_records_the_selection_for_its_assignment(self):
        self.record(entry("opus-5", "high", "implementation", "adequate"),
                    entry("sonnet-5", "high", "implementation", "inadequate"))
        document = self.plan()
        self.assertEqual(document["schema_version"], PLAN_SCHEMA_VERSION)
        record = document["selection"]["developer"]
        self.assertEqual((record["agent"], record["tiered"], record["model"], record["effort"], record["tier_row"]),
                         ("claude", True, "opus-5", "high", "build"))
        self.assertEqual(record["required_capabilities"], {"model": ["implementation"], "worker": []})
        self.assertEqual(record["capability"], "adequate")
        self.assertEqual(record["evidence"], [{"capability": "implementation", "verdict": "adequate",
                                               "source": {"kind": "project", "ref": "fixture", "dated": "2026-09-23"}}])
        # No isolated billing evidence: the cost is unknown, never assumed.
        self.assertEqual(record["cost"], {"billing_window": "unknown", "effective_multiplier": 3.0, "known": False})
        cheaper = record["cheaper"]
        self.assertIsNone(cheaper["floor"])
        self.assertEqual([(row["tier_row"], row["model"], row["verdict"]) for row in cheaper["candidates"]],
                         [("fix", "sonnet-5", "inadequate")])
        self.assertIn("conditions", record["escalation"])

    def test_a_cheaper_row_without_evidence_reads_unknown_with_no_source(self):
        record = self.plan()["selection"]["developer"]
        self.assertEqual(record["evidence"], [{"capability": "implementation", "verdict": "unknown", "source": None}])
        self.assertEqual([(row["verdict"], row["sources"]) for row in record["cheaper"]["candidates"]], [("unknown", [])])

    def test_round_context_reaches_the_escalation_record(self):
        context = self.tmp / "context.json"
        context.write_text(json.dumps({"developer": {"input_bytes": 1}}), encoding="utf-8")
        record = self.plan("--round-context", str(context))["selection"]["developer"]
        self.assertEqual(fired(record["escalation"])["input_bytes"], False)
        context.write_text(json.dumps({"developer": {"prior_high_miss": True}}), encoding="utf-8")
        self.settings["agents"][0]["tiers"]["review"] = {"model": "opus-5", "effort": "high", "multiplier": 3.0}
        self.config.write_text(json.dumps(self.settings), encoding="utf-8")
        record = self.plan("--round-context", str(context))["selection"]["developer"]
        self.assertTrue(fired(record["escalation"])["prior_high_miss"])
        self.assertEqual(record["effort"], "xhigh")


class UntieredSelectionTest(CliCase):
    def test_an_untiered_worker_records_unknown_rather_than_a_guess(self):
        rc, output, error = self.run_cli(["plan", *self.base(), "--roles", "developer,reviewer",
                                          "--snapshot", str(self.snapshot), "--now", AT])
        self.assertEqual(rc, 0, error)
        document = json.loads(output)
        self.assertEqual(set(document["selection"]), {"developer", "reviewer"})
        for record in document["selection"].values():
            self.assertFalse(record["tiered"])
            for field in ("model", "effort", "round", "tier_row", "capability", "evidence", "escalation"):
                self.assertEqual(record[field], "unknown", field)
            self.assertEqual(record["required_capabilities"]["model"], "unknown")
            self.assertEqual(record["cheaper"]["candidates"], "unknown")
            self.assertFalse(record["cost"]["known"])


if __name__ == "__main__":
    unittest.main()

skills

herdr-foreman

tests

__init__.py

fakes.py

test_assign.py

test_attention.py

test_billing.py

test_bounded_run.sh

test_capabilities.py

test_capability_routing.py

test_chronology.py

test_churn.py

test_classify.sh

test_claude_native.py

test_cli.py

test_compose_briefs.sh

test_composer.py

test_composition.py

test_config.py

test_continuity_cli.py

test_cost_report.py

test_diagnostics.py

test_engagement.py

test_entrypoints.py

test_foreman_launcher.sh

test_foreman_queue.py

test_foreman_reset.py

test_foreman_seat.py

test_foreman_tier_check.py

test_freeze.py

test_herdr.py

test_historical.py

test_home.py

test_label_workspaces.sh

test_launch.py

test_legacy_recovery.py

test_load_set.py

test_measure.py

test_members.py

test_memory.py

test_oracle.py

test_parsers.py

test_partition.py

test_planner.py

test_probe.py

test_provision_worktree.sh

test_prune_remote_branches.sh

test_prune_report_caches.py

test_prune_result.py

test_prune_worktrees.sh

test_recovery_cli.py

test_recovery.py

test_renderable.py

test_report_contract.py

test_report_delivery.py

test_report_gates.py

test_report_verdict.py

test_resolve_gates.sh

test_resolve_policy_paths.py

test_restoration.py

test_retrospective_runtime.py

test_retrospective.py

test_review_package.py

test_role_clear.py

test_roster.sh

test_round_preflight.sh

test_runnable.py

test_scoring.py

test_script_dir_newline.sh

test_seat_holds.py

test_selection.py

test_skill_invocations.sh

test_slice_scope_parity.py

test_specialist_cli.py

test_specialist_delivery.py

test_specialist_recovery.py

test_specialist_retention.py

test_stale_grok_delivery.py

test_start_judge_worker.py

test_state.py

test_supervision_cli.py

test_supervision_diagnostics.py

test_supervision_gate.py

test_supervision_replay.py

test_supervision.py

test_sweep_worktrees.sh

test_tier_integration.py

test_tiers.py

test_triggers.py

test_typesafe_client.py

test_verdict_gates.py

test_verify_authority.sh

test_wait_report.sh

tier_fixture.py

bounded-run.sh

compose-briefs.sh

config.example.json

foreman-tier-check.py

foreman.sh

label-workspaces.sh

provision-worktree.sh

prune-remote-branches.sh

prune-report-caches.py

prune-worktrees.sh

resolve-gates.sh

resolve-policy-paths.sh

review-package.sh

roster.sh

round-preflight.sh

SKILL.md

start-judge-worker.sh

state-schema.md

sweep-worktrees.sh

verify-authority.sh

wait-report.sh

README.md

tile.json