CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/coding-policy

General-purpose coding policy for Baruch's AI agents

74

Quality

93%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide
SecuritybySnyk

Medium

Suggest reviewing before use

Overview
Quality
Evals
Security
Files

test_composition.pyskills/herdr-foreman/tests/

"""Specialist selection respects eligibility and actual contribution history."""

import copy
import sys
import unittest
from pathlib import Path
from types import SimpleNamespace

sys.path.insert(0, str(Path(__file__).resolve().parents[1]))

from foreman.composition import REQUIREMENTS_SCHEMA_VERSION, normalize_requirement, parse_requirements, selection_constraints
from foreman.errors import PlanError, UsageError
from foreman.planner import plan
from tests.test_planner import snapshot


def requirement(independent=False, engagement="onboarding-flow"):
    return {"specialty": "ux", "required_capabilities": ["browser", "ux"],
            "independent": independent, "engagement": engagement}


def worker(name, capabilities=("browser", "ux")):
    return SimpleNamespace(name=name, capabilities=capabilities)


def assignment(name="prior", role="advisor", status="applied", **extra):
    return {"task": "onboarding", "role": role, "agent": name, "status": status,
            "requirements": requirement(), **extra}


class RequirementTest(unittest.TestCase):
    def test_legacy_roles_need_no_requirements(self):
        self.assertEqual(parse_requirements(None, ["developer", "reviewer", "tester"], None), {})
        self.assertEqual(parse_requirements(None, ["architect"], None, allow_historical_architect=True), {})

    def test_normalized_shape_preserves_responsibility_and_engagement(self):
        data = {"schema_version": 1, "assignments": {"advisor": requirement()}}
        data["assignments"]["advisor"]["required_capabilities"].reverse()
        saved = copy.deepcopy(data)
        result = parse_requirements(data, ["advisor", "developer"], "onboarding")
        self.assertEqual(result, {"advisor": requirement()})
        self.assertEqual(data, saved)

    def test_consultations_require_explicit_task_and_requirements(self):
        for role in ("advisor", "investigator", "architect"):
            with self.subTest(role=role), self.assertRaisesRegex(UsageError, "require explicit"):
                parse_requirements(None, [role], "onboarding")
        with self.assertRaisesRegex(UsageError, "--task"):
            parse_requirements({"schema_version": 1, "assignments": {"advisor": requirement()}}, ["advisor"], None)

    def test_malformed_or_unknown_envelopes_refuse(self):
        cases = [[], {}, {"schema_version": True, "assignments": {}},
                 {"schema_version": 2, "assignments": {"advisor": requirement()}},
                 {"schema_version": 1, "assignments": []},
                 {"schema_version": 1, "assignments": {}},
                 {"schema_version": 1, "assignments": {"typo": requirement()}},
                 {"schema_version": 1, "assignments": {"advisor": requirement()}, "task": "other"}]
        for data in cases:
            with self.subTest(data=data), self.assertRaises(UsageError):
                parse_requirements(data, ["advisor"], "onboarding")

    def test_invalid_requirements_never_silently_weaken_the_gate(self):
        cases = [None, {}, {**requirement(), "unknown": True},
                 {**requirement(), "specialty": "UX"}, {**requirement(), "specialty": " ux"},
                 {**requirement(), "required_capabilities": []},
                 {**requirement(), "required_capabilities": ["ux", "ux"]},
                 {**requirement(), "required_capabilities": [[]]},
                 {**requirement(), "independent": "false"},
                 {**requirement(), "engagement": ""}, {**requirement(), "engagement": " new"},
                 {**requirement(), "engagement": "new\nflow"}]
        for data in cases:
            with self.subTest(data=data), self.assertRaises(UsageError):
                normalize_requirement(data, "advisor")
        with self.assertRaisesRegex(UsageError, "pinned judge"):
            normalize_requirement(requirement(), "judge")
        for role in ("reviewer", "tester"):
            with self.subTest(role=role), self.assertRaisesRegex(UsageError, "independent:true"):
                normalize_requirement(requirement(), role)


class EligibilityTest(unittest.TestCase):
    def test_empty_specialist_bench_reports_the_actual_eligibility_gap(self):
        requests = {"advisor": requirement()}
        bench = [worker("spare", ()), worker("other", ())]
        constraints = selection_constraints(["advisor"], bench, requests, [], "onboarding")
        with self.assertRaises(PlanError) as caught:
            plan(["advisor"], snapshot(spare=99, other=80), exclude=constraints["exclude"], requirements=requests,
                 selection_rationale=constraints["rationale"])
        self.assertIn("missing declared capabilities browser, ux", str(caught.exception))
        self.assertIn("preserve independence", str(caught.exception))
        self.assertEqual(caught.exception.details["eligibility"], constraints["rationale"])

    def test_lower_headroom_capable_worker_beats_unqualified_worker(self):
        agents = [worker("qualified"), worker("spare", ("ux",))]
        requests = {"advisor": requirement()}
        constraints = selection_constraints(["advisor"], agents, requests, [], "onboarding")
        result = plan(["advisor"], snapshot(qualified=30, spare=99), exclude=constraints["exclude"],
                      requirements=requests, familiarity=constraints["familiarity"])
        self.assertEqual(result["assignments"], {"advisor": "qualified"})
        self.assertIn("missing declared capabilities browser", " ".join(constraints["rationale"]))

    def test_unconfigured_snapshot_worker_cannot_claim_specialty(self):
        constraints = selection_constraints(["advisor"], [worker("configured")], {"advisor": requirement()},
                                            [], "onboarding", candidate_names=["configured", "ghost"])
        self.assertEqual(constraints["exclude"]["advisor"], ["ghost"])
        self.assertIn("absent from current config", " ".join(constraints["rationale"]))

    def test_old_config_capabilities_do_not_come_from_model_name(self):
        constraints = selection_constraints(["advisor"], [worker("best-ux-model", ())],
                                            {"advisor": requirement()}, [], "onboarding")
        self.assertEqual(constraints["exclude"]["advisor"], ["best-ux-model"])

    def test_independence_follows_possible_contributions_across_context_and_model_changes(self):
        for role in ("developer", "architect", "advisor", "investigator"):
            for status in ("applied", "unknown", "sending", "sent_but_not_started"):
                with self.subTest(role=role, status=status):
                    history = [assignment(role=role, status=status),
                               assignment(role="release", requirements=None, context_session={"value": "different"})]
                    constraints = selection_constraints(["reviewer", "tester"], [worker("prior"), worker("fresh")],
                                                        {}, history, "onboarding")
                    self.assertEqual(constraints["exclude"], {"reviewer": ["prior"], "tester": ["prior"]})

    def test_non_independent_consultation_can_continue_author_work(self):
        constraints = selection_constraints(["advisor"], [worker("prior")], {"advisor": requirement()},
                                            [assignment(role="developer")], "onboarding")
        self.assertEqual(constraints["exclude"]["advisor"], [])

    def test_pending_send_is_excluded_even_before_assignment_row_exists(self):
        constraints = selection_constraints(["reviewer"], [worker("prior")], {}, [], "onboarding",
                                            dispatches=[assignment(role="developer", status="sending")])
        self.assertEqual(constraints["exclude"]["reviewer"], ["prior"])
        for status in ("reserved", "not_sent"):
            constraints = selection_constraints(["reviewer"], [worker("prior")], {}, [], "onboarding",
                                                dispatches=[assignment(role="developer", status=status)])
            self.assertEqual(constraints["exclude"]["reviewer"], [])

    def test_other_tasks_and_unbound_legacy_runs_keep_candidates(self):
        for task in (None, "another-task"):
            constraints = selection_constraints(["reviewer"], [worker("prior")], {}, [assignment()], task)
            self.assertEqual(constraints["exclude"]["reviewer"], [])

    def test_architecture_under_reviewer_role_is_still_a_contribution(self):
        for tier in (None, {"round": "architect"}, {"round": "reconciliation"}):
            constraints = selection_constraints(["tester"], [worker("prior")], {},
                                                [assignment(role="reviewer", requirements=None, tier=tier)], "onboarding")
            self.assertEqual(constraints["exclude"]["tester"], ["prior"])
        constraints = selection_constraints(["tester"], [worker("prior")], {},
                                            [assignment(role="reviewer", requirements=None, tier={"round": "review"},
                                                        reviewer_scope="verification")], "onboarding")
        self.assertEqual(constraints["exclude"]["tester"], [])

    def test_new_untiered_verification_remains_eligible_for_followup(self):
        history = [assignment(role="reviewer", requirements=None, tier=None, reviewer_scope="verification")]
        constraints = selection_constraints(["reviewer", "tester"], [worker("prior")], {}, history, "onboarding")
        self.assertEqual(constraints["exclude"], {"reviewer": [], "tester": []})

    def test_predevelopment_tester_stays_excluded_whatever_its_assessment_says(self):
        # Add-only (#625): a recorded `none` never lifts the exclusion.
        history = [assignment(role="tester", tier={"round": "test_plan"})]
        constraints = selection_constraints(["reviewer", "tester"], [worker("prior")], {}, history, "onboarding")
        self.assertEqual(constraints["exclude"], {"reviewer": ["prior"], "tester": ["prior"]})
        assessment = {"assignment_index": 0, "task": "onboarding", "agent": "prior", "contribution": "none"}
        constraints = selection_constraints(["reviewer", "tester"], [worker("prior")], {}, history, "onboarding", assessments=[assessment])
        self.assertEqual(constraints["exclude"], {"reviewer": ["prior"], "tester": ["prior"]})

    def test_migration_or_requirements_never_invent_verification_provenance(self):
        for scope in (None, "unknown", "design"):
            for tier in (None, {"round": "review"}):
                with self.subTest(scope=scope, tier=tier):
                    history = [assignment(role="reviewer", schema_version=6, reviewer_scope=scope,
                                          requirements=requirement(independent=True), tier=tier)]
                    constraints = selection_constraints(["reviewer"], [worker("prior")], {}, history, "onboarding")
                    self.assertEqual(constraints["exclude"]["reviewer"], ["prior"])

    def test_authored_tier_overrides_a_verification_label_and_nothing_clears_it(self):
        for round_type in ("architect", "reconciliation"):
            with self.subTest(round_type=round_type):
                history = [assignment(role="reviewer", reviewer_scope="verification", tier={"round": round_type})]
                constraints = selection_constraints(["reviewer"], [worker("prior")], {}, history, "onboarding")
                self.assertEqual(constraints["exclude"]["reviewer"], ["prior"])
                assessment = {"assignment_index": 0, "task": "onboarding", "agent": "prior", "contribution": "none"}
                constraints = selection_constraints(["reviewer"], [worker("prior")], {}, history, "onboarding", assessments=[assessment])
                self.assertEqual(constraints["exclude"]["reviewer"], ["prior"])

    def test_a_none_assessment_never_subtracts_a_contributor(self):
        # Add-only (#625): a consultation worker stays excluded from verifying
        # its own task whatever its assessment, worker-declared or legacy, says.
        assessment = {"assignment_index": 0, "task": "onboarding", "agent": "prior", "contribution": "none"}
        for history in ([assignment()], [assignment(role="developer")], [assignment(), assignment(role="architect")]):
            with self.subTest(history=[row["role"] for row in history]):
                constraints = selection_constraints(["reviewer"], [worker("prior")], {}, history, "onboarding",
                                                    dispatches=[assignment(assignment_index=0)], assessments=[assessment])
                self.assertEqual(constraints["exclude"]["reviewer"], ["prior"])

    def test_a_declared_contribution_adds_an_exclusion_with_no_dispatch_history(self):
        assessment = {"assignment_index": 0, "task": "onboarding", "agent": "verifier", "contribution": "design"}
        history = [assignment(name="verifier", role="reviewer", reviewer_scope="verification")]
        constraints = selection_constraints(["reviewer"], [worker("verifier")], {}, history, "onboarding")
        self.assertEqual(constraints["exclude"]["reviewer"], [])
        constraints = selection_constraints(["reviewer"], [worker("verifier")], {}, history, "onboarding", assessments=[assessment])
        self.assertEqual(constraints["exclude"]["reviewer"], ["verifier"])

    def test_any_assessed_design_or_implementation_remains_a_contribution(self):
        for contribution in ("design", "implementation"):
            assessments = [
                {"assignment_index": 0, "task": "onboarding", "agent": "prior", "contribution": contribution},
                {"assignment_index": 1, "task": "onboarding", "agent": "prior", "contribution": "none"},
            ]
            constraints = selection_constraints(["reviewer"], [worker("prior")], {}, [], "onboarding", assessments=assessments)
            self.assertEqual(constraints["exclude"]["reviewer"], ["prior"])

    def test_familiarity_requires_confirmed_same_task_role_and_requirements(self):
        requests = {"advisor": requirement()}
        variants = [assignment(), assignment(name="unsent", status="sent_but_not_started"),
                    assignment(name="other-task", task="other"), assignment(name="other-role", role="architect"),
                    assignment(name="other-engagement", requirements=requirement(engagement="other")),
                    assignment(name="old", requirements=None)]
        agents = [worker(row["agent"]) for row in variants]
        constraints = selection_constraints(["advisor"], agents, requests, variants, "onboarding")
        self.assertEqual(constraints["familiarity"]["advisor"], {
            "prior": 1, "unsent": 0, "other-task": 0, "other-role": 0, "other-engagement": 0, "old": 0,
        })
        self.assertIn("never expertise or completion", " ".join(constraints["rationale"]))

    def test_independent_specialist_rejects_a_familiar_author(self):
        request = requirement(independent=True)
        constraints = selection_constraints(["advisor"], [worker("prior")], {"advisor": request},
                                            [assignment(requirements=request)], "onboarding")
        self.assertEqual(constraints["exclude"]["advisor"], ["prior"])
        self.assertEqual(constraints["familiarity"]["advisor"], {})



class SeatResponsibilityTest(unittest.TestCase):
    """A seat is its role for every responsibility check (#434)."""

    def agent(self, name):
        return SimpleNamespace(name=name, capabilities=("review",))

    def history(self):
        return [{"task": "t1", "role": "developer", "agent": "alpha", "status": "applied",
                 "reviewer_scope": None}]

    def test_a_contributor_is_barred_from_every_seat_of_the_role(self):
        for role in ("reviewer", "reviewer#api"):
            with self.subTest(role=role):
                result = selection_constraints(
                    [role], [self.agent("alpha"), self.agent("beta")], {}, self.history(), "t1")
                self.assertEqual(result["exclude"][role], ["alpha"])

    def test_a_pending_design_seat_contributes_like_its_role(self):
        # A dispatch keeps the SEAT while the ledger keeps the responsibility,
        # so the contributor classifier reads `reviewer#api` on the pending row.
        # Reading it literally leaves a design reviewer eligible for an
        # independent seat on its own task before the send even resolves (#434).
        for role in ("reviewer", "reviewer#api"):
            with self.subTest(role=role):
                dispatch = {"task": "t1", "role": role, "agent": "alpha", "status": "sending",
                            "reviewer_scope": "design"}
                result = selection_constraints(
                    ["reviewer#core"], [self.agent("alpha"), self.agent("beta")], {}, [], "t1",
                    dispatches=[dispatch])
                self.assertEqual(result["exclude"]["reviewer#core"], ["alpha"])

    def test_a_seat_requirement_still_requires_independence(self):
        record = {"specialty": "api-review", "required_capabilities": ["review"],
                  "independent": False, "engagement": "review the api slice"}
        for role in ("reviewer", "reviewer#api"):
            with self.subTest(role=role):
                with self.assertRaisesRegex(UsageError, "require independent:true"):
                    normalize_requirement(record, role)

    def test_a_seat_inherits_its_role_requirement(self):
        # One `reviewer` entry covers every slice (#434).
        record = {"specialty": "api-review", "required_capabilities": ["review"],
                  "independent": True, "engagement": "review the slice"}
        payload = {"schema_version": REQUIREMENTS_SCHEMA_VERSION,
                   "assignments": {"reviewer": record}}
        parsed = parse_requirements(payload, ["reviewer#api", "reviewer#core"], "t1")
        self.assertEqual(parsed["reviewer#api"]["specialty"], "api-review")
        self.assertEqual(parsed["reviewer#core"]["specialty"], "api-review")

    def test_a_seat_requirement_beside_its_role_is_refused(self):
        # The seat entry would decide that seat's capabilities instead of its
        # role's, so a seat could require less than the responsibility does and
        # admit a worker the role's capabilities bar (#434).
        role_record = {"specialty": "api-review", "required_capabilities": ["review", "security"],
                       "independent": True, "engagement": "review the slice"}
        weaker = {"specialty": "api-review", "required_capabilities": ["review"],
                  "independent": True, "engagement": "review the slice"}
        payload = {"schema_version": REQUIREMENTS_SCHEMA_VERSION,
                   "assignments": {"reviewer": role_record, "reviewer#api": weaker}}
        with self.assertRaisesRegex(UsageError, "beside its responsibility"):
            parse_requirements(payload, ["reviewer#api", "reviewer#core"], "t1")

    def test_an_explicitly_null_requirement_is_refused(self):
        # `assignments.get(...)` returns None for an absent key and for an
        # explicit null alike, so a sentinel keeps `{"advisor": null}` from
        # reading as "no requirement" and bypassing the specialist contract.
        payload = {"schema_version": REQUIREMENTS_SCHEMA_VERSION,
                   "assignments": {"advisor": None}}
        with self.assertRaisesRegex(UsageError, "Each specialist requirement needs"):
            parse_requirements(payload, ["advisor"], "t1")

    def test_a_requirement_for_an_unassigned_role_is_still_refused(self):
        record = {"specialty": "api-review", "required_capabilities": ["review"],
                  "independent": True, "engagement": "review the slice"}
        payload = {"schema_version": REQUIREMENTS_SCHEMA_VERSION,
                   "assignments": {"tester": record}}
        with self.assertRaisesRegex(UsageError, "only roles this plan assigns"):
            parse_requirements(payload, ["reviewer#api"], "t1")

    def test_a_seat_requirement_resolves_its_role(self):
        record = {"specialty": "api-review", "required_capabilities": ["review"],
                  "independent": True, "engagement": "review the api slice"}
        self.assertEqual(normalize_requirement(record, "reviewer#api")["specialty"], "api-review")
        with self.assertRaisesRegex(UsageError, "invent a responsibility"):
            normalize_requirement(record, "nonsense#api")


if __name__ == "__main__":
    unittest.main()

skills

herdr-foreman

tests

__init__.py

fakes.py

test_assign.py

test_attention.py

test_billing.py

test_bounded_run.sh

test_capabilities.py

test_capability_routing.py

test_chronology.py

test_churn.py

test_classify.sh

test_claude_native.py

test_cli.py

test_compose_briefs.sh

test_composer.py

test_composition.py

test_config.py

test_continuity_cli.py

test_cost_report.py

test_diagnostics.py

test_engagement.py

test_entrypoints.py

test_foreman_launcher.sh

test_foreman_queue.py

test_foreman_reset.py

test_foreman_seat.py

test_foreman_tier_check.py

test_freeze.py

test_herdr.py

test_historical.py

test_home.py

test_label_workspaces.sh

test_launch.py

test_legacy_recovery.py

test_lifecycle.py

test_load_set.py

test_measure.py

test_members.py

test_memory.py

test_minimum_adequate.py

test_oracle.py

test_parsers.py

test_partition.py

test_planner.py

test_probe.py

test_provision_worktree.sh

test_prune_remote_branches.sh

test_prune_report_caches.py

test_prune_result.py

test_prune_worktrees.sh

test_recovery_cli.py

test_recovery.py

test_renderable.py

test_report_contract.py

test_report_delivery.py

test_report_gates.py

test_report_verdict.py

test_reset_input_hook.py

test_resolve_gates.sh

test_resolve_policy_paths.py

test_restoration.py

test_retrospective_runtime.py

test_retrospective.py

test_review_package.py

test_role_clear.py

test_roster.sh

test_round_preflight.sh

test_runnable.py

test_scoring.py

test_script_dir_newline.sh

test_seat_holds.py

test_selection.py

test_skill_invocations.sh

test_slice_scope_parity.py

test_specialist_cli.py

test_specialist_delivery.py

test_specialist_recovery.py

test_specialist_retention.py

test_stale_grok_delivery.py

test_start_judge_worker.py

test_state.py

test_successors.py

test_supervision_cli.py

test_supervision_diagnostics.py

test_supervision_gate.py

test_supervision_replay.py

test_supervision.py

test_sweep_worktrees.sh

test_tier_integration.py

test_tiers.py

test_triggers.py

test_typesafe_client.py

test_verdict_gates.py

test_verify_authority.sh

test_wait_report.sh

tier_fixture.py

bounded-run.sh

compose-briefs.sh

config.example.json

foreman-tier-check.py

foreman.sh

label-workspaces.sh

provision-worktree.sh

prune-remote-branches.sh

prune-report-caches.py

prune-worktrees.sh

resolve-gates.sh

resolve-policy-paths.sh

review-package.sh

roster.sh

round-preflight.sh

SKILL.md

start-judge-worker.sh

state-schema.md

sweep-worktrees.sh

verify-authority.sh

wait-report.sh

README.md

tile.json