CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/coding-policy

General-purpose coding policy for Baruch's AI agents

73

Quality

91%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide

SecuritybySnyk

Low

Low-risk findings worth noting

Overview
Quality
Evals
Security
Files

test_report_contract.pyskills/herdr-foreman/tests/

"""Report contract lines: every form the parser accepts and every refusal class it names."""

import sys
import unittest
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parents[1]))

from foreman import report_contract
from foreman.errors import UsageError

CRITERIA_BRIEF = """# Brief

## Assignment

- Question: settle it

## Acceptance Criteria

CRITERION 1: the flow is named
CRITERION 2: the alternatives are compared

## Evidence and Capabilities

{inputs}
"""


def brief(inputs="Read the issue."):
    return CRITERIA_BRIEF.format(inputs=inputs)


def consultation(*lines):
    return "Findings.\n" + "\n".join(lines) + "\n"


MET = ("ACCEPTANCE 1/2: met — the flow is in section 1", "ACCEPTANCE 2/2: met — table in section 2")


class BriefCriteriaTest(unittest.TestCase):
    def test_contiguous_block_counts(self):
        self.assertEqual(report_contract.brief_criteria(brief()), 2)

    def test_criterion_injected_into_another_section_is_ignored(self):
        self.assertEqual(report_contract.brief_criteria(brief("CRITERION 3: sneaked in through inputs")), 2)

    def test_gapped_duplicate_malformed_missing_and_doubled_sections_are_refused(self):
        cases = {
            "gap": brief().replace("CRITERION 2:", "CRITERION 3:"),
            "duplicate": brief().replace("CRITERION 2: the alternatives are compared", "CRITERION 1: again"),
            "malformed": brief().replace("CRITERION 2: the", "CRITERION two: the"),
            "empty": brief().replace("CRITERION 1: the flow is named\nCRITERION 2: the alternatives are compared\n", ""),
            "no section": "# Brief\n\nCRITERION 1: outside any section\n",
            "two sections": brief("## Acceptance Criteria\nCRITERION 3: second section"),
        }
        for name, text in cases.items():
            with self.subTest(name), self.assertRaises(UsageError) as caught:
                report_contract.brief_criteria(text)
            self.assertTrue(caught.exception.details["gaps"])

    def test_markup_on_criterion_lines_is_tolerated(self):
        text = brief().replace("CRITERION 1:", "- CRITERION 1:").replace("CRITERION 2:", "> `CRITERION 2:")
        self.assertEqual(report_contract.brief_criteria(text), 2)


class ReportLinesTest(unittest.TestCase):
    def gaps(self, text, role, specialty=None, criteria=None):
        with self.assertRaises(UsageError) as caught:
            report_contract.report_lines(text, role, specialty, criteria)
        return " | ".join(caught.exception.details["gaps"])

    def test_reviewer_and_tester_verdicts(self):
        for role in ("reviewer", "tester"):
            for verdict in ("blocking", "approved"):
                with self.subTest(role=role, verdict=verdict):
                    lines = report_contract.report_lines("Body\nVERDICT: {}\n".format(verdict), role)
                    self.assertEqual(lines, {"verdict": verdict, "acceptance": None, "contribution": None})

    def test_tolerated_markup(self):
        for line in ("- VERDICT: approved", "* VERDICT: approved", "> VERDICT: approved", "`VERDICT: approved`",
                     "  - `VERDICT: approved`  "):
            with self.subTest(line=line):
                self.assertEqual(report_contract.report_lines(line + "\n", "reviewer")["verdict"], "approved")

    def test_consultation_acceptance_with_each_dash(self):
        for dash in ("—", "–", "-"):
            with self.subTest(dash=dash):
                text = consultation("ACCEPTANCE 1/2: met {} a".format(dash), "ACCEPTANCE 2/2: unmet {} b".format(dash))
                lines = report_contract.report_lines(text, "advisor", "ux", 2)
                self.assertEqual(lines["acceptance"], [{"k": 1, "state": "met", "evidence": "a"},
                                                       {"k": 2, "state": "unmet", "evidence": "b"}])
                self.assertIsNone(lines["verdict"])

    def test_verdict_specialties_carry_both(self):
        for specialty in sorted(report_contract.VERDICT_SPECIALTIES):
            with self.subTest(specialty=specialty):
                lines = report_contract.report_lines(consultation(*MET, "VERDICT: blocking"), "advisor", specialty, 2)
                self.assertEqual(lines["verdict"], "blocking")
                self.assertIn("missing VERDICT", self.gaps(consultation(*MET), "advisor", specialty, 2))

    def test_optional_contribution(self):
        lines = report_contract.report_lines(consultation(*MET, "CONTRIBUTION: design"), "architect", None, 2)
        self.assertEqual(lines["contribution"], "design")
        self.assertIn("duplicate CONTRIBUTION",
                      self.gaps(consultation(*MET, "CONTRIBUTION: none", "CONTRIBUTION: none"), "architect", None, 2))
        self.assertIn("malformed CONTRIBUTION", self.gaps(consultation(*MET, "CONTRIBUTION: some"), "architect", None, 2))

    def test_missing(self):
        self.assertIn("missing VERDICT", self.gaps("No verdict here.\n", "reviewer"))
        self.assertIn("missing ACCEPTANCE 2", self.gaps(consultation(MET[0]), "investigator", None, 2))

    def test_duplicate_including_identical_repeat(self):
        self.assertIn("duplicate VERDICT", self.gaps("VERDICT: approved\nVERDICT: approved\n", "reviewer"))
        self.assertIn("duplicate VERDICT", self.gaps("VERDICT: approved\nVERDICT: blocking\n", "tester"))
        self.assertIn("duplicate ACCEPTANCE 1", self.gaps(consultation(*MET, MET[0]), "advisor", None, 2))

    def test_extra(self):
        self.assertIn("extra VERDICT", self.gaps(consultation(*MET, "VERDICT: approved"), "architect", None, 2))
        self.assertIn("extra VERDICT", self.gaps(consultation(*MET, "VERDICT: approved"), "advisor", "accessibility", 2))
        self.assertIn("extra ACCEPTANCE", self.gaps("VERDICT: approved\nACCEPTANCE 1/1: met — x\n", "reviewer"))
        self.assertIn("extra ACCEPTANCE 3", self.gaps(consultation(*MET, "ACCEPTANCE 3/2: met — x"), "advisor", None, 2))

    def test_a_criterion_line_in_a_report_is_extra(self):
        self.assertIn("extra CRITERION", self.gaps("VERDICT: approved\nCRITERION 1: restated from the brief\n", "reviewer"))
        self.assertIn("extra CRITERION", self.gaps(consultation(*MET, "- CRITERION 1: the flow is named"), "advisor", None, 2))

    def test_n_mismatch(self):
        text = consultation("ACCEPTANCE 1/1: met — only one")
        self.assertIn("recorded 2", self.gaps(text, "advisor", None, 2))

    def test_malformed_candidates(self):
        for line in ("VERDICT: looks good", "VERDICT approved", "ACCEPTANCE 1/2: met", "ACCEPTANCE 1/2: met —   ",
                     "ACCEPTANCE 1/2: done — x", "ACCEPTANCE one/2: met — x"):
            with self.subTest(line=line):
                self.assertIn("malformed", self.gaps(consultation(*MET, line), "advisor", "security", 2))

    def test_prose_mentioning_keywords_in_lower_case_is_not_a_candidate(self):
        text = consultation(*MET, "The acceptance criteria and the verdict are discussed above.")
        self.assertEqual(report_contract.report_lines(text, "advisor", None, 2)["verdict"], None)

    def test_declared_contributions_survive_other_gaps(self):
        text = "No verdict.\n- CONTRIBUTION: implementation\nCONTRIBUTION: design\nCONTRIBUTION: lots\n"
        self.assertEqual(report_contract.declared_contributions(text), {"implementation", "design"})
        self.assertEqual(report_contract.declared_contributions("contribution: design\n"), set())

    def test_every_gap_is_named_at_once(self):
        gaps = self.gaps("VERDICT: approved\nVERDICT: approved\nACCEPTANCE 1/1: met — x\n", "reviewer")
        self.assertIn("duplicate VERDICT", gaps)
        self.assertIn("extra ACCEPTANCE", gaps)

    def test_role_and_criteria_preconditions(self):
        for role, criteria in (("developer", None), ("advisor", None), ("advisor", 0), ("reviewer", 2)):
            with self.subTest(role=role, criteria=criteria), self.assertRaises(UsageError):
                report_contract.report_lines("VERDICT: approved\n", role, None, criteria)


if __name__ == "__main__":
    unittest.main()

skills

herdr-foreman

tests

__init__.py

fakes.py

test_assign.py

test_attention.py

test_billing.py

test_bounded_run.sh

test_capabilities.py

test_capability_routing.py

test_chronology.py

test_churn.py

test_classify.sh

test_claude_native.py

test_cli.py

test_compose_briefs.sh

test_composer.py

test_composition.py

test_config.py

test_continuity_cli.py

test_cost_report.py

test_diagnostics.py

test_engagement.py

test_entrypoints.py

test_foreman_launcher.sh

test_foreman_queue.py

test_foreman_reset.py

test_foreman_seat.py

test_foreman_tier_check.py

test_freeze.py

test_herdr.py

test_historical.py

test_home.py

test_label_workspaces.sh

test_launch.py

test_legacy_recovery.py

test_load_set.py

test_measure.py

test_members.py

test_memory.py

test_oracle.py

test_parsers.py

test_partition.py

test_planner.py

test_probe.py

test_provision_worktree.sh

test_prune_remote_branches.sh

test_prune_report_caches.py

test_prune_result.py

test_prune_worktrees.sh

test_recovery_cli.py

test_recovery.py

test_renderable.py

test_report_contract.py

test_report_delivery.py

test_report_gates.py

test_report_verdict.py

test_resolve_gates.sh

test_resolve_policy_paths.py

test_restoration.py

test_retrospective_runtime.py

test_retrospective.py

test_review_package.py

test_role_clear.py

test_roster.sh

test_round_preflight.sh

test_runnable.py

test_scoring.py

test_script_dir_newline.sh

test_seat_holds.py

test_selection.py

test_skill_invocations.sh

test_slice_scope_parity.py

test_specialist_cli.py

test_specialist_delivery.py

test_specialist_recovery.py

test_specialist_retention.py

test_stale_grok_delivery.py

test_start_judge_worker.py

test_state.py

test_supervision_cli.py

test_supervision_diagnostics.py

test_supervision_gate.py

test_supervision_replay.py

test_supervision.py

test_sweep_worktrees.sh

test_tier_integration.py

test_tiers.py

test_triggers.py

test_typesafe_client.py

test_verdict_gates.py

test_verify_authority.sh

test_wait_report.sh

tier_fixture.py

bounded-run.sh

compose-briefs.sh

config.example.json

foreman-tier-check.py

foreman.sh

label-workspaces.sh

provision-worktree.sh

prune-remote-branches.sh

prune-report-caches.py

prune-worktrees.sh

resolve-gates.sh

resolve-policy-paths.sh

review-package.sh

roster.sh

round-preflight.sh

SKILL.md

start-judge-worker.sh

state-schema.md

sweep-worktrees.sh

verify-authority.sh

wait-report.sh

README.md

tile.json