CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/coding-policy

General-purpose coding policy for Baruch's AI agents

73

Quality

91%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide

SecuritybySnyk

Low

Low-risk findings worth noting

Overview
Quality
Evals
Security
Files

test_foreman_tier_check.pyskills/herdr-foreman/tests/

"""foreman-tier-check.py owns the composite verdict and the evidence each row needs.

Every test drives `main()` with the owner CLI replaced by recorded outputs, so
nothing spawns a process or contacts Herdr (#626).
"""

import contextlib
import copy
import importlib.util
import io
import json
import unittest
from pathlib import Path
from unittest.mock import patch

SCRIPT = Path(__file__).resolve().parents[1] / "foreman-tier-check.py"
_SPEC = importlib.util.spec_from_file_location("foreman_tier_check", SCRIPT)
assert _SPEC is not None and _SPEC.loader is not None
check = importlib.util.module_from_spec(_SPEC)
_SPEC.loader.exec_module(check)

MEASURED = {"schema_version": 3, "agents": {}, "failed_agents": []}
#: What `foreman verify-foreman` emits for a proven seat.
PROVEN = {
    "configured": True, "agent": "foreman", "pane": "w1:p0",
    "tier": {"model": "sonnet-5", "effort": "medium", "tier_row": "coordination"},
    "argv_verified": True,
    "verified": {"source": "process_argv", "argv": ["claude", "--model", "sonnet-5", "--effort", "medium"],
                 "model": "sonnet-5", "effort": "medium", "pid": 400, "pane_id": "w1:p0"},
}
UNCONFIGURED = {"configured": False, "warning": "add a `foreman` block to config.json, then run start-foreman"}


def proven(**changes):
    result = copy.deepcopy(PROVEN)
    for key, value in changes.items():
        if key == "verified":
            result["verified"].update(value)
        else:
            result[key] = value
    return result


class CompositeTest(unittest.TestCase):
    def verdict(self, verify_output, argv=(), measure=(0, MEASURED)):
        replies = {"measure": (measure[0], json.dumps(measure[1])), "verify-foreman": (0, json.dumps(verify_output))}

        def fake_run(args):
            return replies["measure" if "measure" in args else "verify-foreman"]

        out = io.StringIO()
        with patch.object(check, "run", fake_run), contextlib.redirect_stdout(out):
            code = check.main(list(argv))
        return code, json.loads(out.getvalue())

    def test_a_complete_process_argv_proof_is_ready(self):
        code, rows = self.verdict(PROVEN)
        self.assertEqual(code, 0)
        self.assertEqual(rows["foreman_tier"]["status"], "ok")
        self.assertEqual(rows["headroom"]["status"], "ok")

    def test_an_ok_result_without_its_process_argv_proof_fails(self):
        without = {key: value for key, value in PROVEN.items() if key != "verified"}
        for result in (without,
                       {"configured": True, "argv_verified": True},
                       proven(argv_verified=False),
                       proven(verified={"source": "launch_argv"}),
                       proven(verified={"argv": []})):
            with self.subTest(result=result):
                code, rows = self.verdict(result)
                self.assertEqual(code, 1)
                self.assertEqual(rows["foreman_tier"]["status"], "failed")
                self.assertIn("unproven", rows["foreman_tier"]["reason"])

    def test_a_proof_of_another_tier_fails(self):
        for changes in ({"model": "opus-5"}, {"effort": "high"}):
            with self.subTest(proven=changes):
                code, rows = self.verdict(proven(verified=changes))
                self.assertEqual(code, 1)
                self.assertEqual(rows["foreman_tier"]["status"], "failed")

    def test_unconfigured_with_its_warning_is_ready(self):
        code, rows = self.verdict(UNCONFIGURED)
        self.assertEqual(code, 0)
        self.assertEqual(rows["foreman_tier"]["status"], "unconfigured")
        self.assertIn("warning", rows["foreman_tier"]["detail"])

    def test_unconfigured_without_a_warning_fails(self):
        for result in ({"configured": False}, {"configured": False, "warning": "  "}):
            with self.subTest(result=result):
                code, rows = self.verdict(result)
                self.assertEqual(code, 1)
                self.assertEqual(rows["foreman_tier"]["status"], "failed")

    def test_skipped_headroom_with_an_unconfigured_seat_is_ready(self):
        code, rows = self.verdict(UNCONFIGURED, argv=["--no-measure"])
        self.assertEqual(code, 0)
        self.assertEqual(rows["headroom"]["status"], "skipped")

    def test_a_failed_measurement_is_not_ready(self):
        code, rows = self.verdict(PROVEN, measure=(3, {}))
        self.assertEqual(code, 1)
        self.assertEqual(rows["headroom"]["status"], "failed")


if __name__ == "__main__":
    unittest.main()

skills

herdr-foreman

tests

__init__.py

fakes.py

test_assign.py

test_attention.py

test_billing.py

test_bounded_run.sh

test_capabilities.py

test_capability_routing.py

test_chronology.py

test_churn.py

test_classify.sh

test_claude_native.py

test_cli.py

test_compose_briefs.sh

test_composer.py

test_composition.py

test_config.py

test_continuity_cli.py

test_cost_report.py

test_diagnostics.py

test_engagement.py

test_entrypoints.py

test_foreman_launcher.sh

test_foreman_queue.py

test_foreman_reset.py

test_foreman_seat.py

test_foreman_tier_check.py

test_freeze.py

test_herdr.py

test_historical.py

test_home.py

test_label_workspaces.sh

test_launch.py

test_legacy_recovery.py

test_load_set.py

test_measure.py

test_members.py

test_memory.py

test_oracle.py

test_parsers.py

test_partition.py

test_planner.py

test_probe.py

test_provision_worktree.sh

test_prune_remote_branches.sh

test_prune_report_caches.py

test_prune_result.py

test_prune_worktrees.sh

test_recovery_cli.py

test_recovery.py

test_renderable.py

test_report_contract.py

test_report_delivery.py

test_report_gates.py

test_report_verdict.py

test_resolve_gates.sh

test_resolve_policy_paths.py

test_restoration.py

test_retrospective_runtime.py

test_retrospective.py

test_review_package.py

test_role_clear.py

test_roster.sh

test_round_preflight.sh

test_runnable.py

test_scoring.py

test_script_dir_newline.sh

test_seat_holds.py

test_selection.py

test_skill_invocations.sh

test_slice_scope_parity.py

test_specialist_cli.py

test_specialist_delivery.py

test_specialist_recovery.py

test_specialist_retention.py

test_stale_grok_delivery.py

test_start_judge_worker.py

test_state.py

test_supervision_cli.py

test_supervision_diagnostics.py

test_supervision_gate.py

test_supervision_replay.py

test_supervision.py

test_sweep_worktrees.sh

test_tier_integration.py

test_tiers.py

test_triggers.py

test_typesafe_client.py

test_verdict_gates.py

test_verify_authority.sh

test_wait_report.sh

tier_fixture.py

bounded-run.sh

compose-briefs.sh

config.example.json

foreman-tier-check.py

foreman.sh

label-workspaces.sh

provision-worktree.sh

prune-remote-branches.sh

prune-report-caches.py

prune-worktrees.sh

resolve-gates.sh

resolve-policy-paths.sh

review-package.sh

roster.sh

round-preflight.sh

SKILL.md

start-judge-worker.sh

state-schema.md

sweep-worktrees.sh

verify-authority.sh

wait-report.sh

README.md

tile.json