CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/coding-policy

General-purpose coding policy for Baruch's AI agents

74

Quality

93%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide
SecuritybySnyk

Medium

Suggest reviewing before use

Overview
Quality
Evals
Security
Files

test_continuity_cli.pyskills/herdr-foreman/tests/

"""Public CLI continuity flows stay offline and preserve independent owners."""

import io
import json
import sys
import tempfile
import unittest
from pathlib import Path
from unittest.mock import patch

sys.path.insert(0, str(Path(__file__).resolve().parents[1]))

from foreman import attention, cli, memory


NOW = "2025-01-01T12:00:00Z"
LATER = "2025-01-01T12:01:00Z"


class ContinuityCliTests(unittest.TestCase):
    def setUp(self):
        self.temp = tempfile.TemporaryDirectory()
        self.addCleanup(self.temp.cleanup)
        self.root = Path(self.temp.name).resolve()
        self.state = self.root / "state.json"
        self.source = self.root / "TASK-LEDGER.md"
        self.source.write_text("Verified task evidence.\n", encoding="utf-8")

    def invoke(self, *arguments, now=NOW):
        out, err = io.StringIO(), io.StringIO()
        with patch.object(cli, "load_config", side_effect=AssertionError("offline command read config")), \
                patch.object(cli, "_client", side_effect=AssertionError("offline command contacted Herdr")), \
                patch.object(cli, "load_state_checked", side_effect=AssertionError("offline command read dispatch state")):
            code = cli.main([*arguments, "--state", str(self.state), "--now", now], stdout=out, stderr=err)
        return code, json.loads(out.getvalue()) if out.getvalue() else None, err.getvalue()

    def record_input(self, name, value):
        path = self.root / name
        path.write_text(json.dumps(value), encoding="utf-8")
        return str(path)

    def obligation(self):
        return {"id": "review-result", "kind": "review", "task": "task-1",
                "title": "Review the proposed result", "context": "The draft is ready.",
                "consequence": "Publication awaits review.",
                "resolution_condition": "Record the user's review outcome.",
                "sources": [{"schema_version": 1, "kind": "task_ledger", "ref": str(self.source)}]}

    def test_empty_read_creates_no_files_even_with_unreadable_dispatch_content(self):
        self.state.write_text("not valid dispatch JSON", encoding="utf-8")
        before = {str(path): path.read_bytes() for path in self.root.iterdir()}
        for command in ("memory-list", "memory-show", "catch-up", "attention-list"):
            with self.subTest(command=command):
                code, payload, errors = self.invoke(command)
                if command == "memory-show":
                    self.assertEqual(code, 1)
                    self.assertIsNone(payload)
                    self.assertIn("No saved memory matches", errors)
                else:
                    self.assertEqual(code, 0, errors)
                    self.assertIsNotNone(payload)
        after = {str(path): path.read_bytes() for path in self.root.iterdir()}
        self.assertEqual(before, after)

    def test_lesson_and_stow_survive_new_cli_invocations_and_lost_source(self):
        lesson = {"id": "lesson-1", "lesson_id": "report-source", "supersedes": None,
                  "status": "active", "lesson": "Check the report before relying on its claim.",
                  "scopes": ["task:task-1"], "sources": [str(self.source)],
                  "last_verified_at": NOW, "expires_at": None,
                  "revalidate_when": "The report is replaced.", "reason": "Observed during review."}
        code, payload, errors = self.invoke("memory-record", "--record", self.record_input("lesson.json", lesson))
        self.assertEqual(code, 0, errors)
        assert payload is not None, "successful write must return its receipt"
        self.assertFalse(payload["replayed"])
        stow = {"id": "handoff-1", "capture": "Review is pending and evidence is in the task ledger.",
                "unresolved_work": ["Read the task ledger before proceeding."], "gaps": [],
                "required_reads": [str(self.source)]}
        code, _, errors = self.invoke("memory-stow", "--record", self.record_input("stow.json", stow))
        self.assertEqual(code, 0, errors)
        saved = memory.location(self.state).read_bytes()
        self.source.unlink()
        code, shown, errors = self.invoke("memory-show", "--id", "handoff-1", now=LATER)
        self.assertEqual(code, 0, errors)
        self.assertIn("Review is pending", json.dumps(shown))
        assert shown is not None, "successful read must return the saved capture"
        self.assertFalse(shown["record"]["reset_ready"])
        self.assertEqual(saved, memory.location(self.state).read_bytes())
        self.assertFalse(self.state.exists())

    def test_presented_review_stays_in_catch_up_until_review_outcome(self):
        code, _, errors = self.invoke("attention-record", "--record", self.record_input("entry.json", self.obligation()))
        self.assertEqual(code, 0, errors)
        present = {"event_id": "show-review", "id": "review-result", "expected_revision": 1,
                   "action": "present", "reason": "The review request was shown.",
                   "evidence": {"schema_version": 1, "kind": "delivery", "ref": "message:2", "summary": "Presented the draft for review."}}
        code, _, errors = self.invoke("attention-update", "--record", self.record_input("present.json", present))
        self.assertEqual(code, 0, errors)
        saved = attention.storage_path(self.state).read_bytes()
        code, view, errors = self.invoke("catch-up", now=LATER)
        self.assertEqual(code, 0, errors)
        assert view is not None, "successful catch-up must return its view"
        self.assertEqual(view["attention"]["total"], 1)
        self.assertEqual(view["attention"]["items"][0]["status"], "open")
        self.assertEqual(saved, attention.storage_path(self.state).read_bytes())
        answer = {"event_id": "review-answer", "id": "review-result", "expected_revision": 2,
                  "action": "resolve", "reason": "The user reviewed the draft.",
                  "evidence": {"schema_version": 1, "kind": "review_outcome", "ref": "message:3", "summary": "The user approved the draft for publication."}}
        code, _, errors = self.invoke("attention-update", "--record", self.record_input("answer.json", answer), now=LATER)
        self.assertEqual(code, 0, errors)
        code, view, errors = self.invoke("catch-up", "--include-closed", now=LATER)
        self.assertEqual(code, 0, errors)
        assert view is not None, "successful catch-up must return its view"
        self.assertEqual(view["attention"]["total"], 0)
        self.assertEqual(view["closed"]["total"], 1)
        self.assertFalse(self.state.exists())

    def test_future_attention_schema_is_preserved_with_structured_error(self):
        path = attention.storage_path(self.state)
        path.write_text('{"schema_version": 999, "events": []}', encoding="utf-8")
        before = path.read_bytes()
        code, output, errors = self.invoke("catch-up")
        self.assertEqual(code, 1)
        self.assertIsNone(output)
        self.assertIsInstance(json.loads(errors), dict)
        self.assertEqual(path.read_bytes(), before)
        self.assertFalse(Path(str(path) + ".lock").exists())


if __name__ == "__main__":
    unittest.main()

skills

herdr-foreman

tests

__init__.py

fakes.py

test_assign.py

test_attention.py

test_billing.py

test_bounded_run.sh

test_capabilities.py

test_capability_routing.py

test_chronology.py

test_churn.py

test_classify.sh

test_claude_native.py

test_cli.py

test_compose_briefs.sh

test_composer.py

test_composition.py

test_config.py

test_continuity_cli.py

test_cost_report.py

test_diagnostics.py

test_engagement.py

test_entrypoints.py

test_foreman_launcher.sh

test_foreman_queue.py

test_foreman_reset.py

test_foreman_seat.py

test_foreman_tier_check.py

test_freeze.py

test_herdr.py

test_historical.py

test_home.py

test_label_workspaces.sh

test_launch.py

test_legacy_recovery.py

test_lifecycle.py

test_load_set.py

test_measure.py

test_members.py

test_memory.py

test_minimum_adequate.py

test_oracle.py

test_parsers.py

test_partition.py

test_planner.py

test_probe.py

test_provision_worktree.sh

test_prune_remote_branches.sh

test_prune_report_caches.py

test_prune_result.py

test_prune_worktrees.sh

test_recovery_cli.py

test_recovery.py

test_renderable.py

test_report_contract.py

test_report_delivery.py

test_report_gates.py

test_report_verdict.py

test_reset_input_hook.py

test_resolve_gates.sh

test_resolve_policy_paths.py

test_restoration.py

test_retrospective_runtime.py

test_retrospective.py

test_review_package.py

test_role_clear.py

test_roster.sh

test_round_preflight.sh

test_runnable.py

test_scoring.py

test_script_dir_newline.sh

test_seat_holds.py

test_selection.py

test_skill_invocations.sh

test_slice_scope_parity.py

test_specialist_cli.py

test_specialist_delivery.py

test_specialist_recovery.py

test_specialist_retention.py

test_stale_grok_delivery.py

test_start_judge_worker.py

test_state.py

test_successors.py

test_supervision_cli.py

test_supervision_diagnostics.py

test_supervision_gate.py

test_supervision_replay.py

test_supervision.py

test_sweep_worktrees.sh

test_tier_integration.py

test_tiers.py

test_triggers.py

test_typesafe_client.py

test_verdict_gates.py

test_verify_authority.sh

test_wait_report.sh

tier_fixture.py

bounded-run.sh

compose-briefs.sh

config.example.json

foreman-tier-check.py

foreman.sh

label-workspaces.sh

provision-worktree.sh

prune-remote-branches.sh

prune-report-caches.py

prune-worktrees.sh

resolve-gates.sh

resolve-policy-paths.sh

review-package.sh

roster.sh

round-preflight.sh

SKILL.md

start-judge-worker.sh

state-schema.md

sweep-worktrees.sh

verify-authority.sh

wait-report.sh

README.md

tile.json