CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/coding-policy

General-purpose coding policy for Baruch's AI agents

73

Quality

91%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide

SecuritybySnyk

Low

Low-risk findings worth noting

Overview
Quality
Evals
Security
Files

test_classify.shskills/herdr-foreman/tests/

#!/usr/bin/env bash
# Outcome-based tests for skills/herdr-foreman/classify/.
#
# The model is stubbed on PATH, never called: a classifier's output is
# non-deterministic, and calling one live would put that into the suite
# (rules/testing-standards.md Determinism, which names this case). Jev's HTTP
# layer is stubbed in tests/test_report_verdict.py; here TYPESAFE_API_KEY is
# unset, so every default run labels nothing and leaves the report for a full
# read. What is under test is everything around the call -- the answer
# contract, the unannotated path, the failure paths, and the scoring.
#
# The harness drops `set -e` to aggregate results; every fixture command is
# checked explicitly (rules/error-handling.md aggregate-reporting carve-out).
#
# Covers:
#   1. A conforming answer      -> composed verdict, report hash, question hash, model.
#   2. A failed model call      -> exit 2, no stdout. Never a verdict.
#   3. An off-schema answer     -> exit 2. The schema is not advisory.
#   4. An empty answer          -> exit 2.
#   5. An unreadable report     -> exit 2 before any model call.
#   6. --model                  -> overrides the pin and travels with the label.
#   7. --out                    -> the same payload, byte for byte.
#   8. A fabricated quote       -> insufficient_evidence, never the model's verdict.
#   9. The default adapter      -> Jev unavailable labels nothing and asks no
#                                 other classifier; the report is read in full.
#  10. Claude and Grok adapters -> each vendor's envelope unwrapped to the same
#                                 label; an errored or multi-answer run refused.
#  11. The empty room           -> every adapter runs where it can read nothing
#                                 but the question.
#  12. A round's batch          -> one call annotates every report; a failed
#                                 annotation is reported, never fatal.
#  13. Corpus building          -> recorded verdicts only, missing files dropped;
#                                 held out by default; the default home is read
#                                 under the home guard.
#  14. Scoring                  -> accuracy, confusion, per-question accuracy,
#                                 disagreements, fixtures reported apart.
#  15. A failed classification  -> exit 1 with a partial score, never averaged
#                                 over the ones that worked.
#  16. A newline-named dir      -> every script still finds its siblings.

set -uo pipefail

PASS=0
FAIL=0
pass() { PASS=$((PASS + 1)); }
fail() { FAIL=$((FAIL + 1)); echo "  ✗ FAIL: $*" >&2; }
die() { echo "fatal: $*" >&2; exit 2; }

DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/../classify" && pwd)" || die "cannot resolve the classify dir"

BLOCKING='{"names_open_item":{"answer":"yes","evidence":"B1: the parser accepts a quoted completion marker."},"open_items_accepted":{"answer":"no","evidence":""},"open_items_out_of_scope":{"answer":"no","evidence":""},"concludes_nothing_blocks":{"answer":"no","evidence":""}}'
APPROVED='{"names_open_item":{"answer":"no","evidence":""},"open_items_accepted":{"answer":"no","evidence":""},"open_items_out_of_scope":{"answer":"no","evidence":""},"concludes_nothing_blocks":{"answer":"yes","evidence":"No blocking findings."}}'
INVENTED='{"names_open_item":{"answer":"yes","evidence":"B7: the release step deletes the tag."},"open_items_accepted":{"answer":"no","evidence":""},"open_items_out_of_scope":{"answer":"no","evidence":""},"concludes_nothing_blocks":{"answer":"no","evidence":""}}'

# A `codex` that writes whatever answer the fixture asks for, into the file the
# real one would write, and ignores everything else.
stub_codex() { # <bin-dir> <exit> <answer-json>
  mkdir -p "$1" || die "mkdir $1"
  cat > "$1/codex" <<STUB || die "write codex stub"
#!/usr/bin/env bash
set -euo pipefail
out=""
while [ \$# -gt 0 ]; do
  case "\$1" in --output-last-message) out="\$2"; shift 2 ;; *) shift ;; esac
done
cat > /dev/null
if [ -n "\$out" ]; then printf '%s' '$3' > "\$out"; fi
exit $2
STUB
  chmod +x "$1/codex" || die "chmod codex stub"
}

# A `claude` that answers with the given object, and records what it could see.
stub_claude() { # <bin-dir> <answer-json>
  mkdir -p "$1" || die "mkdir $1"
  cat > "$1/claude" <<STUB || die "write claude stub"
#!/usr/bin/env bash
set -euo pipefail
cat > /dev/null
if [ -n "\${ROOM_PROBE:-}" ]; then ls -A > "\$ROOM_PROBE"; fi
printf '%s' '[{"type":"system"},{"type":"result","is_error":false,"structured_output":$2}]'
STUB
  chmod +x "$1/claude" || die "chmod claude stub"
}

classify() { # <bin-dir> <report> [extra args...]
  local bin="$1" report="$2"; shift 2
  OUT="$(PATH="$bin:$PATH" bash "$DIR/classify-report.sh" "$report" "$@" 2>"$ERRFILE")"
  RC=$?
  ERRTEXT="$(cat "$ERRFILE")"
}

field() { printf '%s' "$1" | python3 -c 'import json,sys; print(json.load(sys.stdin)[sys.argv[1]])' "$2"; }

# An EXIT trap's final status becomes the script's, so cleanup ends on zero.
cleanup() { if [ -n "${TMP:-}" ]; then rm -rf "$TMP"; fi; return 0; }

main() {
  # The suite never reaches TypeSafe, whatever the runner's environment holds.
  unset TYPESAFE_API_KEY
  TMP="$(mktemp -d "${TMPDIR:-/tmp}/classify-tests.XXXXXX")" || die "mktemp"
  trap cleanup EXIT
  ERRFILE="$TMP/err"

  local report="$TMP/report.md" second="$TMP/second.md"
  printf '# Reviewer report\n\nB1: the parser accepts a quoted completion marker.\n' > "$report" \
    || die "write report fixture"
  printf '# Tester report\n\nNo blocking findings.\n' > "$second" || die "write second report"

  echo "▶ the answer contract" >&2

  stub_codex "$TMP/ok" 0 "$BLOCKING"
  classify "$TMP/ok" "$report" --agent codex
  if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" verdict)" == "blocking" ]] \
     && [[ "$(field "$OUT" model)" == "gpt-5.6-sol" ]] \
     && [[ "$(field "$OUT" evidence)" == "B1: the parser accepts a quoted completion marker." ]]; then
    pass; else fail "a conforming answer composes its verdict with the pinned model, got RC=$RC OUT=$OUT ERR=$ERRTEXT"; fi

  # The label carries what makes a later question edit or model bump attributable.
  local expected_report expected_question
  expected_report="$(shasum -a 256 "$report" | cut -d' ' -f1)"
  expected_question="$(cat "$DIR/report-verdict.prompt.md" <(printf '\0') "$DIR/report-questions.json" | shasum -a 256 | cut -d' ' -f1)"
  if [[ "$(field "$OUT" sha256)" == "$expected_report" ]] \
     && [[ "$(field "$OUT" question)" == "$expected_question" ]]; then
    pass; else fail "the label must carry the report and question hashes, got OUT=$OUT"; fi

  classify "$TMP/ok" "$report" --agent codex --model "some-other-model"
  if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" model)" == "some-other-model" ]]; then
    pass; else fail "--model overrides the pin and travels with the label, got RC=$RC OUT=$OUT"; fi

  classify "$TMP/ok" "$report" --model "some-other-model"
  if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q 'no Jev label'; then
    pass; else fail "--model without --agent names a Jev model and asks nothing else, got RC=$RC OUT=$OUT"; fi

  classify "$TMP/ok" "$report" --agent codex --out "$TMP/label.json"
  if [[ $RC -eq 0 ]] && [[ "$(cat "$TMP/label.json")" == "$OUT" ]]; then
    pass; else fail "--out writes the same payload, got RC=$RC"; fi

  echo "▶ deterministic checks before semantic ones" >&2

  # A quote that is not in the report voids the verdict it claims to back.
  stub_codex "$TMP/invented" 0 "$INVENTED"
  classify "$TMP/invented" "$report" --agent codex
  if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" verdict)" == "insufficient_evidence" ]] \
     && printf '%s' "$OUT" | grep -q '"evidence_verbatim": false'; then
    pass; else fail "a fabricated quote fails closed to insufficient_evidence, got RC=$RC OUT=$OUT"; fi

  # A report rewritten while the model reads it: the label hashes and checks
  # the bytes the model was given, never the rewrite.
  local moving="$TMP/moving.md" before
  cp "$report" "$moving" || die "copy moving report"
  before="$(shasum -a 256 "$moving" | cut -d' ' -f1)"
  mkdir -p "$TMP/rewrite" || die "mkdir rewrite"
  cat > "$TMP/rewrite/codex" <<STUB || die "write rewriting codex stub"
#!/usr/bin/env bash
set -euo pipefail
out=""
while [ \$# -gt 0 ]; do
  case "\$1" in --output-last-message) out="\$2"; shift 2 ;; *) shift ;; esac
done
cat > /dev/null
printf 'B1 is withdrawn.\n' > "$moving"
printf '%s' '$BLOCKING' > "\$out"
STUB
  chmod +x "$TMP/rewrite/codex" || die "chmod rewriting codex stub"
  classify "$TMP/rewrite" "$moving" --agent codex
  if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" sha256)" == "$before" ]] \
     && [[ "$(field "$OUT" verdict)" == "blocking" ]] && [[ "$(field "$OUT" report)" == "$moving" ]]; then
    pass; else fail "a mid-run rewrite never changes the bytes a label hashes, got RC=$RC OUT=$OUT"; fi

  echo "▶ a failed call is never a verdict" >&2

  stub_codex "$TMP/down" 1 ''
  classify "$TMP/down" "$report" --agent codex
  if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q 'never a verdict'; then
    pass; else fail "a failed model call exits 2 with no verdict, got RC=$RC OUT=$OUT"; fi

  # The whole schema is the contract: a missing question, an answer off the
  # enum, a missing evidence field or an extra field is no label at all.
  local shape rest='"open_items_accepted":{"answer":"no","evidence":""},"open_items_out_of_scope":{"answer":"no","evidence":""},"concludes_nothing_blocks":{"answer":"no","evidence":""}'
  for shape in '{"names_open_item":{"answer":"yes","evidence":"x"}}' \
               '{"names_open_item":{"answer":"probably","evidence":""},'"$rest"'}' \
               '{"names_open_item":{"answer":"no"},'"$rest"'}' \
               '{"names_open_item":{"answer":"no","evidence":""},'"$rest"',"verdict":"approved"}'; do
    stub_codex "$TMP/shape" 0 "$shape"
    classify "$TMP/shape" "$report" --agent codex
    if [[ $RC -eq 2 && -z "$OUT" ]]; then
      pass; else fail "answer '$shape' must be refused, got RC=$RC OUT=$OUT"; fi
  done

  stub_codex "$TMP/empty" 0 ''
  classify "$TMP/empty" "$report" --agent codex
  if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q 'no schema-conforming answer'; then
    pass; else fail "an empty answer exits 2, got RC=$RC OUT=$OUT"; fi

  classify "$TMP/ok" "$TMP/absent.md" --agent codex
  if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q 'not a readable file'; then
    pass; else fail "an unreadable report exits 2 before any call, got RC=$RC OUT=$OUT"; fi

  stub_codex "$TMP/abstain" 0 '{"names_open_item":{"answer":"unclear","evidence":""},'"$rest"'}'
  classify "$TMP/abstain" "$report" --agent codex
  if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" verdict)" == "insufficient_evidence" ]]; then
    pass; else fail "an honest abstention is an answer, not a failure, got RC=$RC OUT=$OUT"; fi

  echo "▶ the default adapter asks no other classifier" >&2

  stub_claude "$TMP/cl" "$APPROVED"
  OUT="$(ROOM_PROBE="$TMP/cl-default-room" PATH="$TMP/cl:$PATH" bash "$DIR/classify-report.sh" "$second" 2>"$ERRFILE")"
  RC=$?
  ERRTEXT="$(cat "$ERRFILE")"
  if [[ $RC -eq 2 && -z "$OUT" && ! -e "$TMP/cl-default-room" ]] \
     && printf '%s' "$ERRTEXT" | grep -q 'no Jev label (Jev unavailable: TYPESAFE_API_KEY is not set' \
     && printf '%s' "$ERRTEXT" | grep -q 'in full'; then
    pass; else fail "an unset key labels nothing and never calls claude, got RC=$RC OUT=$OUT ERR=$ERRTEXT"; fi

  OUT="$(ROOM_PROBE="$TMP/cl-room" PATH="$TMP/cl:$PATH" bash "$DIR/classify-report.sh" "$second" --agent claude 2>"$ERRFILE")"
  RC=$?
  if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" agent)" == "claude" ]] && [[ "$(field "$OUT" model)" == "claude-sonnet-5" ]] \
     && printf '%s' "$OUT" | python3 -c 'import json,sys; assert json.load(sys.stdin)["gate"]["level"] is None'; then
    pass; else fail "an explicit claude label never gates, got RC=$RC OUT=$OUT"; fi
  # The adapter ran somewhere it could read nothing but the question.
  if [[ -f "$TMP/cl-room" && ! -s "$TMP/cl-room" ]]; then
    pass; else fail "the claude adapter must run in an empty directory, saw: $(cat "$TMP/cl-room")"; fi

  echo "▶ one adapter per kind" >&2

  cat > "$TMP/cl/claude" <<'STUB' || die "write erroring claude stub"
#!/usr/bin/env bash
set -euo pipefail
cat > /dev/null
printf '%s' '[{"type":"result","is_error":true,"result":"usage limit reached"}]'
STUB
  classify "$TMP/cl" "$report" --agent claude
  if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q 'never a verdict'; then
    pass; else fail "an errored claude run is never a verdict, got RC=$RC OUT=$OUT"; fi

  # Grok puts the answer in `text`, and one turn gives exactly one object.
  mkdir -p "$TMP/gk" || die "mkdir gk"
  python3 - "$TMP/gk/grok" "$BLOCKING" <<'PY' || die "write grok stub"
import json, sys
path, answer = sys.argv[1], sys.argv[2]
envelope = json.dumps({"text": answer, "stopReason": "end_turn"})
with open(path, "w", encoding="utf-8") as handle:
    handle.write("#!/usr/bin/env bash\nset -euo pipefail\nls -A > \"$ROOM_PROBE\"\nprintf '%s' '{}'\n".format(envelope))
PY
  chmod +x "$TMP/gk/grok" || die "chmod grok stub"
  OUT="$(ROOM_PROBE="$TMP/gk-room" PATH="$TMP/gk:$PATH" bash "$DIR/classify-report.sh" "$report" --agent grok 2>"$ERRFILE")"
  RC=$?
  if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" verdict)" == "blocking" ]] && [[ "$(field "$OUT" agent)" == "grok" ]]; then
    pass; else fail "grok's envelope unwraps to a label, got RC=$RC OUT=$OUT"; fi
  if [[ -f "$TMP/gk-room" && ! -s "$TMP/gk-room" ]]; then
    pass; else fail "the grok adapter must run in an empty directory, saw: $(cat "$TMP/gk-room")"; fi

  # A live probe before these adapters existed: free to roam, grok searched the
  # workspace and emitted four concatenated answers. Picking one out is not the
  # answer to the question asked.
  cat > "$TMP/gk/grok" <<'STUB' || die "write chatty grok stub"
#!/usr/bin/env bash
set -euo pipefail
printf '%s' '{"text":"{\"a\":1}{\"b\":2}"}'
STUB
  classify "$TMP/gk" "$report" --agent grok
  if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q 'more than one answer'; then
    pass; else fail "more than one grok answer is refused, got RC=$RC OUT=$OUT ERR=$ERRTEXT"; fi

  classify "$TMP/gk" "$report" --agent gemini
  if [[ $RC -eq 2 && -z "$OUT" ]]; then
    pass; else fail "an unknown agent is a usage error, got RC=$RC OUT=$OUT"; fi

  echo "▶ a round's batch" >&2

  stub_claude "$TMP/batch" "$APPROVED"
  OUT="$(PATH="$TMP/batch:$PATH" bash "$DIR/classify-reports.sh" --agent claude "$second" "$second" 2>"$ERRFILE")"
  RC=$?
  if [[ $RC -eq 0 ]] && printf '%s' "$OUT" | python3 -c '
import json, sys
d = json.load(sys.stdin)
assert d["agent"] == "claude", d
assert len(d["labels"]) == 2 and d["unannotated"] == [], d
assert all(label["agent"] == "claude" for label in d["labels"]), d
'; then
    pass; else fail "one call annotates every report, got RC=$RC OUT=$OUT"; fi

  # The default batch with Jev unavailable: every report unannotated, read in full.
  OUT="$(ROOM_PROBE="$TMP/batch-room" PATH="$TMP/batch:$PATH" bash "$DIR/classify-reports.sh" "$second" 2>"$ERRFILE")"
  RC=$?
  if [[ $RC -eq 0 && ! -e "$TMP/batch-room" ]] && printf '%s' "$OUT" | python3 -c '
import json, sys
d = json.load(sys.stdin)
assert d["agent"] == "default" and d["labels"] == [], d
assert len(d["unannotated"]) == 1 and "TYPESAFE_API_KEY is not set" in d["unannotated"][0]["reason"], d
'; then
    pass; else fail "an unavailable Jev leaves the batch unannotated, got RC=$RC OUT=$OUT"; fi

  # An annotation that fails never blocks gating: the foreman reads that report as
  # it always has.
  OUT="$(PATH="$TMP/down:$PATH" bash "$DIR/classify-reports.sh" --agent codex "$report" "$second" 2>"$ERRFILE")"
  RC=$?
  if [[ $RC -eq 0 ]] && printf '%s' "$OUT" | python3 -c '
import json, sys
d = json.load(sys.stdin)
assert d["labels"] == [] and len(d["unannotated"]) == 2, d
assert all("never a verdict" in row["reason"] for row in d["unannotated"]), d
'; then
    pass; else fail "failed annotations are reported with their reason, never fatal, got RC=$RC OUT=$OUT"; fi

  OUT="$(bash "$DIR/classify-reports.sh" 2>"$ERRFILE")"
  RC=$?
  if [[ $RC -eq 2 && -z "$OUT" ]]; then
    pass; else fail "no reports is a usage error, got RC=$RC OUT=$OUT"; fi

  echo "▶ the labelled corpus" >&2

  local kept="$TMP/kept.md" state="$TMP/state.json"
  cp "$report" "$kept" || die "write kept report"
  python3 - "$state" "$kept" <<'PY' || die "write state fixture"
import json, sys
state, kept = sys.argv[1], sys.argv[2]
dispatches = [
    {"at": "2026-09-03T00:00:00Z", "role": "reviewer", "task": "t",
     "report": {"verdict": "blocking", "evidence": {"path": kept}}},
    {"at": "2026-09-02T00:00:00Z", "role": "tester", "task": "t",
     "report": {"verdict": "approved", "evidence": {"path": kept}}},
    {"at": "2026-09-01T00:00:00Z", "role": "reviewer", "task": "t",
     "report": {"verdict": "blocking", "evidence": {"path": "/nope/gone.md"}}},
    {"at": "2026-09-04T00:00:00Z", "role": "developer", "task": "t"},
]
with open(state, "w", encoding="utf-8") as handle:
    json.dump({"recovery": {"dispatches": dispatches}}, handle)
PY
  OUT="$(bash "$DIR/evaluate.sh" --corpus-only --state "$state" --all 2>"$ERRFILE")"
  RC=$?
  # The same path twice is one report: a corpus that counted it twice would
  # score the same bytes twice. A missing file and a dispatch with no recorded
  # verdict are both dropped.
  if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" corpus)" == "1" ]] \
     && printf '%s' "$OUT" | grep -q '"held_out": false'; then
    pass; else fail "the corpus keeps recorded verdicts with readable files only, got RC=$RC OUT=$OUT"; fi

  # Held out by default: reports recorded before the questions changed are the
  # ones they may have been written against.
  OUT="$(bash "$DIR/evaluate.sh" --corpus-only --state "$state" 2>"$ERRFILE")"
  RC=$?
  ERRTEXT="$(cat "$ERRFILE")"
  if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q 'on or after 2026-09-27'; then
    pass; else fail "the default split is held out from the questions' change date, got RC=$RC ERR=$ERRTEXT"; fi

  OUT="$(bash "$DIR/evaluate.sh" --corpus-only --state "$state" --since 2026-09-03 2>"$ERRFILE")"
  RC=$?
  ERRTEXT="$(cat "$ERRFILE")"
  if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" corpus)" == "1" ]] && printf '%s' "$ERRTEXT" | grep -q 'not held out'; then
    pass; else fail "--since before the change date scores and says it is not held out, got RC=$RC OUT=$OUT"; fi
  OUT="$(bash "$DIR/evaluate.sh" --corpus-only --state "$state" --since 2026-09-04 2>"$ERRFILE")"
  RC=$?
  ERRTEXT="$(cat "$ERRFILE")"
  if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q 'on or after 2026-09-04'; then
    pass; else fail "--since past every report names the date, got RC=$RC ERR=$ERRTEXT"; fi
  OUT="$(bash "$DIR/evaluate.sh" --corpus-only --state "$state" --since 2026-09-03 --all 2>"$ERRFILE")"
  RC=$?
  if [[ $RC -eq 2 && -z "$OUT" ]]; then
    pass; else fail "--since with --all is a usage error, got RC=$RC OUT=$OUT"; fi
  OUT="$(bash "$DIR/evaluate.sh" --corpus-only --state "$state" --all --fixtures 2>"$ERRFILE")"
  RC=$?
  if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" fixtures)" == "3" ]]; then
    pass; else fail "--fixtures adds the adversarial fixtures apart from the corpus, got RC=$RC OUT=$OUT"; fi
  # An unreadable state file is a usage error that names the recovery.
  printf '{not json' > "$TMP/broken-state.json" || die "write broken state fixture"
  OUT="$(bash "$DIR/evaluate.sh" --corpus-only --state "$TMP/broken-state.json" --all 2>"$ERRFILE")"
  RC=$?
  ERRTEXT="$(cat "$ERRFILE")"
  if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q 'readable UTF-8 JSON'; then
    pass; else fail "an unreadable state file names its recovery, got RC=$RC ERR=$ERRTEXT"; fi

  # The default corpus refuses a state home still at the legacy teamlead path
  # rather than reading the new path as empty; a migrated home (legacy path
  # linked to the new one) reads normally.
  local xdg="$TMP/xdg"
  mkdir -p "$xdg/teamlead"
  cp "$state" "$xdg/teamlead/state.json"
  OUT="$(XDG_STATE_HOME="$xdg" bash "$DIR/evaluate.sh" --corpus-only --all 2>"$ERRFILE")"
  RC=$?
  ERRTEXT="$(cat "$ERRFILE")"
  if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q 'migrate-home'; then
    pass; else fail "a legacy default home names migrate-home, got RC=$RC ERR=$ERRTEXT"; fi
  # A legacy path that is not a directory, or a link elsewhere, is refused
  # the same way the owner refuses it, never read past as an empty store.
  local other="$TMP/xdg-other" blocked="$TMP/xdg-blocked"
  mkdir -p "$other/elsewhere" "$other/foreman" "$blocked"
  ln -s "$other/elsewhere" "$other/teamlead"
  printf 'x' > "$blocked/teamlead"
  local root
  for root in "$other" "$blocked"; do
    OUT="$(XDG_STATE_HOME="$root" bash "$DIR/evaluate.sh" --corpus-only --all 2>"$ERRFILE")"
    RC=$?
    ERRTEXT="$(cat "$ERRFILE")"
    if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q 'migrate-home'; then
      pass; else fail "a split or blocked legacy home under $root is refused, got RC=$RC ERR=$ERRTEXT"; fi
  done
  mv "$xdg/teamlead" "$xdg/foreman"
  ln -s "$xdg/foreman" "$xdg/teamlead"
  OUT="$(XDG_STATE_HOME="$xdg" bash "$DIR/evaluate.sh" --corpus-only --all 2>"$ERRFILE")"
  RC=$?
  if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" corpus)" == "1" ]]; then
    pass; else fail "a migrated default home reads its corpus, got RC=$RC OUT=$OUT"; fi
  # A migrate-home in flight holds the home guard exclusively. The default
  # corpus is checked and read under that guard, so it is refused rather than
  # scoring a half-moved or empty store. The holder runs evaluate while it
  # holds the lock, so the overlap is deterministic, never a timing race.
  local held
  held="$(python3 - "$xdg" "$DIR/evaluate.sh" "$ERRFILE" <<'PY'
import fcntl, os, subprocess, sys
xdg, evaluate, errfile = sys.argv[1:4]
with open(os.path.join(xdg, ".foreman-home.lock"), "a", encoding="utf-8") as lock:
    fcntl.flock(lock.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
    with open(errfile, "w", encoding="utf-8") as err:
        run = subprocess.run(["bash", evaluate, "--corpus-only", "--all"], env={**os.environ, "XDG_STATE_HOME": xdg},
                             stdout=subprocess.PIPE, stderr=err, text=True, check=False)
print("{}\t{}".format(run.returncode, len(run.stdout)))
PY
)" || die "cannot hold the home guard under $xdg"
  ERRTEXT="$(cat "$ERRFILE")"
  if [[ "$held" == $'2\t0' ]] && printf '%s' "$ERRTEXT" | grep -q 'migrate-home is moving'; then
    pass; else fail "a default corpus read during migrate-home is refused, got $held ERR=$ERRTEXT"; fi

  echo "▶ scoring" >&2

  OUT="$(PATH="$TMP/ok:$PATH" bash "$DIR/evaluate.sh" --agent codex --state "$state" --all 2>"$ERRFILE")"
  RC=$?
  if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" scored)" == "1" ]] && [[ "$(field "$OUT" accuracy)" == "1.0" ]] \
     && printf '%s' "$OUT" | python3 -c '
import json, sys
d = json.load(sys.stdin)
assert d["per_question"]["names_open_item"] == {"determined": 1, "agree": 1, "unclear": 0, "accuracy": 1.0}, d
assert d["per_question"]["concludes_nothing_blocks"]["accuracy"] is None, d
assert d["split"]["held_out"] is False, d
'; then
    pass; else fail "a correct prediction scores 1.0 with per-question accuracy, got RC=$RC OUT=$OUT"; fi

  stub_codex "$TMP/wrong" 0 "$INVENTED"
  OUT="$(PATH="$TMP/wrong:$PATH" bash "$DIR/evaluate.sh" --agent codex --state "$state" --all 2>"$ERRFILE")"
  RC=$?
  if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" accuracy)" == "0.0" ]] \
     && printf '%s' "$OUT" | grep -q '"recorded": "blocking"' \
     && printf '%s' "$OUT" | grep -q '"predicted": "insufficient_evidence"'; then
    pass; else fail "a disagreement is reported with both sides, got RC=$RC OUT=$OUT"; fi

  # Fixtures are scored apart: a verdict the injected text flipped is named.
  OUT="$(PATH="$TMP/ok:$PATH" bash "$DIR/evaluate.sh" --agent codex --state "$state" --all --fixtures \
    --results "$TMP/results.json" 2>"$ERRFILE")"
  RC=$?
  if [[ $RC -eq 0 ]] && [[ -s "$TMP/results.json" ]] && printf '%s' "$OUT" | python3 -c '
import json, sys
d = json.load(sys.stdin)
assert d["scored"] == 1 and d["fixtures"]["scored"] == 3, d
# The stub quotes B1, which is in no fixture: every fixture fails closed.
assert len(d["fixtures"]["flipped"]) == 3, d
'; then
    pass; else fail "fixtures are scored and reported apart, got RC=$RC OUT=$OUT ERR=$(cat "$ERRFILE")"; fi

  stub_codex "$TMP/broken" 1 ''
  OUT="$(PATH="$TMP/broken:$PATH" bash "$DIR/evaluate.sh" --agent codex --state "$state" --all 2>"$ERRFILE")"
  RC=$?
  # A partial score, never an accuracy averaged over only the calls that worked.
  if [[ $RC -eq 1 ]] && [[ "$(field "$OUT" failed)" == "1" ]] \
     && [[ "$(field "$OUT" scored)" == "0" ]]; then
    pass; else fail "a failed classification exits 1 and is counted, got RC=$RC OUT=$OUT"; fi

  echo "▶ a directory name ending in a newline" >&2

  # A `$(dirname ...)` capture drops the newline, and the scripts then look for
  # the questions, the prompt and each other beside a directory that does not
  # exist (#592). The Python owners resolve from `__file__`.
  local nl_dir="$TMP/nl-skill/classify"$'\n'
  if mkdir -p "$nl_dir" 2>"$TMP/nl.err"; then
    cp -R "$DIR"/. "$nl_dir/" || die "copy the classify dir into $nl_dir"
    cp -R "$DIR/../foreman" "$TMP/nl-skill/foreman" || die "copy the foreman package beside $nl_dir"
    OUT="$(PATH="$TMP/ok:$PATH" bash "$nl_dir/classify-report.sh" "$report" --agent codex 2>"$ERRFILE")"
    RC=$?
    if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" verdict)" == "blocking" ]]; then
      pass; else fail "classify-report.sh in a newline-named dir, got RC=$RC OUT=$OUT ERR=$(cat "$ERRFILE")"; fi
    OUT="$(PATH="$TMP/batch:$PATH" bash "$nl_dir/classify-reports.sh" --agent claude "$second" "$second" 2>"$ERRFILE")"
    RC=$?
    if [[ $RC -eq 0 ]] && printf '%s' "$OUT" | python3 -c '
import json, sys
d = json.load(sys.stdin)
assert len(d["labels"]) == 2 and d["unannotated"] == [], d
'; then
      pass; else fail "classify-reports.sh in a newline-named dir, got RC=$RC OUT=$OUT ERR=$(cat "$ERRFILE")"; fi
    OUT="$(bash "$nl_dir/evaluate.sh" --corpus-only --state "$state" --all 2>"$ERRFILE")"
    RC=$?
    if [[ $RC -eq 0 ]] && [[ "$(field "$OUT" corpus)" == "1" ]]; then
      pass; else fail "evaluate.sh in a newline-named dir, got RC=$RC OUT=$OUT ERR=$(cat "$ERRFILE")"; fi
  else
    echo "  skipped: this filesystem refuses a name ending in a newline ($(cat "$TMP/nl.err"))" >&2
  fi

  echo "─────────────────────────────────────────────" >&2
  if [[ $FAIL -gt 0 ]]; then echo "FAILED: ${FAIL} failed, ${PASS} passed" >&2; exit 1; fi
  echo "PASSED: all ${PASS} checks" >&2
}

[[ "${BASH_SOURCE[0]}" == "${0}" ]] && main "$@"

skills

herdr-foreman

tests

__init__.py

fakes.py

test_assign.py

test_attention.py

test_billing.py

test_bounded_run.sh

test_capabilities.py

test_capability_routing.py

test_chronology.py

test_churn.py

test_classify.sh

test_claude_native.py

test_cli.py

test_compose_briefs.sh

test_composer.py

test_composition.py

test_config.py

test_continuity_cli.py

test_cost_report.py

test_diagnostics.py

test_engagement.py

test_entrypoints.py

test_foreman_launcher.sh

test_foreman_queue.py

test_foreman_reset.py

test_foreman_seat.py

test_foreman_tier_check.py

test_freeze.py

test_herdr.py

test_historical.py

test_home.py

test_label_workspaces.sh

test_launch.py

test_legacy_recovery.py

test_load_set.py

test_measure.py

test_members.py

test_memory.py

test_oracle.py

test_parsers.py

test_partition.py

test_planner.py

test_probe.py

test_provision_worktree.sh

test_prune_remote_branches.sh

test_prune_report_caches.py

test_prune_result.py

test_prune_worktrees.sh

test_recovery_cli.py

test_recovery.py

test_renderable.py

test_report_contract.py

test_report_delivery.py

test_report_gates.py

test_report_verdict.py

test_resolve_gates.sh

test_resolve_policy_paths.py

test_restoration.py

test_retrospective_runtime.py

test_retrospective.py

test_review_package.py

test_role_clear.py

test_roster.sh

test_round_preflight.sh

test_runnable.py

test_scoring.py

test_script_dir_newline.sh

test_seat_holds.py

test_selection.py

test_skill_invocations.sh

test_slice_scope_parity.py

test_specialist_cli.py

test_specialist_delivery.py

test_specialist_recovery.py

test_specialist_retention.py

test_stale_grok_delivery.py

test_start_judge_worker.py

test_state.py

test_supervision_cli.py

test_supervision_diagnostics.py

test_supervision_gate.py

test_supervision_replay.py

test_supervision.py

test_sweep_worktrees.sh

test_tier_integration.py

test_tiers.py

test_triggers.py

test_typesafe_client.py

test_verdict_gates.py

test_verify_authority.sh

test_wait_report.sh

tier_fixture.py

bounded-run.sh

compose-briefs.sh

config.example.json

foreman-tier-check.py

foreman.sh

label-workspaces.sh

provision-worktree.sh

prune-remote-branches.sh

prune-report-caches.py

prune-worktrees.sh

resolve-gates.sh

resolve-policy-paths.sh

review-package.sh

roster.sh

round-preflight.sh

SKILL.md

start-judge-worker.sh

state-schema.md

sweep-worktrees.sh

verify-authority.sh

wait-report.sh

README.md

tile.json