CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/coding-policy

General-purpose coding policy for Baruch's AI agents

74

Quality

93%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide
SecuritybySnyk

Medium

Suggest reviewing before use

Overview
Quality
Evals
Security
Files

evaluate.shskills/herdr-foreman/classify/

#!/usr/bin/env bash
# Run the report classifier over the labelled corpus and score it.
#
# The labels are not invented for this: the foreman recorded a verdict against
# every delivered report at the time it gated the round, and those verdicts sit
# in the recovery store. Written by a different agent, on a different day, for a
# different purpose -- which is what makes them usable as ground truth for a
# classifier written afterwards.
#
# Only 7 of the first 90 carry a `## BLOCKED` heading. A grep would score about
# 8% on recall, which is the measurement that says this destination is a
# classifier and not a script.
#
# Held out by default: a question or model change is scored only on reports
# recorded on or after the date the questions last changed (`changed` in
# report-questions.json), which the change cannot have been written against.
#
# Usage: evaluate.sh [--agent jev|codex|claude|grok] [--limit N] [--model <id>]
#                    [--state FILE] [--since DATE | --all] [--fixtures]
#                    [--results FILE] [--corpus-only]
#
#   --agent        the adapter scored, passed through explicitly so every
#                  label comes from that one adapter. Default claude.
#   --since DATE   score reports recorded on or after DATE (ISO). Defaults to
#                  the questions' `changed` date.
#   --all          score the whole corpus; the split is marked not held out.
#   --fixtures     also score the adversarial reports adversarial.py builds
#                  into this run's temp dir, reported apart from the corpus.
#   --results FILE keep every label with its recorded verdict and date, the
#                  input to `scoring.py calibrate`.
#   --corpus-only  build and print the labelled corpus, call no model, spend no
#                  quota. Use it to see what would be scored.
#   --limit N      score the N most recent reports instead of all of them.
#
# Output contract (rules/script-delegation.md -- structured stdout):
#   stdout: one JSON object, `scoring.py score`'s report --
#     {"schema_version": 2, "agent", "model", "split": {"since", "changed",
#      "held_out"}, "scored", "failed", "accuracy", "confusion",
#      "per_question": {<id>: {"determined", "agree", "unclear", "accuracy"}},
#      "fixtures": {"scored", "per_question", "flipped"}, "disagreements"}
#   `disagreements` is the useful half: a label the classifier and the foreman
#   disagree on is either a classifier error or a report whose verdict was
#   never legible from its own text, and only reading it says which.
#   stderr: per-report progress.
#
# Exit 0 when every selected report was scored. Exit 1 when any classification
# failed -- a partial score is reported, never silently averaged over fewer.
# Exit 2 on a usage or tool error.
#
# THIS SPENDS MODEL QUOTA: one call per report. `--corpus-only` spends none.

set -uo pipefail

# Command substitution strips every trailing newline, so the script directory
# never passes through one bare: parameter expansion derives it (#487), and a
# sentinel carries `pwd` across the strip (#466).
case "${BASH_SOURCE[0]}" in
  */*) HERE_SRC="${BASH_SOURCE[0]%/*}" ;;
  *) HERE_SRC=. ;;
esac
if ! HERE="$(CDPATH='' cd -- "${HERE_SRC:-/}" && pwd && printf x)"; then
  echo "evaluate: cannot enter the script directory ${HERE_SRC:-/} — restore read and search access to the plugin directory, or reinstall the plugin, then re-run" >&2
  exit 2
fi
HERE="${HERE%x}"
HERE="${HERE%$'\n'}"

SCRATCH=""
cleanup() { if [ -n "$SCRATCH" ]; then rm -rf "$SCRATCH"; fi; return 0; }
trap cleanup EXIT

die() { echo "evaluate: $*" >&2; exit 2; }

corpus() { # <state-file-or-empty> <limit> <since-or-empty> <state-root> <skill-dir>
  # An empty state file reads the default home. That run checks the home and
  # reads the store, and checks every report path it names, under one shared
  # hold of the home guard
  # (skills/herdr-foreman/foreman/home.py `guard`, `require_current`), so a
  # migrate-home starting between the check and the read is refused instead of
  # leaving a half-moved or empty corpus. An explicit --state is never moved
  # and takes no guard.
  XDG_STATE_HOME="$4" PYTHONPATH="$5" python3 - "$1" "$2" "${3-}" <<'PY'
import json, pathlib, sys

explicit, limit, since = sys.argv[1], int(sys.argv[2]), sys.argv[3]


def read(state):
    try:
        return json.loads(state.read_text(encoding="utf-8"))
    except (OSError, ValueError) as exc:
        sys.stderr.write("evaluate: cannot read {}: {}. Restore it as readable UTF-8 JSON (the foreman's state "
                         "file), or pass --state with a readable copy.\n".format(state, exc))
        raise SystemExit(2)


def build(data):
    rows, seen = [], set()
    for dispatch in data.get("recovery", {}).get("dispatches", []):
        report = dispatch.get("report") or {}
        evidence = report.get("evidence")
        verdict = report.get("verdict")
        if not (isinstance(evidence, dict) and evidence.get("path")) or verdict not in {"blocking", "approved"}:
            continue
        path = pathlib.Path(evidence["path"])
        # ISO timestamps order as strings, so a date prefix selects everything
        # recorded on or after it.
        if since and dispatch.get("at", "") < since:
            continue
        if path in seen or not path.is_file():
            continue
        seen.add(path)
        rows.append({"report": str(path), "recorded": verdict, "at": dispatch.get("at", ""),
                     "role": dispatch.get("role"), "task": dispatch.get("task"), "source": "corpus"})
    rows.sort(key=lambda row: row["at"], reverse=True)
    return json.dumps(rows[:limit] if limit > 0 else rows)


if explicit:
    corpus = build(read(pathlib.Path(explicit).expanduser()))
else:
    from foreman import home
    from foreman.errors import ForemanError
    try:
        # Held through the report-file checks too: reports can live under the
        # home, so a migration after the read could still drop them.
        with home.guard(False):
            home.require_current({"state"})
            corpus = build(read(home.roots()["state"] / home.CURRENT / home.STATE_FILE))
    except ForemanError as exc:
        sys.stderr.write("evaluate: {} Or pass --state.\n".format(exc.message))
        raise SystemExit(2)
print(corpus)
PY
}

with_fixtures() { # <corpus.json> <built-fixtures.json>
  python3 - "$1" "$2" <<'PY'
import json, pathlib, sys
selected, built = pathlib.Path(sys.argv[1]), pathlib.Path(sys.argv[2])
rows = json.loads(selected.read_text(encoding="utf-8"))
for row in json.loads(built.read_text(encoding="utf-8"))["fixtures"]:
    rows.append({"report": row["report"], "recorded": row["verdict"], "at": "",
                 "role": None, "task": None, "source": "fixture", "expected_answers": row["answers"]})
selected.write_text(json.dumps(rows), encoding="utf-8")
PY
}

main() {
  local state_root="${XDG_STATE_HOME:-${HOME}/.local/state}"
  # HERE is absolute, so its parent comes from parameter expansion too (#487).
  local skill_dir="${HERE%/*}"
  skill_dir="${skill_dir:-/}"
  local limit=0 since="" all=0 model="" agent="claude" state="" corpus_only=0 fixtures=0 keep=""
  while [ $# -gt 0 ]; do
    case "$1" in
      --limit) limit="${2-}"; shift 2 || die "--limit needs a count" ;;
      --since) since="${2-}"; shift 2 || die "--since needs an ISO date" ;;
      --all) all=1; shift ;;
      --model) model="${2-}"; shift 2 || die "--model needs an id" ;;
      --agent) agent="${2-}"; shift 2 || die "--agent needs jev, codex, claude or grok" ;;
      --state) state="${2-}"; shift 2 || die "--state needs a file" ;;
      --fixtures) fixtures=1; shift ;;
      --results) keep="${2-}"; shift 2 || die "--results needs a file" ;;
      --corpus-only) corpus_only=1; shift ;;
      -h|--help) sed -n '2,50p' "${BASH_SOURCE[0]}"; exit 0 ;;
      *) die "unknown argument '$1' -- see --help" ;;
    esac
  done
  case "$limit" in ''|*[!0-9]*) die "--limit takes a non-negative integer" ;; esac
  case "$agent" in jev|codex|claude|grok) ;; *) die "--agent '${agent}' is not one of jev, codex, claude, grok" ;; esac
  [ "$all" -eq 0 ] || [ -z "$since" ] || die "--since and --all contradict each other; pass one"
  local changed
  changed="$(python3 -c 'import json,sys; print(json.load(open(sys.argv[1]))["changed"])' "${HERE}/report-questions.json")" \
    || die "cannot read the questions' change date from ${HERE}/report-questions.json; restore that file (reinstall the plugin: tessl install jbaruch/coding-policy), then rerun"
  if [ "$all" -eq 0 ] && [ -z "$since" ]; then since="$changed"; fi
  local work
  work="$(mktemp -d "${TMPDIR:-/tmp}/classify-eval.XXXXXX")" || die "cannot create a temporary directory; repair write access to \${TMPDIR:-/tmp} (or point TMPDIR at a writable directory), then rerun"
  SCRATCH="$work"

  local split="${work}/split.json"
  python3 -c 'import json,sys; since, changed = sys.argv[2], sys.argv[3]
json.dump({"since": since or None, "changed": changed, "held_out": bool(since) and since >= changed}, open(sys.argv[1], "w"))' \
    "$split" "$since" "$changed" || die "cannot record the split in ${work}; repair write access to \${TMPDIR:-/tmp} (or point TMPDIR at a writable directory), then rerun"
  if [ -z "$since" ] || [[ "$since" < "$changed" ]]; then
    echo "evaluate: not held out -- the questions changed on ${changed}, and reports before it may be what they were written against" >&2
  fi

  local selected="${work}/corpus.json"
  local source="${state:-the default home under ${state_root}}"
  corpus "$state" "$limit" "$since" "$state_root" "$skill_dir" > "$selected" \
    || die "cannot build the labelled corpus from ${source}; fix the cause reported above, or pass --state with a readable state file"
  if [ "$fixtures" -eq 1 ]; then
    mkdir "${work}/fixtures" || die "cannot create ${work}/fixtures; repair write access to \${TMPDIR:-/tmp} (or point TMPDIR at a writable directory), then rerun"
    python3 "${HERE}/adversarial.py" "${work}/fixtures" > "${work}/fixtures.json" \
      || die "cannot build the adversarial reports; see the diagnostic above"
    with_fixtures "$selected" "${work}/fixtures.json" || die "cannot add the adversarial reports to ${selected}; repair write access to \${TMPDIR:-/tmp} (or point TMPDIR at a writable directory), then rerun"
  fi
  local total
  total="$(python3 -c 'import json,sys; print(len(json.load(open(sys.argv[1]))))' "$selected")" \
    || die "cannot read the corpus it just built at ${selected}; repair write access to \${TMPDIR:-/tmp} (or point TMPDIR at a writable directory), then rerun"
  case "$total" in ''|*[!0-9]*) die "the corpus count '${total}' is not a number; inspect ${selected}" ;; esac
  if [ "$total" -eq 0 ]; then
    if [ -n "$since" ]; then
      die "no labelled report was recorded on or after ${since}; run more rounds, pass an earlier --since, or --all for a score that is not held out"
    fi
    die "the corpus is empty: no delivered report carries a recorded verdict"
  fi

  if [ "$corpus_only" -eq 1 ]; then
    python3 -c 'import json,sys,collections; rows=json.load(open(sys.argv[1])); split=json.load(open(sys.argv[2]))
print(json.dumps({"schema_version":2,"scored":0,"split":split,"corpus":sum(r["source"]=="corpus" for r in rows),"fixtures":sum(r["source"]=="fixture" for r in rows),"recorded":dict(collections.Counter(r["recorded"] for r in rows if r["source"]=="corpus")),"reports":rows}, sort_keys=True))' "$selected" "$split" \
      || die "cannot summarize the corpus at ${selected}; repair write access to \${TMPDIR:-/tmp} (or point TMPDIR at a writable directory), then rerun"
    return 0
  fi

  echo "evaluate: scoring ${total} report(s) on ${agent}; this spends one model call each" >&2
  local results="${work}/results.json" failures=0 index=0 report
  printf '[]' > "$results" || die "cannot write to ${work}; repair write access to \${TMPDIR:-/tmp} (or point TMPDIR at a writable directory), then rerun"
  for ((index = 0; index < total; index++)); do
    report="$(python3 -c 'import json,sys; print(json.load(open(sys.argv[1]))[int(sys.argv[2])]["report"])' "$selected" "$index")" \
      || die "cannot read row ${index} of ${selected}; repair write access to \${TMPDIR:-/tmp} (or point TMPDIR at a writable directory), then rerun"
    echo "  [$((index + 1))/${total}] ${report}" >&2
    local answer="${work}/answer-${index}.json"
    if bash "${HERE}/classify-report.sh" "$report" --agent "$agent" ${model:+--model "$model"} --out "$answer" >/dev/null 2>"${work}/err-${index}"; then
      if ! python3 - "$results" "$answer" "$selected" "$index" <<'PY'
import json, sys
results, answer, selected, index = sys.argv[1], sys.argv[2], sys.argv[3], int(sys.argv[4])
with open(results, encoding="utf-8") as handle:
    rows = json.load(handle)
with open(answer, encoding="utf-8") as handle:
    label = json.load(handle)
with open(selected, encoding="utf-8") as handle:
    row = json.load(handle)[index]
label.update(recorded=row["recorded"], recorded_at=row["at"], source=row["source"])
if "expected_answers" in row:
    label["expected_answers"] = row["expected_answers"]
rows.append(label)
with open(results, "w", encoding="utf-8") as handle:
    json.dump(rows, handle)
PY
      then die "cannot record the label for ${report} in ${results}; repair write access to \${TMPDIR:-/tmp} (or point TMPDIR at a writable directory), then rerun"; fi
    else
      failures=$((failures + 1))
      cat "${work}/err-${index}" >&2
    fi
  done

  # Every selected report is either a recorded label or a counted failure; a
  # shortfall means the loop lost rows, and a partial score is never reported
  # as a whole one.
  local labelled
  labelled="$(python3 -c 'import json,sys; print(len(json.load(open(sys.argv[1]))))' "$results")" \
    || die "cannot count the labels in ${results}; repair write access to \${TMPDIR:-/tmp} (or point TMPDIR at a writable directory), then rerun"
  if [ $((labelled + failures)) -ne "$total" ]; then
    die "scored ${labelled} and failed ${failures} of ${total} selected reports; the run lost reports, so no score is reported"
  fi

  if [ -n "$keep" ]; then
    cp "$results" "$keep" || die "cannot keep the labels at ${keep}; pass a writable --results path, then rerun"
  fi
  python3 "${HERE}/scoring.py" score "$results" "$failures" "$agent" "${model:-pinned}" "$split" \
    || die "cannot assemble the accuracy report from ${results}"
  [ "$failures" -eq 0 ] || return 1
  return 0
}

[[ "${BASH_SOURCE[0]}" == "${0}" ]] && main "$@"

skills

herdr-foreman

bounded-run.sh

compose-briefs.sh

config.example.json

foreman-tier-check.py

foreman.sh

label-workspaces.sh

provision-worktree.sh

prune-remote-branches.sh

prune-report-caches.py

prune-worktrees.sh

resolve-gates.sh

resolve-policy-paths.sh

review-package.sh

roster.sh

round-preflight.sh

SKILL.md

start-judge-worker.sh

state-schema.md

sweep-worktrees.sh

verify-authority.sh

wait-report.sh

README.md

tile.json