General-purpose coding policy for Baruch's AI agents
73
91%
Does it follow best practices?
Run evals on this skill
Adds up to 20 points to the overall score
View guide
Low
Low-risk findings worth noting
#!/usr/bin/env bash
# Outcome-based tests for skills/herdr-standup/standup-ask.sh.
#
# Every case points HERDR_BIN at a fake this harness writes, which records the
# argv it was handed (rules/testing-standards.md — no live Herdr session).
#
# The harness drops `set -e` to aggregate results, so every fixture-setup
# command is checked explicitly and aborts with a fatal diagnostic on failure
# (rules/error-handling.md aggregate-reporting carve-out).
#
# Covers:
# 1. Idle worker -> prompted, sent true.
# 2. Done worker -> prompted too; done is ready, not busy.
# 3. Working worker -> exit 3, NOTHING sent (a standup never interrupts).
# 4. Blocked worker -> exit 3, nothing sent.
# 5. Message shape -> `agent prompt`, never a slash command, and the four
# field names plus the report path are in the text.
# 6. Relative path -> exit 1 before any herdr call.
# 6b. Over-long path -> exit 1; the coarse length bound, before Herdr.
# 6e. Control char -> exit 1 before Herdr.
# 6f. DEL, C1, U+2028/9 -> exit 1 before Herdr, in a UTF-8 locale too.
# 6g. Non-ASCII path -> a letter such as U+00E9 is asked normally.
# 6c. Bad limit -> a non-integer override is exit 1, not an abort.
# 6d. `0100` is 100 -> a leading zero is decimal, never octal, downstream.
# 7. Outside Herdr -> exit 1.
# 8. herdr failure -> exit 2, no verdict.
# 9. Bad payload -> exit 2.
# 10. Narrow pane -> exit 4, nothing sent, the verdict names the width
# and the columns the marker needs (#515).
# 11. Exact fit -> a pane exactly as wide as that need is asked.
# 11b. Turn started -> idle at the first read, working at the measurement:
# exit 3, nothing sent.
# 11c. Turn at send -> idle at both earlier reads, working at the last read
# before the send: exit 3, nothing sent (#585).
# 11d. Read-then-send -> that last `agent get` is the herdr call right
# before `agent prompt`.
# 11e. Last read fails -> exit 2, nothing sent.
# 12. Measure fails -> a failed `foreman marker-fit` is exit 2, nothing sent.
# 13. No foreman.sh -> exit 1 before any herdr call.
#
# The width gate runs the real sibling foreman.sh against the fake herdr, with
# empty XDG homes so the host's own foreman home never decides an outcome.
# Run: bash skills/herdr-standup/tests/test_standup_ask.sh
set -uo pipefail
die() { echo "fatal: $*" >&2; exit 2; }
cleanup() { [[ -n "${TMP:-}" ]] && ! rm -rf "$TMP" && echo "warn: could not remove $TMP" >&2; return 0; }
pass() { PASS=$((PASS+1)); }
fail() { FAIL=$((FAIL+1)); echo " ✗ FAIL: $1" >&2; }
mk_fake_herdr() { # <path>
cat > "$1" <<'FAKE' || die "could not write the fake herdr"
#!/usr/bin/env bash
set -euo pipefail
if [[ -n "${FAKE_ARGV_FILE:-}" ]]; then printf '%s\n' "$*" >> "$FAKE_ARGV_FILE"; fi
case "${1:-} ${2:-}" in
"agent get")
if [[ -n "${FAKE_GET_ERR:-}" ]]; then printf '{"error":{"code":"agent_not_found"}}\n' >&2; exit 1; fi
if [[ -n "${FAKE_GET_BAD:-}" ]]; then printf '{"id":"cli:agent:get","result":{}}\n'; exit 0; fi
status="${FAKE_STATUS:-idle}"
# FAKE_STATUS_LATER answers every read after the first: a worker that
# starts a turn between the readiness read and the measurement.
if [[ -n "${FAKE_STATUS_LATER:-}" ]]; then
if [[ -e "$FAKE_GET_SEEN" ]]; then status="$FAKE_STATUS_LATER"; fi
: > "$FAKE_GET_SEEN"
fi
# FAKE_STATUS_SEQ answers the Nth read with its Nth word; the last word
# repeats. FAKE_GET_COUNT is the file that counts the reads.
if [[ -n "${FAKE_STATUS_SEQ:-}" ]]; then
read -r -a seq <<<"$FAKE_STATUS_SEQ"
printf 'x\n' >> "$FAKE_GET_COUNT"
n="$(wc -l < "$FAKE_GET_COUNT")"
n=$(( n > ${#seq[@]} ? ${#seq[@]} : n ))
status="${seq[$((n - 1))]}"
# The word ERR makes that read a herdr failure.
if [[ "$status" == "ERR" ]]; then printf '{"error":{"code":"agent_not_found"}}\n' >&2; exit 1; fi
fi
printf '{"id":"cli:agent:get","result":{"type":"agent_info","agent":{"agent":"claude","agent_status":"%s","pane_id":"w2:p1","name":"%s"}}}\n' \
"$status" "${3:-worker}"
exit 0
;;
"pane layout")
if [[ -n "${FAKE_LAYOUT_ERR:-}" ]]; then printf '{"error":{"code":"pane_not_found"}}\n' >&2; exit 1; fi
printf '{"result":{"type":"pane_layout","layout":{"panes":[{"pane_id":"w2:p1","rect":{"height":48,"width":%s,"x":0,"y":0}}]}}}\n' \
"${FAKE_WIDTH:-400}"
exit 0
;;
"agent prompt")
if [[ -n "${FAKE_PROMPT_ERR:-}" ]]; then printf '{"error":{"code":"agent_blocked"}}\n' >&2; exit 1; fi
printf '{"id":"cli:agent:prompt","result":{"type":"agent_prompt"}}\n'
exit 0
;;
esac
printf '{"error":{"code":"unsupported"}}\n' >&2
exit 2
FAKE
chmod +x "$1" || die "could not chmod the fake herdr"
}
run() { # [env...] -- runs standup-ask worker <report>
RUN_SEQ=$((RUN_SEQ+1))
ARGV="$TMP/argv.$RUN_SEQ"
: > "$ARGV" || die "could not create $ARGV"
OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" FAKE_ARGV_FILE="$ARGV" \
XDG_STATE_HOME="$TMP/xdg/state" XDG_CONFIG_HOME="$TMP/xdg/config" "$@" \
bash "${RUN_SCRIPT:-$SCRIPT}" worker "$REPORT" 2>"$TMP/err.$RUN_SEQ")"
RC=$?
ERRTEXT="$(cat "$TMP/err.$RUN_SEQ")"
ARGVTEXT="$(cat "$ARGV")"
}
main() {
SCRIPT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)/standup-ask.sh"
[[ -f "$SCRIPT" && -r "$SCRIPT" ]] || die "standup-ask.sh not found at $SCRIPT"
command -v jq >/dev/null 2>&1 || die "jq required for these tests"
TMP="$(mktemp -d -t standup-ask-test.XXXXXX)" || die "mktemp failed"
trap cleanup EXIT
FAKE="$TMP/herdr"; mk_fake_herdr "$FAKE"
REPORT="$TMP/reports/worker.md"
# The temp dir alone is near the production path limit; the limit has its
# own case (6b) and every other case runs under a limit it cannot hit.
export STANDUP_REPORT_PATH_MAX_COLS=1000
FAIL=0; PASS=0; RUN_SEQ=0
# 1. An idle worker is asked.
run FAKE_STATUS=idle
if [[ $RC -eq 0 ]] && printf '%s' "$OUT" | jq -e '.sent == true and .agent == "worker" and .state == "idle"' >/dev/null 2>&1; then
pass; else fail "idle: expected sent true, got RC=$RC OUT=$OUT ERR=$ERRTEXT"; fi
# 2. `done` is the same idle state after unseen work — also ready.
run FAKE_STATUS=done
if [[ $RC -eq 0 ]] && printf '%s' "$OUT" | jq -e '.sent == true' >/dev/null 2>&1; then
pass; else fail "done: expected sent true, got RC=$RC OUT=$OUT"; fi
# 3. A standup is worth less than somebody's turn.
run FAKE_STATUS=working
if [[ $RC -eq 3 ]] && printf '%s' "$OUT" | jq -e '.sent == false and .state == "working"' >/dev/null 2>&1 \
&& ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt"; then
pass; else fail "working: expected exit 3 with nothing sent, got RC=$RC ARGV=$ARGVTEXT"; fi
# 4. Same for a worker at a dialog.
run FAKE_STATUS=blocked
if [[ $RC -eq 3 ]] && ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt"; then
pass; else fail "blocked: expected exit 3 with nothing sent, got RC=$RC ARGV=$ARGVTEXT"; fi
# 5. The question goes as a MESSAGE, and carries the contract it asks for.
run FAKE_STATUS=idle
if printf '%s' "$ARGVTEXT" | grep -q "agent prompt worker"; then
pass; else fail "delivery: expected an agent prompt call, got ARGV=$ARGVTEXT"; fi
if ! printf '%s' "$ARGVTEXT" | grep -qE 'agent prompt worker /|pane send-text'; then
pass; else fail "delivery: a standup question must not go as a slash command"; fi
local field ok=1
for field in "DONE:" "PLAN:" "BLOCKED:" "REPORT:"; do
printf '%s' "$ARGVTEXT" | grep -q "$field" || ok=0
done
if [[ $ok -eq 1 ]] && printf '%s' "$ARGVTEXT" | grep -q "$REPORT"; then
pass; else fail "prompt text: expected the four fields and the report path, got ARGV=$ARGVTEXT"; fi
# 6. A relative report path resolves in the WORKER's cwd, not the foreman's.
RUN_SEQ=$((RUN_SEQ+1))
OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" bash "$SCRIPT" worker "reports/w.md" 2>"$TMP/e6")"; RC=$?
if [[ $RC -eq 1 && -z "$OUT" ]] && grep -q "relative" "$TMP/e6"; then
pass; else fail "relative path: expected exit 1, got RC=$RC"; fi
# 6b. A report path over the coarse length bound is refused before Herdr.
RUN_SEQ=$((RUN_SEQ+1))
OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" STANDUP_REPORT_PATH_MAX_COLS=100 bash "$SCRIPT" worker "/very/long/reports/directory/that/keeps/going/and/going/round-3/reports/standup-answer-from-worker.md" 2>"$TMP/e6b")"; RC=$?
if [[ $RC -eq 1 && -z "$OUT" ]] && grep -q "limit" "$TMP/e6b"; then
pass; else fail "long path: expected exit 1 naming the limit, got RC=$RC OUT=$OUT"; fi
# 6c. A bad limit override is a precondition failure, never an arithmetic abort.
RUN_SEQ=$((RUN_SEQ+1))
OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" STANDUP_REPORT_PATH_MAX_COLS=soon bash "$SCRIPT" worker "$REPORT" 2>"$TMP/e6c")"; RC=$?
if [[ $RC -eq 1 && -z "$OUT" ]] && grep -q "STANDUP_REPORT_PATH_MAX_COLS must be a positive integer" "$TMP/e6c"; then
pass; else fail "bad limit override: expected exit 1 naming it, got RC=$RC OUT=$OUT"; fi
# 6d. A leading-zero override is decimal downstream: `0100` rejects the long
# path like `100` does, with no octal reparse.
RUN_SEQ=$((RUN_SEQ+1))
OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" STANDUP_REPORT_PATH_MAX_COLS=0100 bash "$SCRIPT" worker "/very/long/reports/directory/that/keeps/going/and/going/round-3/reports/standup-answer-from-worker.md" 2>"$TMP/e6d")"; RC=$?
if [[ $RC -eq 1 && -z "$OUT" ]] && grep -q "limit is 100" "$TMP/e6d" && ! grep -q "value too great" "$TMP/e6d"; then
pass; else fail "leading-zero limit: expected exit 1 naming limit 100, got RC=$RC OUT=$OUT ERR=$(cat "$TMP/e6d")"; fi
# 6e. A control character in the path is a precondition, never a
# measurement failure.
RUN_SEQ=$((RUN_SEQ+1))
OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" FAKE_ARGV_FILE="$TMP/argv6e" bash "$SCRIPT" worker "$TMP/reports/a"$'\t'"b.md" 2>"$TMP/e6e")"; RC=$?
if [[ $RC -eq 1 && -z "$OUT" && ! -s "$TMP/argv6e" ]] && grep -q "control character" "$TMP/e6e"; then
pass; else fail "control character: expected exit 1 before herdr, got RC=$RC OUT=$OUT ERR=$(cat "$TMP/e6e")"; fi
# 6f. DEL and UTF-8 C1 controls are the same precondition, whatever the locale.
local ctl
for ctl in $'\x7f' $'\xc2\x85' $'\xc2\x9b' $'\xe2\x80\xa8' $'\xe2\x80\xa9'; do
RUN_SEQ=$((RUN_SEQ+1))
: > "$TMP/argv6f" || die "could not create $TMP/argv6f"
OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" FAKE_ARGV_FILE="$TMP/argv6f" LC_ALL=en_US.UTF-8 \
bash "$SCRIPT" worker "$TMP/reports/a${ctl}b.md" 2>"$TMP/e6f")"; RC=$?
if [[ $RC -eq 1 && -z "$OUT" && ! -s "$TMP/argv6f" ]] && grep -q "control character" "$TMP/e6f"; then
pass; else fail "control $(printf '%s' "$ctl" | od -An -tx1): expected exit 1 before herdr, got RC=$RC ERR=$(cat "$TMP/e6f")"; fi
done
# 6g. A non-ASCII letter (UTF-8 lead byte 0xC3) is not a control character.
RUN_SEQ=$((RUN_SEQ+1))
OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" XDG_STATE_HOME="$TMP/xdg/state" XDG_CONFIG_HOME="$TMP/xdg/config" \
bash "$SCRIPT" worker "$TMP/reports/caf"$'\xc3\xa9'"-"$'\xc3\x85'".md" 2>"$TMP/e6g")"; RC=$?
if [[ $RC -eq 0 ]] && ! grep -q "control character" "$TMP/e6g"; then
pass; else fail "non-ASCII path: expected it to be asked, got RC=$RC ERR=$(cat "$TMP/e6g")"; fi
# 7. Outside Herdr.
OUT="$(env -u HERDR_ENV HERDR_BIN="$FAKE" bash "$SCRIPT" worker "$REPORT" 2>"$TMP/e7")"; RC=$?
if [[ $RC -eq 1 && -z "$OUT" ]] && grep -q "Herdr" "$TMP/e7"; then
pass; else fail "outside Herdr: expected exit 1, got RC=$RC"; fi
# 8. A herdr failure is never a verdict about the worker.
run FAKE_GET_ERR=1
if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q "agent_not_found"; then
pass; else fail "herdr failure: expected exit 2, got RC=$RC OUT=$OUT"; fi
# 9. An unreadable payload is a tool failure too.
run FAKE_GET_BAD=1
if [[ $RC -eq 2 && -z "$OUT" ]]; then
pass; else fail "bad payload: expected exit 2, got RC=$RC OUT=$OUT"; fi
# 9b. A refused prompt (the agent went blocked between the read and the send)
# is surfaced, never reported as sent.
run FAKE_STATUS=idle FAKE_PROMPT_ERR=1
if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q "agent_blocked"; then
pass; else fail "prompt refused: expected exit 2, got RC=$RC OUT=$OUT ERR=$ERRTEXT"; fi
# 10. A pane too narrow for the marker is refused before anything is sent.
run FAKE_STATUS=idle FAKE_WIDTH=40
if [[ $RC -eq 4 ]] && printf '%s' "$OUT" | jq -e '.sent == false and .pane_width == 40 and .needed > 40' >/dev/null 2>&1 \
&& ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt" \
&& printf '%s' "$ERRTEXT" | grep -q "40 columns"; then
pass; else fail "narrow pane: expected exit 4 with nothing sent, got RC=$RC OUT=$OUT ARGV=$ARGVTEXT ERR=$ERRTEXT"; fi
# 11. The owner's own need is enough: a pane that wide is asked, one column
# narrower is not. Its setup measures the need itself through the
# sibling foreman.sh, independent of check 10.
local fit needed
fit="$(env XDG_STATE_HOME="$TMP/xdg/state" XDG_CONFIG_HOME="$TMP/xdg/config" FAKE_WIDTH=1 \
bash "$(dirname "$SCRIPT")/../herdr-foreman/foreman.sh" marker-fit --herdr-bin "$FAKE" \
--agent worker --report "$REPORT")" || die "foreman marker-fit setup failed"
needed="$(printf '%s' "$fit" | jq -er '.needed')" || die "could not read the needed columns from: $fit"
run FAKE_STATUS=idle FAKE_WIDTH="$needed"
if [[ $RC -eq 0 ]] && printf '%s' "$OUT" | jq -e '.sent == true' >/dev/null 2>&1 \
&& printf '%s' "$ARGVTEXT" | grep -q "agent prompt worker"; then
pass; else fail "exact fit: expected a ${needed}-column pane to be asked, got RC=$RC OUT=$OUT ERR=$ERRTEXT"; fi
run FAKE_STATUS=idle FAKE_WIDTH="$((needed - 1))"
if [[ $RC -eq 4 ]] && ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt"; then
pass; else fail "one short: expected a $((needed - 1))-column pane to be refused, got RC=$RC OUT=$OUT"; fi
# 11b. A worker that starts a turn while its pane is measured is not asked.
run FAKE_STATUS=idle FAKE_STATUS_LATER=working FAKE_GET_SEEN="$TMP/get-seen.$((RUN_SEQ+1))"
if [[ $RC -eq 3 ]] && printf '%s' "$OUT" | jq -e '.sent == false and .state == "working"' >/dev/null 2>&1 \
&& ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt" \
&& [[ "$(printf '%s\n' "$ARGVTEXT" | grep -c "agent get")" -eq 2 ]]; then
pass; else fail "idle then working: expected exit 3 with nothing sent, got RC=$RC OUT=$OUT ARGV=$ARGVTEXT ERR=$ERRTEXT"; fi
# 11c. A worker idle at the first read and at the measurement but working
# at the last read before the send is not asked (#585).
run FAKE_STATUS_SEQ="idle idle working" FAKE_GET_COUNT="$TMP/get-count.$((RUN_SEQ+1))"
if [[ $RC -eq 3 ]] && printf '%s' "$OUT" | jq -e '.sent == false and .state == "working"' >/dev/null 2>&1 \
&& ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt" \
&& [[ "$(printf '%s\n' "$ARGVTEXT" | grep -c "agent get")" -eq 3 ]]; then
pass; else fail "working at the send: expected exit 3 with nothing sent, got RC=$RC OUT=$OUT ARGV=$ARGVTEXT ERR=$ERRTEXT"; fi
# 11d. The last read sits immediately before the send: the prompt call is
# the very next herdr call after that third `agent get`.
# The prompt text spans lines, so only lines naming a herdr call count.
local calls
run FAKE_STATUS_SEQ="idle" FAKE_GET_COUNT="$TMP/get-count.$((RUN_SEQ+1))"
calls="$(printf '%s\n' "$ARGVTEXT" | grep -E '^(agent|pane) ')"
if [[ $RC -eq 0 ]] && [[ "$(printf '%s\n' "$calls" | tail -n 2 | head -n 1)" == "agent get worker" ]] \
&& [[ "$(printf '%s\n' "$calls" | grep -c "^agent get")" -eq 3 ]] \
&& printf '%s\n' "$calls" | tail -n 1 | grep -q "^agent prompt worker"; then
pass; else fail "read before send: expected agent get then agent prompt last, got RC=$RC ARGV=$ARGVTEXT"; fi
# 11e. A failed last read fails closed: exit 2, nothing sent.
run FAKE_STATUS_SEQ="idle idle ERR" FAKE_GET_COUNT="$TMP/get-count.$((RUN_SEQ+1))"
if [[ $RC -eq 2 && -z "$OUT" ]] && ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt" \
&& printf '%s' "$ERRTEXT" | grep -q "nothing was sent"; then
pass; else fail "failed last read: expected exit 2 with nothing sent, got RC=$RC OUT=$OUT ARGV=$ARGVTEXT ERR=$ERRTEXT"; fi
# 12. A measurement that fails is a tool failure, never a send.
run FAKE_STATUS=idle FAKE_LAYOUT_ERR=1
if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q "marker-fit" \
&& ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt"; then
pass; else fail "measure failure: expected exit 2 with nothing sent, got RC=$RC OUT=$OUT ERR=$ERRTEXT"; fi
# 13. Installed without its herdr-foreman sibling, nothing is asked.
mkdir -p "$TMP/lonely/herdr-standup" || die "could not create the lonely skill dir"
cp "$SCRIPT" "$TMP/lonely/herdr-standup/standup-ask.sh" || die "could not copy standup-ask.sh"
RUN_SCRIPT="$TMP/lonely/herdr-standup/standup-ask.sh"
run FAKE_STATUS=idle
RUN_SCRIPT=""
if [[ $RC -eq 1 && -z "$OUT" && -z "$ARGVTEXT" ]] && printf '%s' "$ERRTEXT" | grep -q "foreman.sh not found"; then
pass; else fail "missing foreman.sh: expected exit 1 before herdr, got RC=$RC ARGV=$ARGVTEXT ERR=$ERRTEXT"; fi
echo "─────────────────────────────────────────────" >&2
if [[ $FAIL -gt 0 ]]; then echo "FAILED: ${FAIL} failed, ${PASS} passed" >&2; exit 1; fi
echo "PASSED: all ${PASS} checks" >&2
}
if [[ "${BASH_SOURCE[0]}" == "$0" ]]; then
main "$@"
fi.tessl-plugin
hooks
rules
skills
adopt-fork-pr
herdr-foreman
classify
foreman
references
templates
tests
herdr-standup
migrate-to-plugin
onboard-repo
release
references
tests