CtrlK
BlogDocsLog inGet started
Tessl Logo

jbaruch/coding-policy

General-purpose coding policy for Baruch's AI agents

73

Quality

91%

Does it follow best practices?

Run evals on this skill

Adds up to 20 points to the overall score

View guide

SecuritybySnyk

Low

Low-risk findings worth noting

Overview
Quality
Evals
Security
Files

test_standup_ask.shskills/herdr-standup/tests/

#!/usr/bin/env bash
# Outcome-based tests for skills/herdr-standup/standup-ask.sh.
#
# Every case points HERDR_BIN at a fake this harness writes, which records the
# argv it was handed (rules/testing-standards.md — no live Herdr session).
#
# The harness drops `set -e` to aggregate results, so every fixture-setup
# command is checked explicitly and aborts with a fatal diagnostic on failure
# (rules/error-handling.md aggregate-reporting carve-out).
#
# Covers:
#   1. Idle worker    -> prompted, sent true.
#   2. Done worker    -> prompted too; done is ready, not busy.
#   3. Working worker -> exit 3, NOTHING sent (a standup never interrupts).
#   4. Blocked worker -> exit 3, nothing sent.
#   5. Message shape  -> `agent prompt`, never a slash command, and the four
#                        field names plus the report path are in the text.
#   6. Relative path  -> exit 1 before any herdr call.
#  6b. Over-long path -> exit 1; the coarse length bound, before Herdr.
#  6e. Control char   -> exit 1 before Herdr.
#  6f. DEL, C1, U+2028/9 -> exit 1 before Herdr, in a UTF-8 locale too.
#  6g. Non-ASCII path -> a letter such as U+00E9 is asked normally.
#  6c. Bad limit      -> a non-integer override is exit 1, not an abort.
#  6d. `0100` is 100  -> a leading zero is decimal, never octal, downstream.
#   7. Outside Herdr  -> exit 1.
#   8. herdr failure  -> exit 2, no verdict.
#   9. Bad payload    -> exit 2.
#  10. Narrow pane    -> exit 4, nothing sent, the verdict names the width
#                        and the columns the marker needs (#515).
#  11. Exact fit      -> a pane exactly as wide as that need is asked.
#  11b. Turn started  -> idle at the first read, working at the measurement:
#                        exit 3, nothing sent.
#  11c. Turn at send  -> idle at both earlier reads, working at the last read
#                        before the send: exit 3, nothing sent (#585).
#  11d. Read-then-send -> that last `agent get` is the herdr call right
#                        before `agent prompt`.
#  11e. Last read fails -> exit 2, nothing sent.
#  12. Measure fails  -> a failed `foreman marker-fit` is exit 2, nothing sent.
#  13. No foreman.sh  -> exit 1 before any herdr call.
#
# The width gate runs the real sibling foreman.sh against the fake herdr, with
# empty XDG homes so the host's own foreman home never decides an outcome.
# Run: bash skills/herdr-standup/tests/test_standup_ask.sh
set -uo pipefail

die() { echo "fatal: $*" >&2; exit 2; }
cleanup() { [[ -n "${TMP:-}" ]] && ! rm -rf "$TMP" && echo "warn: could not remove $TMP" >&2; return 0; }
pass() { PASS=$((PASS+1)); }
fail() { FAIL=$((FAIL+1)); echo "  ✗ FAIL: $1" >&2; }

mk_fake_herdr() { # <path>
  cat > "$1" <<'FAKE' || die "could not write the fake herdr"
#!/usr/bin/env bash
set -euo pipefail
if [[ -n "${FAKE_ARGV_FILE:-}" ]]; then printf '%s\n' "$*" >> "$FAKE_ARGV_FILE"; fi
case "${1:-} ${2:-}" in
  "agent get")
    if [[ -n "${FAKE_GET_ERR:-}" ]]; then printf '{"error":{"code":"agent_not_found"}}\n' >&2; exit 1; fi
    if [[ -n "${FAKE_GET_BAD:-}" ]]; then printf '{"id":"cli:agent:get","result":{}}\n'; exit 0; fi
    status="${FAKE_STATUS:-idle}"
    # FAKE_STATUS_LATER answers every read after the first: a worker that
    # starts a turn between the readiness read and the measurement.
    if [[ -n "${FAKE_STATUS_LATER:-}" ]]; then
      if [[ -e "$FAKE_GET_SEEN" ]]; then status="$FAKE_STATUS_LATER"; fi
      : > "$FAKE_GET_SEEN"
    fi
    # FAKE_STATUS_SEQ answers the Nth read with its Nth word; the last word
    # repeats. FAKE_GET_COUNT is the file that counts the reads.
    if [[ -n "${FAKE_STATUS_SEQ:-}" ]]; then
      read -r -a seq <<<"$FAKE_STATUS_SEQ"
      printf 'x\n' >> "$FAKE_GET_COUNT"
      n="$(wc -l < "$FAKE_GET_COUNT")"
      n=$(( n > ${#seq[@]} ? ${#seq[@]} : n ))
      status="${seq[$((n - 1))]}"
      # The word ERR makes that read a herdr failure.
      if [[ "$status" == "ERR" ]]; then printf '{"error":{"code":"agent_not_found"}}\n' >&2; exit 1; fi
    fi
    printf '{"id":"cli:agent:get","result":{"type":"agent_info","agent":{"agent":"claude","agent_status":"%s","pane_id":"w2:p1","name":"%s"}}}\n' \
      "$status" "${3:-worker}"
    exit 0
    ;;
  "pane layout")
    if [[ -n "${FAKE_LAYOUT_ERR:-}" ]]; then printf '{"error":{"code":"pane_not_found"}}\n' >&2; exit 1; fi
    printf '{"result":{"type":"pane_layout","layout":{"panes":[{"pane_id":"w2:p1","rect":{"height":48,"width":%s,"x":0,"y":0}}]}}}\n' \
      "${FAKE_WIDTH:-400}"
    exit 0
    ;;
  "agent prompt")
    if [[ -n "${FAKE_PROMPT_ERR:-}" ]]; then printf '{"error":{"code":"agent_blocked"}}\n' >&2; exit 1; fi
    printf '{"id":"cli:agent:prompt","result":{"type":"agent_prompt"}}\n'
    exit 0
    ;;
esac
printf '{"error":{"code":"unsupported"}}\n' >&2
exit 2
FAKE
  chmod +x "$1" || die "could not chmod the fake herdr"
}

run() { # [env...] -- runs standup-ask worker <report>
  RUN_SEQ=$((RUN_SEQ+1))
  ARGV="$TMP/argv.$RUN_SEQ"
  : > "$ARGV" || die "could not create $ARGV"
  OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" FAKE_ARGV_FILE="$ARGV" \
    XDG_STATE_HOME="$TMP/xdg/state" XDG_CONFIG_HOME="$TMP/xdg/config" "$@" \
    bash "${RUN_SCRIPT:-$SCRIPT}" worker "$REPORT" 2>"$TMP/err.$RUN_SEQ")"
  RC=$?
  ERRTEXT="$(cat "$TMP/err.$RUN_SEQ")"
  ARGVTEXT="$(cat "$ARGV")"
}

main() {
  SCRIPT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)/standup-ask.sh"
  [[ -f "$SCRIPT" && -r "$SCRIPT" ]] || die "standup-ask.sh not found at $SCRIPT"
  command -v jq >/dev/null 2>&1 || die "jq required for these tests"
  TMP="$(mktemp -d -t standup-ask-test.XXXXXX)" || die "mktemp failed"
  trap cleanup EXIT
  FAKE="$TMP/herdr"; mk_fake_herdr "$FAKE"
  REPORT="$TMP/reports/worker.md"
  # The temp dir alone is near the production path limit; the limit has its
  # own case (6b) and every other case runs under a limit it cannot hit.
  export STANDUP_REPORT_PATH_MAX_COLS=1000
  FAIL=0; PASS=0; RUN_SEQ=0

  # 1. An idle worker is asked.
  run FAKE_STATUS=idle
  if [[ $RC -eq 0 ]] && printf '%s' "$OUT" | jq -e '.sent == true and .agent == "worker" and .state == "idle"' >/dev/null 2>&1; then
    pass; else fail "idle: expected sent true, got RC=$RC OUT=$OUT ERR=$ERRTEXT"; fi

  # 2. `done` is the same idle state after unseen work — also ready.
  run FAKE_STATUS=done
  if [[ $RC -eq 0 ]] && printf '%s' "$OUT" | jq -e '.sent == true' >/dev/null 2>&1; then
    pass; else fail "done: expected sent true, got RC=$RC OUT=$OUT"; fi

  # 3. A standup is worth less than somebody's turn.
  run FAKE_STATUS=working
  if [[ $RC -eq 3 ]] && printf '%s' "$OUT" | jq -e '.sent == false and .state == "working"' >/dev/null 2>&1 \
     && ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt"; then
    pass; else fail "working: expected exit 3 with nothing sent, got RC=$RC ARGV=$ARGVTEXT"; fi

  # 4. Same for a worker at a dialog.
  run FAKE_STATUS=blocked
  if [[ $RC -eq 3 ]] && ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt"; then
    pass; else fail "blocked: expected exit 3 with nothing sent, got RC=$RC ARGV=$ARGVTEXT"; fi

  # 5. The question goes as a MESSAGE, and carries the contract it asks for.
  run FAKE_STATUS=idle
  if printf '%s' "$ARGVTEXT" | grep -q "agent prompt worker"; then
    pass; else fail "delivery: expected an agent prompt call, got ARGV=$ARGVTEXT"; fi
  if ! printf '%s' "$ARGVTEXT" | grep -qE 'agent prompt worker /|pane send-text'; then
    pass; else fail "delivery: a standup question must not go as a slash command"; fi
  local field ok=1
  for field in "DONE:" "PLAN:" "BLOCKED:" "REPORT:"; do
    printf '%s' "$ARGVTEXT" | grep -q "$field" || ok=0
  done
  if [[ $ok -eq 1 ]] && printf '%s' "$ARGVTEXT" | grep -q "$REPORT"; then
    pass; else fail "prompt text: expected the four fields and the report path, got ARGV=$ARGVTEXT"; fi

  # 6. A relative report path resolves in the WORKER's cwd, not the foreman's.
  RUN_SEQ=$((RUN_SEQ+1))
  OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" bash "$SCRIPT" worker "reports/w.md" 2>"$TMP/e6")"; RC=$?
  if [[ $RC -eq 1 && -z "$OUT" ]] && grep -q "relative" "$TMP/e6"; then
    pass; else fail "relative path: expected exit 1, got RC=$RC"; fi

  # 6b. A report path over the coarse length bound is refused before Herdr.
  RUN_SEQ=$((RUN_SEQ+1))
  OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" STANDUP_REPORT_PATH_MAX_COLS=100 bash "$SCRIPT" worker "/very/long/reports/directory/that/keeps/going/and/going/round-3/reports/standup-answer-from-worker.md" 2>"$TMP/e6b")"; RC=$?
  if [[ $RC -eq 1 && -z "$OUT" ]] && grep -q "limit" "$TMP/e6b"; then
    pass; else fail "long path: expected exit 1 naming the limit, got RC=$RC OUT=$OUT"; fi

  # 6c. A bad limit override is a precondition failure, never an arithmetic abort.
  RUN_SEQ=$((RUN_SEQ+1))
  OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" STANDUP_REPORT_PATH_MAX_COLS=soon bash "$SCRIPT" worker "$REPORT" 2>"$TMP/e6c")"; RC=$?
  if [[ $RC -eq 1 && -z "$OUT" ]] && grep -q "STANDUP_REPORT_PATH_MAX_COLS must be a positive integer" "$TMP/e6c"; then
    pass; else fail "bad limit override: expected exit 1 naming it, got RC=$RC OUT=$OUT"; fi

  # 6d. A leading-zero override is decimal downstream: `0100` rejects the long
  #     path like `100` does, with no octal reparse.
  RUN_SEQ=$((RUN_SEQ+1))
  OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" STANDUP_REPORT_PATH_MAX_COLS=0100 bash "$SCRIPT" worker "/very/long/reports/directory/that/keeps/going/and/going/round-3/reports/standup-answer-from-worker.md" 2>"$TMP/e6d")"; RC=$?
  if [[ $RC -eq 1 && -z "$OUT" ]] && grep -q "limit is 100" "$TMP/e6d" && ! grep -q "value too great" "$TMP/e6d"; then
    pass; else fail "leading-zero limit: expected exit 1 naming limit 100, got RC=$RC OUT=$OUT ERR=$(cat "$TMP/e6d")"; fi

  # 6e. A control character in the path is a precondition, never a
  #     measurement failure.
  RUN_SEQ=$((RUN_SEQ+1))
  OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" FAKE_ARGV_FILE="$TMP/argv6e" bash "$SCRIPT" worker "$TMP/reports/a"$'\t'"b.md" 2>"$TMP/e6e")"; RC=$?
  if [[ $RC -eq 1 && -z "$OUT" && ! -s "$TMP/argv6e" ]] && grep -q "control character" "$TMP/e6e"; then
    pass; else fail "control character: expected exit 1 before herdr, got RC=$RC OUT=$OUT ERR=$(cat "$TMP/e6e")"; fi

  # 6f. DEL and UTF-8 C1 controls are the same precondition, whatever the locale.
  local ctl
  for ctl in $'\x7f' $'\xc2\x85' $'\xc2\x9b' $'\xe2\x80\xa8' $'\xe2\x80\xa9'; do
    RUN_SEQ=$((RUN_SEQ+1))
    : > "$TMP/argv6f" || die "could not create $TMP/argv6f"
    OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" FAKE_ARGV_FILE="$TMP/argv6f" LC_ALL=en_US.UTF-8 \
      bash "$SCRIPT" worker "$TMP/reports/a${ctl}b.md" 2>"$TMP/e6f")"; RC=$?
    if [[ $RC -eq 1 && -z "$OUT" && ! -s "$TMP/argv6f" ]] && grep -q "control character" "$TMP/e6f"; then
      pass; else fail "control $(printf '%s' "$ctl" | od -An -tx1): expected exit 1 before herdr, got RC=$RC ERR=$(cat "$TMP/e6f")"; fi
  done

  # 6g. A non-ASCII letter (UTF-8 lead byte 0xC3) is not a control character.
  RUN_SEQ=$((RUN_SEQ+1))
  OUT="$(env HERDR_ENV=1 HERDR_BIN="$FAKE" XDG_STATE_HOME="$TMP/xdg/state" XDG_CONFIG_HOME="$TMP/xdg/config" \
    bash "$SCRIPT" worker "$TMP/reports/caf"$'\xc3\xa9'"-"$'\xc3\x85'".md" 2>"$TMP/e6g")"; RC=$?
  if [[ $RC -eq 0 ]] && ! grep -q "control character" "$TMP/e6g"; then
    pass; else fail "non-ASCII path: expected it to be asked, got RC=$RC ERR=$(cat "$TMP/e6g")"; fi

  # 7. Outside Herdr.
  OUT="$(env -u HERDR_ENV HERDR_BIN="$FAKE" bash "$SCRIPT" worker "$REPORT" 2>"$TMP/e7")"; RC=$?
  if [[ $RC -eq 1 && -z "$OUT" ]] && grep -q "Herdr" "$TMP/e7"; then
    pass; else fail "outside Herdr: expected exit 1, got RC=$RC"; fi

  # 8. A herdr failure is never a verdict about the worker.
  run FAKE_GET_ERR=1
  if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q "agent_not_found"; then
    pass; else fail "herdr failure: expected exit 2, got RC=$RC OUT=$OUT"; fi

  # 9. An unreadable payload is a tool failure too.
  run FAKE_GET_BAD=1
  if [[ $RC -eq 2 && -z "$OUT" ]]; then
    pass; else fail "bad payload: expected exit 2, got RC=$RC OUT=$OUT"; fi

  # 9b. A refused prompt (the agent went blocked between the read and the send)
  #     is surfaced, never reported as sent.
  run FAKE_STATUS=idle FAKE_PROMPT_ERR=1
  if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q "agent_blocked"; then
    pass; else fail "prompt refused: expected exit 2, got RC=$RC OUT=$OUT ERR=$ERRTEXT"; fi

  # 10. A pane too narrow for the marker is refused before anything is sent.
  run FAKE_STATUS=idle FAKE_WIDTH=40
  if [[ $RC -eq 4 ]] && printf '%s' "$OUT" | jq -e '.sent == false and .pane_width == 40 and .needed > 40' >/dev/null 2>&1 \
     && ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt" \
     && printf '%s' "$ERRTEXT" | grep -q "40 columns"; then
    pass; else fail "narrow pane: expected exit 4 with nothing sent, got RC=$RC OUT=$OUT ARGV=$ARGVTEXT ERR=$ERRTEXT"; fi

  # 11. The owner's own need is enough: a pane that wide is asked, one column
  #     narrower is not. Its setup measures the need itself through the
  #     sibling foreman.sh, independent of check 10.
  local fit needed
  fit="$(env XDG_STATE_HOME="$TMP/xdg/state" XDG_CONFIG_HOME="$TMP/xdg/config" FAKE_WIDTH=1 \
    bash "$(dirname "$SCRIPT")/../herdr-foreman/foreman.sh" marker-fit --herdr-bin "$FAKE" \
    --agent worker --report "$REPORT")" || die "foreman marker-fit setup failed"
  needed="$(printf '%s' "$fit" | jq -er '.needed')" || die "could not read the needed columns from: $fit"
  run FAKE_STATUS=idle FAKE_WIDTH="$needed"
  if [[ $RC -eq 0 ]] && printf '%s' "$OUT" | jq -e '.sent == true' >/dev/null 2>&1 \
     && printf '%s' "$ARGVTEXT" | grep -q "agent prompt worker"; then
    pass; else fail "exact fit: expected a ${needed}-column pane to be asked, got RC=$RC OUT=$OUT ERR=$ERRTEXT"; fi
  run FAKE_STATUS=idle FAKE_WIDTH="$((needed - 1))"
  if [[ $RC -eq 4 ]] && ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt"; then
    pass; else fail "one short: expected a $((needed - 1))-column pane to be refused, got RC=$RC OUT=$OUT"; fi

  # 11b. A worker that starts a turn while its pane is measured is not asked.
  run FAKE_STATUS=idle FAKE_STATUS_LATER=working FAKE_GET_SEEN="$TMP/get-seen.$((RUN_SEQ+1))"
  if [[ $RC -eq 3 ]] && printf '%s' "$OUT" | jq -e '.sent == false and .state == "working"' >/dev/null 2>&1 \
     && ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt" \
     && [[ "$(printf '%s\n' "$ARGVTEXT" | grep -c "agent get")" -eq 2 ]]; then
    pass; else fail "idle then working: expected exit 3 with nothing sent, got RC=$RC OUT=$OUT ARGV=$ARGVTEXT ERR=$ERRTEXT"; fi

  # 11c. A worker idle at the first read and at the measurement but working
  #      at the last read before the send is not asked (#585).
  run FAKE_STATUS_SEQ="idle idle working" FAKE_GET_COUNT="$TMP/get-count.$((RUN_SEQ+1))"
  if [[ $RC -eq 3 ]] && printf '%s' "$OUT" | jq -e '.sent == false and .state == "working"' >/dev/null 2>&1 \
     && ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt" \
     && [[ "$(printf '%s\n' "$ARGVTEXT" | grep -c "agent get")" -eq 3 ]]; then
    pass; else fail "working at the send: expected exit 3 with nothing sent, got RC=$RC OUT=$OUT ARGV=$ARGVTEXT ERR=$ERRTEXT"; fi

  # 11d. The last read sits immediately before the send: the prompt call is
  #      the very next herdr call after that third `agent get`.
  #      The prompt text spans lines, so only lines naming a herdr call count.
  local calls
  run FAKE_STATUS_SEQ="idle" FAKE_GET_COUNT="$TMP/get-count.$((RUN_SEQ+1))"
  calls="$(printf '%s\n' "$ARGVTEXT" | grep -E '^(agent|pane) ')"
  if [[ $RC -eq 0 ]] && [[ "$(printf '%s\n' "$calls" | tail -n 2 | head -n 1)" == "agent get worker" ]] \
     && [[ "$(printf '%s\n' "$calls" | grep -c "^agent get")" -eq 3 ]] \
     && printf '%s\n' "$calls" | tail -n 1 | grep -q "^agent prompt worker"; then
    pass; else fail "read before send: expected agent get then agent prompt last, got RC=$RC ARGV=$ARGVTEXT"; fi

  # 11e. A failed last read fails closed: exit 2, nothing sent.
  run FAKE_STATUS_SEQ="idle idle ERR" FAKE_GET_COUNT="$TMP/get-count.$((RUN_SEQ+1))"
  if [[ $RC -eq 2 && -z "$OUT" ]] && ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt" \
     && printf '%s' "$ERRTEXT" | grep -q "nothing was sent"; then
    pass; else fail "failed last read: expected exit 2 with nothing sent, got RC=$RC OUT=$OUT ARGV=$ARGVTEXT ERR=$ERRTEXT"; fi

  # 12. A measurement that fails is a tool failure, never a send.
  run FAKE_STATUS=idle FAKE_LAYOUT_ERR=1
  if [[ $RC -eq 2 && -z "$OUT" ]] && printf '%s' "$ERRTEXT" | grep -q "marker-fit" \
     && ! printf '%s' "$ARGVTEXT" | grep -q "agent prompt"; then
    pass; else fail "measure failure: expected exit 2 with nothing sent, got RC=$RC OUT=$OUT ERR=$ERRTEXT"; fi

  # 13. Installed without its herdr-foreman sibling, nothing is asked.
  mkdir -p "$TMP/lonely/herdr-standup" || die "could not create the lonely skill dir"
  cp "$SCRIPT" "$TMP/lonely/herdr-standup/standup-ask.sh" || die "could not copy standup-ask.sh"
  RUN_SCRIPT="$TMP/lonely/herdr-standup/standup-ask.sh"
  run FAKE_STATUS=idle
  RUN_SCRIPT=""
  if [[ $RC -eq 1 && -z "$OUT" && -z "$ARGVTEXT" ]] && printf '%s' "$ERRTEXT" | grep -q "foreman.sh not found"; then
    pass; else fail "missing foreman.sh: expected exit 1 before herdr, got RC=$RC ARGV=$ARGVTEXT ERR=$ERRTEXT"; fi

  echo "─────────────────────────────────────────────" >&2
  if [[ $FAIL -gt 0 ]]; then echo "FAILED: ${FAIL} failed, ${PASS} passed" >&2; exit 1; fi
  echo "PASSED: all ${PASS} checks" >&2
}

if [[ "${BASH_SOURCE[0]}" == "$0" ]]; then
  main "$@"
fi

skills

README.md

tile.json