Files
club-3090/scripts/tests/test-rerun-failed-packs.sh
noonghunna 36295beeae Add rerun-failed-packs.sh: re-test only failing packs + flake verdict (#678)
benchlocal-cli's finest run granularity is --pack (no per-scenario run
filter — benchlocal-cli#82), so after a --full run the "are these
failures real?" question cost another 1-2 h full re-run. This script
parses a saved RunResult, re-runs ONLY the packs containing failures
(through quality-test.sh, keeping its hermes-env/timeout guards),
matches the original run's thinking mode from the JSON, passes
--previous-result for benchlocal's own per-scenario delta, and prints a
consolidated REPRODUCED / FIXED(flake) / NEW-regression verdict.
Supports --repeat N passthrough for flakiness rates and RERUN_DRY=1.
Guard test included.


Claude-Session: https://claude.ai/code/session_01EfF565T9eSLaqGzidyJ1Pm

Co-authored-by: noonghunna <10742901+noonghunna@users.noreply.github.com>
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
2026-07-11 22:26:30 +05:00

48 lines
2.1 KiB
Bash
Executable File

#!/usr/bin/env bash
# Guard for scripts/rerun-failed-packs.sh: syntax, arg refusal, dry-run plan
# (pack extraction + thinking-mode derivation from a synthetic RunResult).
set -euo pipefail
ROOT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../.." && pwd)"
S="${ROOT_DIR}/scripts/rerun-failed-packs.sh"
fail() { echo "ASSERTION FAILED: $1" >&2; exit 1; }
# 1. syntax
bash -n "$S" || fail "bash -n syntax check"
# 2. refuses missing/absent JSON with exit 2
set +e
bash "$S" >/dev/null 2>&1; [[ $? -eq 2 ]] || fail "no-arg should exit 2"
bash "$S" /nonexistent.json >/dev/null 2>&1; [[ $? -eq 2 ]] || fail "missing file should exit 2"
set -e
# 3. dry-run plan on a synthetic result: only the failing pack, correct mode
TMP="$(mktemp -d)"; trap 'rm -rf "$TMP"' EXIT
cat > "$TMP/r.json" <<'JSON'
{
"thinking_enabled": true,
"packs": [
{"pack_id": "toolcall-15", "scenarios": [
{"id": "TC-01", "passed": true, "failure_mode": "passed"}]},
{"pack_id": "reasonmath-15", "scenarios": [
{"id": "RM-04", "passed": false, "failure_mode": "wrong_answer"},
{"id": "RM-05", "passed": true, "failure_mode": "passed"}]}
]
}
JSON
out="$(RERUN_DRY=1 bash "$S" "$TMP/r.json" 2>&1)"
echo "$out" | command grep -q "failed packs: reasonmath-15$" || fail "plan should list only reasonmath-15 (got: $out)"
echo "$out" | command grep -q -- "--enable-thinking" || fail "plan should derive --enable-thinking from thinking_enabled=true"
echo "$out" | command grep -q -- "--previous-result $TMP/r.json" || fail "plan should pass --previous-result"
echo "$out" | command grep -qv "toolcall-15" || true
echo "$out" | command grep -q "\[dry\] quality-test.sh --pack toolcall-15" && fail "clean pack must not be re-run"
# 4. all-clean result → exit 0, nothing to re-run
cat > "$TMP/clean.json" <<'JSON'
{"thinking_enabled": false, "packs": [{"pack_id": "toolcall-15", "scenarios": [{"id": "TC-01", "passed": true}]}]}
JSON
out="$(bash "$S" "$TMP/clean.json" 2>&1)"
echo "$out" | command grep -q "nothing to re-run" || fail "clean result should short-circuit"
echo "test-rerun-failed-packs: PASS"