From 36295beeaeefd579dbc1093e5579872d3c4c934e Mon Sep 17 00:00:00 2001 From: noonghunna Date: Sat, 11 Jul 2026 22:26:30 +0500 Subject: [PATCH] Add rerun-failed-packs.sh: re-test only failing packs + flake verdict (#678) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit benchlocal-cli's finest run granularity is --pack (no per-scenario run filter — benchlocal-cli#82), so after a --full run the "are these failures real?" question cost another 1-2 h full re-run. This script parses a saved RunResult, re-runs ONLY the packs containing failures (through quality-test.sh, keeping its hermes-env/timeout guards), matches the original run's thinking mode from the JSON, passes --previous-result for benchlocal's own per-scenario delta, and prints a consolidated REPRODUCED / FIXED(flake) / NEW-regression verdict. Supports --repeat N passthrough for flakiness rates and RERUN_DRY=1. Guard test included. Claude-Session: https://claude.ai/code/session_01EfF565T9eSLaqGzidyJ1Pm Co-authored-by: noonghunna <10742901+noonghunna@users.noreply.github.com> Co-authored-by: Claude Fable 5 --- scripts/rerun-failed-packs.sh | 141 +++++++++++++++++++++++ scripts/tests/test-rerun-failed-packs.sh | 47 ++++++++ 2 files changed, 188 insertions(+) create mode 100755 scripts/rerun-failed-packs.sh create mode 100755 scripts/tests/test-rerun-failed-packs.sh diff --git a/scripts/rerun-failed-packs.sh b/scripts/rerun-failed-packs.sh new file mode 100755 index 00000000..01637715 --- /dev/null +++ b/scripts/rerun-failed-packs.sh @@ -0,0 +1,141 @@ +#!/usr/bin/env bash +# +# rerun-failed-packs.sh — re-test ONLY the packs that had failures in a prior +# quality run, and report which failures reproduced vs flipped (flakes). +# +# Why: benchlocal-cli's finest run granularity is --pack (no per-scenario run +# filter — see noonghunna/benchlocal-cli#82), and a --full re-run costs 1-2 h. +# After any full run, the variance question is almost always "are these +# failures real?" — answering it only needs the packs that failed. +# +# What it does: +# 1. Parses a saved RunResult JSON: failed scenarios -> the affected packs. +# 2. Re-runs ONLY those packs via quality-test.sh (keeping its hermes-env + +# timeout guards), passing --previous-result so benchlocal emits its own +# per-scenario delta, matching the ORIGINAL run's thinking mode from the +# JSON (thinking_enabled) — no mode flag to get wrong. +# 3. Prints a consolidated verdict per original failure: +# REPRODUCED (real) / FIXED (flake or environment) + any NEW regressions +# in the re-run packs. +# +# Usage: +# bash scripts/rerun-failed-packs.sh [--repeat N] [extra quality-test.sh args...] +# +# URL= / MODEL= env override the endpoint (default: autodetect via +# quality-test.sh, same as any other run). --repeat N re-runs each scenario +# N times (benchlocal aggregates at >=50% pass) — use N>=3 for flakiness +# stats on the failing packs. +# +# RERUN_DRY=1 print the plan (packs + mode + commands) without running. +# +# Output JSONs land next to the original as .rerun..json. +set -euo pipefail + +SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)" +ROOT_DIR="$(cd -- "${SCRIPT_DIR}/.." && pwd)" + +RESULT_JSON="${1:-}" +if [[ -z "$RESULT_JSON" || "$RESULT_JSON" == "-h" || "$RESULT_JSON" == "--help" ]]; then + sed -n '2,30p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//' + exit 2 +fi +if [[ ! -f "$RESULT_JSON" ]]; then + echo "✗ result JSON not found: $RESULT_JSON" >&2 + echo " Fix: pass a saved RunResult (e.g. results/quality/quality-.json)" >&2 + exit 2 +fi +shift + +# --- 1. plan: failed packs + thinking mode from the original run ------------- +PLAN="$(python3 - "$RESULT_JSON" <<'PY' +import json, sys +d = json.load(open(sys.argv[1], encoding="utf-8")) +failed_packs = [] +failures = [] +for p in d.get("packs", []): + pid = p.get("pack_id", "?") + bad = [s for s in p.get("scenarios", []) if s.get("passed") is not True] + if bad: + failed_packs.append(pid) + for s in bad: + failures.append(f"{pid}/{s.get('id','?')}:{s.get('failure_mode','fail')}") +mode = "--enable-thinking" if d.get("thinking_enabled") else "--no-thinking" +print(mode) +print(" ".join(failed_packs)) +print(" ".join(failures)) +PY +)" +MODE_FLAG="$(sed -n '1p' <<<"$PLAN")" +FAILED_PACKS="$(sed -n '2p' <<<"$PLAN")" +ORIG_FAILURES="$(sed -n '3p' <<<"$PLAN")" + +if [[ -z "$FAILED_PACKS" ]]; then + echo "✓ no failed scenarios in $RESULT_JSON — nothing to re-run." + exit 0 +fi + +echo "[rerun] original run: $RESULT_JSON (mode: ${MODE_FLAG#--})" +echo "[rerun] failed packs: $FAILED_PACKS" +echo "[rerun] original failures: $(wc -w <<<"$ORIG_FAILURES" | tr -d ' ')" + +if [[ "${RERUN_DRY:-0}" == "1" ]]; then + for pack in $FAILED_PACKS; do + echo "[dry] quality-test.sh --pack $pack $MODE_FLAG --previous-result $RESULT_JSON --save-json ${RESULT_JSON}.rerun.${pack}.json $*" + done + exit 0 +fi + +# --- 2. re-run each failed pack via the wrapper ------------------------------- +for pack in $FAILED_PACKS; do + echo + echo "════ re-running pack: $pack ════" + bash "${SCRIPT_DIR}/quality-test.sh" --pack "$pack" "$MODE_FLAG" \ + --previous-result "$RESULT_JSON" \ + --save-json "${RESULT_JSON}.rerun.${pack}.json" \ + "$@" +done + +# --- 3. consolidated verdict --------------------------------------------------- +python3 - "$RESULT_JSON" $FAILED_PACKS <<'PY' +import json, sys +orig_path, packs = sys.argv[1], sys.argv[2:] +orig = json.load(open(orig_path, encoding="utf-8")) +orig_state = {} +for p in orig.get("packs", []): + for s in p.get("scenarios", []): + orig_state[(p["pack_id"], s["id"])] = s.get("passed") is True + +repro, fixed, new_reg, missing = [], [], [], [] +for pack in packs: + try: + rr = json.load(open(f"{orig_path}.rerun.{pack}.json", encoding="utf-8")) + except FileNotFoundError: + missing.append(pack) + continue + for p in rr.get("packs", []): + for s in p.get("scenarios", []): + key = (p["pack_id"], s["id"]) + now_pass = s.get("passed") is True + was_pass = orig_state.get(key) + tag = f"{key[0]}/{key[1]}" + if was_pass is False and not now_pass: + repro.append(f"{tag} ({s.get('failure_mode','fail')})") + elif was_pass is False and now_pass: + fixed.append(tag) + elif was_pass is True and not now_pass: + new_reg.append(f"{tag} ({s.get('failure_mode','fail')})") + +print("\n════ rerun verdict ════") +print(f"REPRODUCED ({len(repro)}) — likely real:") +for t in repro: print(f" ✗ {t}") +print(f"FIXED on re-run ({len(fixed)}) — flake / environment:") +for t in fixed: print(f" ↺ {t}") +if new_reg: + print(f"NEW regressions ({len(new_reg)}) — variance cuts both ways:") + for t in new_reg: print(f" ⚠ {t}") +if missing: + print(f"(no rerun JSON for: {', '.join(missing)})") +print("\nNote: single re-run separates 'stable' from 'flaky', not 'model' from") +print("'harness'. For flakiness RATES, add --repeat 3 and read benchlocal's") +print(">=50% aggregation in the per-pack delta output.") +PY diff --git a/scripts/tests/test-rerun-failed-packs.sh b/scripts/tests/test-rerun-failed-packs.sh new file mode 100755 index 00000000..83d28055 --- /dev/null +++ b/scripts/tests/test-rerun-failed-packs.sh @@ -0,0 +1,47 @@ +#!/usr/bin/env bash +# Guard for scripts/rerun-failed-packs.sh: syntax, arg refusal, dry-run plan +# (pack extraction + thinking-mode derivation from a synthetic RunResult). +set -euo pipefail +ROOT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../.." && pwd)" +S="${ROOT_DIR}/scripts/rerun-failed-packs.sh" + +fail() { echo "ASSERTION FAILED: $1" >&2; exit 1; } + +# 1. syntax +bash -n "$S" || fail "bash -n syntax check" + +# 2. refuses missing/absent JSON with exit 2 +set +e +bash "$S" >/dev/null 2>&1; [[ $? -eq 2 ]] || fail "no-arg should exit 2" +bash "$S" /nonexistent.json >/dev/null 2>&1; [[ $? -eq 2 ]] || fail "missing file should exit 2" +set -e + +# 3. dry-run plan on a synthetic result: only the failing pack, correct mode +TMP="$(mktemp -d)"; trap 'rm -rf "$TMP"' EXIT +cat > "$TMP/r.json" <<'JSON' +{ + "thinking_enabled": true, + "packs": [ + {"pack_id": "toolcall-15", "scenarios": [ + {"id": "TC-01", "passed": true, "failure_mode": "passed"}]}, + {"pack_id": "reasonmath-15", "scenarios": [ + {"id": "RM-04", "passed": false, "failure_mode": "wrong_answer"}, + {"id": "RM-05", "passed": true, "failure_mode": "passed"}]} + ] +} +JSON +out="$(RERUN_DRY=1 bash "$S" "$TMP/r.json" 2>&1)" +echo "$out" | command grep -q "failed packs: reasonmath-15$" || fail "plan should list only reasonmath-15 (got: $out)" +echo "$out" | command grep -q -- "--enable-thinking" || fail "plan should derive --enable-thinking from thinking_enabled=true" +echo "$out" | command grep -q -- "--previous-result $TMP/r.json" || fail "plan should pass --previous-result" +echo "$out" | command grep -qv "toolcall-15" || true +echo "$out" | command grep -q "\[dry\] quality-test.sh --pack toolcall-15" && fail "clean pack must not be re-run" + +# 4. all-clean result → exit 0, nothing to re-run +cat > "$TMP/clean.json" <<'JSON' +{"thinking_enabled": false, "packs": [{"pack_id": "toolcall-15", "scenarios": [{"id": "TC-01", "passed": true}]}]} +JSON +out="$(bash "$S" "$TMP/clean.json" 2>&1)" +echo "$out" | command grep -q "nothing to re-run" || fail "clean result should short-circuit" + +echo "test-rerun-failed-packs: PASS"