diff --git a/scripts/arch-ab.sh b/scripts/arch-ab.sh index 3af9bf95..c7815bec 100755 --- a/scripts/arch-ab.sh +++ b/scripts/arch-ab.sh @@ -87,6 +87,12 @@ IFS=',' read -ra ARM_LIST <<< "$ARMS" [[ "${#ARM_LIST[@]}" -ge 1 ]] || die "no arms given" for arm in "${ARM_LIST[@]}"; do [[ -n "${ARM_DTYPE[$arm]+x}" ]] || die "unknown arm '${arm}' (valid: e5m2, e4m3, nvfp4, fp8w)" + if [[ "$arm" == "e4m3" ]]; then + # vLLM hard-rejects fp8_e4m3 KV below SM 8.9 at boot — without this guard + # an Ampere rig sits in switch.sh's ready-wait until the 10-min timeout. + awk -v a="$MIN_SM" 'BEGIN{exit !(a>=8.9)}' \ + || die "arm 'e4m3' needs sm>=8.9 (fp8_e4m3 KV is boot-rejected on Ampere); detected min sm_${MIN_SM}. On a 3090-class rig there is no arch delta to measure — this A/B is for Ada/Blackwell rigs (run --arms e5m2 if you just want the control numbers)" + fi if [[ "$arm" == "nvfp4" ]]; then awk -v a="$MIN_SM" 'BEGIN{exit !(a>=10.0)}' \ || die "arm 'nvfp4' needs Blackwell (sm>=10.0); detected min sm_${MIN_SM}. Drop it: --arms e5m2,e4m3" @@ -143,13 +149,25 @@ for arm in "${ARM_LIST[@]}"; do *) die "arm ${arm}: container '${container}' is NOT running --kv-cache-dtype ${assert_dtype} — aborting before benching the wrong config" ;; esac rb_args=(--tag "$tag") + stress_fast=0 if (( FULL )); then rb_args+=(--with-8pack-thinking=both) else rb_args+=(--skip soak) + # A/B tier: STRESS_FAST halves verify-stress's deep-prefill request count + # (large fresh needles skipped + ladder capped at ~3 rungs) while keeping + # the fillable-to proof, ceiling recall, and VRAM margin. + stress_fast=1 fi (( RESUME )) && rb_args+=(--resume) - bash scripts/rebench-full.sh "${rb_args[@]}" + STRESS_FAST="$stress_fast" bash scripts/rebench-full.sh "${rb_args[@]}" + # A margin-advisory rc=1 with a clean ladder scares first-time runners into + # aborting — say explicitly that it is not a failure. + if command grep -q "rungs passed" "results/rebench/${tag}/verify-stress.log" 2>/dev/null \ + && command grep -q "VRAM margin thin at ceiling" "results/rebench/${tag}/verify-stress.log" 2>/dev/null; then + echo "[arch-ab] note: verify-stress exited nonzero on the VRAM-margin ADVISORY only —" + echo "[arch-ab] the NIAH ladder is CLEAN. Expected on tightly-packed configs; carry on." + fi done # --- summary + bundle --------------------------------------------------------- @@ -178,6 +196,10 @@ for arm in arms: (tag / "verify-stress.log").read_text(errors="replace")) if m: ladder = f"{m.group(1)} rungs clean, fillable to {int(m.group(2)):,} tok" + mm = re.search(r"VRAM margin thin at ceiling: (\d+) MB free < (\d+) MB", + (tag / "verify-stress.log").read_text(errors="replace")) + if mm: + ladder += f" (margin advisory: {mm.group(1)} MB < {mm.group(2)} MB — not a failure)" else: ladder = "LADDER DID NOT PASS — check verify-stress.log" except OSError: @@ -202,6 +224,7 @@ tar czf "$bundle" \ $( [[ -s results/rebench/246-ab-rig-report.md ]] && echo results/rebench/246-ab-rig-report.md ) echo "" echo "[arch-ab] bundle written: ${bundle}" -echo "[arch-ab] -> attach that ONE file to https://github.com/noonghunna/club-3090/issues/246" +echo "[arch-ab] -> attach that ONE file to the #246 cross-rig test thread (link pinned on" +echo "[arch-ab] https://github.com/noonghunna/club-3090/issues/246)" echo "[arch-ab] plus a sentence on anything that surprised or annoyed you — the friction" echo "[arch-ab] report is as valuable as the numbers." diff --git a/scripts/tests/test-arch-ab.sh b/scripts/tests/test-arch-ab.sh index dba32f6b..04e06912 100644 --- a/scripts/tests/test-arch-ab.sh +++ b/scripts/tests/test-arch-ab.sh @@ -33,6 +33,15 @@ assert_contains "$out" "arm nvfp4: vllm/dual + KV_CACHE_DTYPE=nvfp4" assert_contains "$out" "arm fp8w: vllm/qwen-27b-dual-max STOCK" # --- refusals (fail-loud, each names the fix) --------------------------------- +# default arms include e4m3 -> a bare run on an Ampere rig must refuse with the +# "no arch delta to measure" explanation, NOT hang in the boot ready-wait +if out="$(CLUB3090_FAKE_GPUS="$GPU_3090" bash scripts/arch-ab.sh --dry-run 2>&1)"; then + fail "default arms on sm_8.6 must refuse (e4m3 boot-rejects on Ampere)" +fi +assert_contains "$out" "no arch delta to measure" +# ...but the explicit control-only run stays possible on Ampere +out="$(CLUB3090_FAKE_GPUS="$GPU_3090" bash scripts/arch-ab.sh --arms e5m2 --dry-run)" +assert_contains "$out" "arm e5m2" if out="$(CLUB3090_FAKE_GPUS="$GPU_3090" bash scripts/arch-ab.sh --arms e5m2,nvfp4 --dry-run 2>&1)"; then fail "nvfp4 on sm_8.6 must refuse" fi diff --git a/scripts/tests/test-verify-stress-ceiling.sh b/scripts/tests/test-verify-stress-ceiling.sh index 92dad717..74144db1 100755 --- a/scripts/tests/test-verify-stress-ceiling.sh +++ b/scripts/tests/test-verify-stress-ceiling.sh @@ -112,7 +112,10 @@ result="$(ENGINE_KIND=llamacpp URL=http://mock CONTAINER=none \ bash -c "source '$HELPERS_FILE'; get_n_ctx")" assert_eq "get_n_ctx 512K compose (nested)" "524288" "$result" -# ---- Test get_vram_free_mb dual GPU (no container context — sums all) ---- +# ---- Test get_vram_free_mb dual GPU (no container context — MIN, not sum) ---- +# TP OOMs on whichever card runs out first, so min per-card free is the honest +# margin; the old sum overstated dual-rig headroom ~2x (corrected 2026-07-04 +# while dogfooding #246). cat > "${tmp_dir}/nvidia-smi" <<'EOF' #!/usr/bin/env bash case "$*" in @@ -124,7 +127,7 @@ chmod +x "${tmp_dir}/nvidia-smi" result="$(CONTAINER=none PATH="${tmp_dir}:/usr/bin:/bin" \ bash -c "source '$HELPERS_FILE'; get_vram_free_mb")" -assert_eq "get_vram_free_mb dual-GPU sum (no container)" "23500" "$result" +assert_eq "get_vram_free_mb dual-GPU min (no container)" "11500" "$result" # ---- Test get_vram_free_mb single GPU ---- cat > "${tmp_dir}/nvidia-smi" <<'EOF' diff --git a/scripts/verify-stress.sh b/scripts/verify-stress.sh index c79445f9..f7863527 100755 --- a/scripts/verify-stress.sh +++ b/scripts/verify-stress.sh @@ -55,6 +55,10 @@ # SKIP_LONGCTX Set to 1 to skip the long-context needle ladder. # SKIP_TOOL_PREFILL Set to 1 to skip the tool-response prefill test. # SKIP_CEILING Set to 1 to skip the context ceiling ladder (#199). +# STRESS_FAST Set to 1 for the A/B-tier fast mode (#246): skips +# probe 7's large fresh needles and caps the ceiling +# ladder at 2 rungs (~2/3 anchor + ceiling). Roughly +# halves wall time; full mode stays the gate default. # PREFILL_TARGET_CHARS Tool-response prefill payload size in chars # (default: 100000 ≈ 25K tokens; set higher to # push closer to the cliff under investigation). @@ -918,6 +922,14 @@ check_longctx_large() { skip "SKIP_LONGCTX=1" return 0 fi + if [[ "${STRESS_FAST:-0}" == "1" ]]; then + # #246 A/B tier: these fresh 60K/90K fills cost ~4 min and near-duplicate + # the ceiling ladder's first-rung depth; depth recall transfers to the + # rungs. Full mode keeps the fresh-context vs accumulated-context + # distinction (this probe is fresh-fill; the ladder is staggered). + skip "STRESS_FAST=1 — large fresh needles skipped (depth coverage via ceiling rungs)" + return 0 + fi LONGCTX_SCALES="900 1400" check_longctx } # Engine health check after crash-prone probes (7, 8). @@ -1050,10 +1062,13 @@ except Exception: # env to identify which GPU(s) the model actually uses. # # Priority: -# 1. Docker HostConfig.DeviceRequests[0].DeviceIDs (compose device_ids) -# 2. Container env CUDA_VISIBLE_DEVICES -# 3. Container env NVIDIA_VISIBLE_DEVICES (if numeric, e.g. "0") -# 4. Fallback: sum all GPUs with a warning (multi-GPU host, can't determine subset) +# 1. Docker HostConfig.DeviceRequests DeviceIDs (compose device_ids), or +# Count: -1 (compose `count: all`) -> ALL host GPUs, no warning +# 2. Container env CUDA_VISIBLE_DEVICES ('all' -> ALL) +# 3. Container env NVIDIA_VISIBLE_DEVICES (numeric, or 'all' -> ALL) +# 4. Fallback: min over all GPUs with a warning (genuinely undetermined) +# Aggregation is MIN free across the model's GPUs (TP OOMs on the first card +# to run out — a sum overstates dual-rig margins ~2x; corrected 2026-07-04). # Returns 0 when nvidia-smi is unavailable. get_vram_free_mb() { if ! command -v nvidia-smi >/dev/null 2>&1; then @@ -1078,16 +1093,27 @@ try: if ids: print(','.join(ids)) sys.exit(0) + # count: all (compose) -> Count: -1, DeviceIDs: null. That IS a + # determined answer: the container sees every host GPU. + if dr.get('Count') == -1: + print('ALL') + sys.exit(0) # Fallback: check env vars for e in cfg.get('Config', {}).get('Env', []) or []: if e.startswith('CUDA_VISIBLE_DEVICES='): val = e.split('=', 1)[1] - if val and val != 'all': + if val == 'all': + print('ALL') + sys.exit(0) + if val: print(val) sys.exit(0) if e.startswith('NVIDIA_VISIBLE_DEVICES='): val = e.split('=', 1)[1] - if val and val != 'all' and val.replace(',', '').isdigit(): + if val == 'all': + print('ALL') + sys.exit(0) + if val and val.replace(',', '').isdigit(): print(val) sys.exit(0) print('') @@ -1096,18 +1122,26 @@ except Exception: " 2>/dev/null)" fi - if [[ -n "$gpu_ids" ]]; then + # Aggregation: MIN free across the model's GPUs, not sum. TP splits the KV + # pool and activations per card and OOM hits whichever card runs out first, + # so min is the honest margin (sum overstated dual-rig headroom ~2x; noted + # 2026-07-04 while dogfooding #246 — this makes the margin advisory + # STRICTER on multi-GPU configs than earlier runs' summed readings). + if [[ "$gpu_ids" == "ALL" ]]; then + nvidia-smi --query-gpu=memory.free --format=csv,noheader,nounits 2>/dev/null \ + | awk 'NR==1||$1/dev/null || echo 0 + elif [[ -n "$gpu_ids" ]]; then nvidia-smi -i "$gpu_ids" --query-gpu=memory.free --format=csv,noheader,nounits 2>/dev/null \ - | awk '{s+=$1} END {printf "%.0f\n", s}' 2>/dev/null || echo 0 + | awk 'NR==1||$1/dev/null || echo 0 else - # Can't determine which GPUs — sum all, but flag it + # Genuinely undetermined (no docker, exotic env) — min over all, flagged local total_gpus total_gpus="$(nvidia-smi --query-gpu=index --format=csv,noheader 2>/dev/null | wc -l)" if [[ "$total_gpus" -gt 1 ]]; then - echo " [vram] WARN: could not determine model GPU(s) on ${total_gpus}-GPU host — summing all (margin may be inflated)" >&2 + echo " [vram] WARN: could not determine model GPU(s) on ${total_gpus}-GPU host — using min free across all" >&2 fi nvidia-smi --query-gpu=memory.free --format=csv,noheader,nounits 2>/dev/null \ - | awk '{s+=$1} END {printf "%.0f\n", s}' 2>/dev/null || echo 0 + | awk 'NR==1||$1/dev/null || echo 0 fi } @@ -1144,6 +1178,21 @@ check_ceiling_ladder() { return 0 fi + # STRESS_FAST (#246 A/B tier): cap the ladder at 2 rungs — one mid anchor + # at ~2/3 of the ceiling target, then the ceiling. Every rung is a FULL + # prefill at its depth (measured: rungs dominate this script's wall time), + # so the rung budget IS the time budget (861s -> ~450s on the 262K dual). + # Depth coverage stays 4-anchor: bench probe 10K + 90K, mid rung, ceiling. + # Never ADDS rungs on small-ctx configs (max() guards). Explicit CEILING_* + # overrides win. + if [[ "${STRESS_FAST:-0}" == "1" ]]; then + [[ -z "${CEILING_START_TOKENS:-}" ]] && \ + ceiling_start="$(python3 -c "print(max(${ceiling_start}, ${ceiling_top} * 2 // 3))")" + [[ -z "${CEILING_STEP_TOKENS:-}" ]] && \ + ceiling_step="$(python3 -c "print(max(${ceiling_step}, ${ceiling_top} - ${ceiling_start} + 1))")" + echo " STRESS_FAST=1 — ladder capped (start=${ceiling_start}, step=${ceiling_step}); full-mode gates use the fine ladder" + fi + # Compute rungs as a space-separated list local rungs rungs="$(python3 -c "