Files
club-3090/scripts/tests/test-launch-compat.sh
T
noonghunnaandClaude Fable 5 e8bcfd8da1 Launcher arch-aware KV dtype injection for pilot slugs (#246 Phase 1)
launch.sh/switch.sh now detect GPU arch and export KV_CACHE_DTYPE=
fp8_e4m3 (native FP8 compute) on sm_89+ cards for the pilot slugs
(vllm/dual, vllm/minimal). The injected value is the hardware
profiles' dormant kv_format_default.balanced -- one source of truth
shared with the pull gates; the Ampere no-op is data equality
(3090-class balanced = fp8_e5m2 = compose default -> nothing emitted),
not a code branch.

Injection guards (all load-bearing):
- pilot allowlist only; expansion gated on the #246 cross-rig A/B
- per-variant: registry kv_format == fp8_e5m2 only (int8-PTH/TQ/bf16
  slugs never touched -- compressed-tensors weights reject fp8 KV)
- vLLM-family variants only; explicit user KV_CACHE_DTYPE= wins;
  unmapped cards / heterogeneous rigs / no nvidia-smi -> no injection
- direct `docker compose up` keeps Ampere-safe compose defaults

Rides the existing resolve-variant-pin export seam (new optional
--gpu-spec); VLLM_ATTENTION_BACKEND is whitelisted but ships no value
(vLLM auto-detect stays the default until measured). Preflight banner
names the detected arch class.

Consistency fixes the injection exposed:
- gates.py _ARCH_KERNEL_SM: fp8_e4m3 9.0 -> 8.9 (vLLM's real floor is
  SM89+; we'd otherwise inject e4m3 on 4090s our own pull-gate calls
  unloadable) + new nvfp4: 10.0 row (v0.24.0 literal had NO gate --
  a 3090 pull of an nvfp4 config wouldn't have been rejected)
- Blackwell hardware profiles: nvfp4 declared as CANDIDATE capability
  (engine list unchanged until validated -- gates take the intersection)
- kv-calc: projected nvfp4 rows (bytes/elem + activation coefs),
  calibration unchanged

Docs: HARDWARE.md new section, KV_MATH/QUANTIZATION/DTYPE_MATRIX rows
(incl. retiring the Genesis-era "e4m3 undertuned per #51" advisory in
favor of the A/B).

Validated: 8-case injection matrix in test-launch-compat; live no-op
on the real 2x3090 (spec built, nothing injected); faked 4090 through
the real bash detection path emits e4m3; compose interpolation both
ways; kv-calc --calibration green; full scripts gate 64/64.

Co-Authored-By: Claude Fable 5 <[email protected]>
Claude-Session: https://claude.ai/code/session_01EfF565T9eSLaqGzidyJ1Pm
2026-07-04 20:45:28 +00:00

147 lines
6.1 KiB
Bash
Executable File

#!/usr/bin/env bash
set -euo pipefail
ROOT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../.." && pwd)"
HELPER="${ROOT_DIR}/scripts/lib/profiles/launch_compat.py"
GPU_3090='0|RTX_3090|24576|8.6'
MTP_SHA="01d4d1ad375dc5854779c593eee093bcebb0cada"
CLEAN_SHA="bf610c2f56764e1b30bc6065f4ceace3d6e59036"
DFLASH_SHA="e47c98ef7a38792996e452ef53914e21e41928e9"
assert_contains() {
local haystack="$1"
local needle="$2"
if [[ "$haystack" != *"$needle"* ]]; then
echo "ASSERTION FAILED: expected output to contain: $needle" >&2
echo "--- output ---" >&2
echo "$haystack" >&2
exit 1
fi
}
assert_not_contains() {
local haystack="$1"
local needle="$2"
if [[ "$haystack" == *"$needle"* ]]; then
echo "ASSERTION FAILED: expected output not to contain: $needle" >&2
echo "--- output ---" >&2
echo "$haystack" >&2
exit 1
fi
}
out="$(python3 "$HELPER" filter-candidates \
--variants vllm/dual,vllm/minimal,llamacpp/default \
--model qwen3.6-27b \
--gpu-spec "$GPU_3090" \
--tp 1 \
--pp 1 \
--workload fast-chat)"
assert_contains "$out" "vllm/minimal"
assert_not_contains "$out" "vllm/dual"
out="$(python3 "$HELPER" filter-candidates \
--variants vllm/minimal,llamacpp/default,llamacpp/mtp \
--model qwen3.6-27b \
--gpu-spec "$GPU_3090" \
--tp 1 \
--pp 1 \
--stable)"
assert_contains "$out" "vllm/minimal"
assert_contains "$out" "llamacpp/default"
assert_contains "$out" "llamacpp/mtp"
if out="$(python3 "$HELPER" validate-variant \
--variant vllm/gemma-mtp-tp1 \
--gpu-spec "$GPU_3090" \
--tp 2 \
--pp 1 \
--no-project-vram 2>&1)"; then
echo "ASSERTION FAILED: invalid Gemma single-card profile unexpectedly passed" >&2
echo "$out" >&2
exit 1
fi
assert_contains "$out" "C1: tp=2 * pp=1 = 2 != 1 cards selected"
assert_contains "$out" "C5: kv_format=fp8_e4m3 not supported by hardware: rtx-3090"
out="$(python3 "$HELPER" validate-variant \
--variant vllm/minimal \
--gpu-spec "$GPU_3090" \
--tp 1 \
--pp 1 \
--no-project-vram \
--verbose 2>&1)"
assert_contains "$out" "Pass 1 fits()"
assert_contains "$out" "Resolved compose: vllm/minimal"
assert_contains "$out" "Pass 2 fits()"
out="$(python3 "$HELPER" resolve-engine-pin --engine-id vllm-nightly-mtp --format shell)"
assert_contains "$out" "VLLM_NIGHTLY_SHA=${MTP_SHA}"
if out="$(python3 "$HELPER" resolve-engine-pin --engine-id vllm-pip-baseline --format shell 2>&1)"; then
echo "ASSERTION FAILED: pip-only vllm-pip-baseline unexpectedly resolved as a docker nightly" >&2
echo "$out" >&2
exit 1
fi
assert_contains "$out" "install.spec is not a docker image"
out="$(python3 "$HELPER" resolve-variant-pin --variant vllm/dual --format shell)"
assert_contains "$out" "VLLM_IMAGE=vllm/vllm-openai:v0.24.0"
out="$(python3 "$HELPER" resolve-variant-pin --variant vllm/gemma-int8-mtp --format shell)"
assert_contains "$out" "VLLM_IMAGE=vllm/vllm-openai:v0.22.0"
# --- #246 arch-aware KV injection matrix (resolve-variant-pin --gpu-spec) ----
GPU_4090='0|NVIDIA GeForce RTX 4090|24564|8.9'
GPU_5090X2='0|NVIDIA GeForce RTX 5090|32607|12.0;1|NVIDIA GeForce RTX 5090|32607|12.0'
# ampere -> NOTHING injected (compose defaults; the no-op is data equality)
out="$(python3 "$HELPER" resolve-variant-pin --variant vllm/dual --format shell --gpu-spec "$GPU_3090")"
assert_not_contains "$out" "KV_CACHE_DTYPE"
# ada / blackwell pilot slugs -> native-fp8 swap
out="$(python3 "$HELPER" resolve-variant-pin --variant vllm/dual --format shell --gpu-spec "$GPU_4090")"
assert_contains "$out" "KV_CACHE_DTYPE=fp8_e4m3"
out="$(python3 "$HELPER" resolve-variant-pin --variant vllm/minimal --format shell --gpu-spec "$GPU_5090X2")"
assert_contains "$out" "KV_CACHE_DTYPE=fp8_e4m3"
# non-pilot slug (same kv_format) -> no injection until the #246 A/B expands the pilot
out="$(python3 "$HELPER" resolve-variant-pin --variant vllm/qwen-27b-dual-fast --format shell --gpu-spec "$GPU_4090")"
assert_not_contains "$out" "KV_CACHE_DTYPE"
# quant-specific KV slug -> never overridden (compressed-tensors reject fp8 KV)
out="$(python3 "$HELPER" resolve-variant-pin --variant vllm/gemma-int8-mtp --format shell --gpu-spec "$GPU_4090")"
assert_not_contains "$out" "KV_CACHE_DTYPE"
# explicit user env pin wins
out="$(KV_CACHE_DTYPE=fp8_e5m2 python3 "$HELPER" resolve-variant-pin --variant vllm/dual --format shell --gpu-spec "$GPU_4090")"
assert_not_contains "$out" "KV_CACHE_DTYPE"
# heterogeneous rig -> no single right answer -> no injection
out="$(python3 "$HELPER" resolve-variant-pin --variant vllm/dual --format shell --gpu-spec "${GPU_3090};1|NVIDIA GeForce RTX 4090|24564|8.9")"
assert_not_contains "$out" "KV_CACHE_DTYPE"
# unmapped card -> degrade to compose defaults, never an error
out="$(python3 "$HELPER" resolve-variant-pin --variant vllm/dual --format shell --gpu-spec "0|Weird GPU|8192|7.0")"
assert_contains "$out" "VLLM_IMAGE=vllm/vllm-openai:v0.24.0"
assert_not_contains "$out" "KV_CACHE_DTYPE"
echo " ok: #246 arch-aware KV injection matrix (8 cases)"
if command -v docker >/dev/null 2>&1 && docker compose version >/dev/null 2>&1; then
out="$(VLLM_NIGHTLY_SHA="$CLEAN_SHA" docker compose -f "$ROOT_DIR/models/qwen3.6-27b/vllm/compose/dual/autoround-int4/fp8-mtp.yml" config 2>/dev/null)"
assert_contains "$out" "image: vllm/vllm-openai:v0.24.0"
out="$(VLLM_NIGHTLY_SHA="$CLEAN_SHA" VLLM_IMAGE=vllm/vllm-openai:latest docker compose -f "$ROOT_DIR/models/qwen3.6-27b/vllm/compose/dual/autoround-int4/fp8-mtp.yml" config 2>/dev/null)"
assert_contains "$out" "image: vllm/vllm-openai:latest"
fi
out="$(python3 - <<'PY'
from scripts.lib.profiles.compat import InstanceSpec
from scripts.lib.profiles.estate_cli import compose_env
clean = compose_env(InstanceSpec(name="qwen", compose_name="vllm/dual", gpu_indices=(0, 1), port=8010))
gemma = compose_env(InstanceSpec(name="gemma", compose_name="vllm/gemma-int8-mtp", gpu_indices=(0, 1), port=8032))
print(clean["VLLM_IMAGE"])
print(gemma["VLLM_IMAGE"])
PY
)"
assert_contains "$out" "vllm/vllm-openai:v0.24.0" # clean (vllm/dual → vllm-stable) bumped to v0.24.0
assert_contains "$out" "vllm/vllm-openai:v0.22.0" # gemma (vllm/gemma-int8-mtp → vllm-gemma-stable) stays v0.22.0
echo "test-launch-compat: ok"