2 Commits
Author SHA1 Message Date
noonghunna 98f0406d0f fix(launch): project TP greater than four
Release / release (push) Failing after 48s
2026-05-14 00:23:32 +00:00
github-actions[bot] 2a92a199ff chore(changelog): regenerate for v0.6.1 [skip ci] 2026-05-14 00:02:46 +00:00
4 changed files with 78 additions and 12 deletions
+21
View File
@@ -16,6 +16,27 @@ history; SemVer takes over from `v0.3.0` onward.
---
## v0.6.1 — 2026-05-14
### ✨ Features
- feat(launch): add hardware-aware launcher ([5882bbe](https://github.com/noonghunna/club-3090/commit/5882bbef6f5ed7e8ceea450fa3a6b167a1bf4926))
- feat(tools): extend kv-calc.py to multi-model (Qwen 3.6 + Gemma 4 31B) ([0d48dac](https://github.com/noonghunna/club-3090/commit/0d48dac818ba889f74b74df1c454bb961a123c37))
### 📝 Documentation
- docs: update launch.sh references for v0.6.1 wizard flow ([e299e70](https://github.com/noonghunna/club-3090/commit/e299e70451c8d146214a6560e582d0e174dd0ebc))
### 🧹 Other
- Merge codex/v0.6.1-launch into master ([056dcb6](https://github.com/noonghunna/club-3090/commit/056dcb643914fee6169b02b89cb420b038c29b0f))
[Pin: `git checkout v0.6.1`] · [Full diff](https://github.com/noonghunna/club-3090/compare/v0.6.0...v0.6.1)
## v0.6.0 — 2026-05-13
-7
View File
@@ -706,12 +706,6 @@ kv_projection() {
echo "[launch] KV projection only available for vLLM variants today." >&2
return 0
fi
if (( TP_VALUE > 4 )); then
echo "[launch] KV projection skipped: tools/kv-calc.py currently models TP up to 4." >&2
echo "[launch] Proceeding with launch-side head-divisibility validation only." >&2
return 0
fi
local kv_model="${mapping%%:*}" kv_compose="${mapping#*:}" kv_json status
if kv_json="$("${ROOT_DIR}/tools/kv-calc.py" --model "$kv_model" --compose "$kv_compose" --vram "$MIN_VRAM_GB" --tp "$TP_VALUE" --json 2>&1)"; then
status=0
@@ -783,7 +777,6 @@ if [[ -z "$VARIANT" ]]; then
VARIANT="${CANDIDATE_VARIANTS[0]}"
fi
echo "[launch] model: $(model_label "$MODEL_NAME")" >&2
echo "[launch] selected variant: ${VARIANT}" >&2
if (( HET_VRAM_MIXED == 1 && TP_VALUE > 1 )); then
echo "[launch] Note: heterogeneous TP is bottlenecked by the smallest selected card (${MIN_VRAM_GB} GB)." >&2
fi
+27
View File
@@ -131,6 +131,7 @@ assert_contains "$out" "[model] SKIP_MODEL=1"
# skip prompts, select the expected variant, and export GPU / TP / PP envs.
mkdir -p "${TMP_DIR}/models/qwen3.6-27b-autoround-int4" \
"${TMP_DIR}/models/gemma-4-31b-autoround-int4"
FAKE_8X3090='0:RTX_3090:24576:8.6,1:RTX_3090:24576:8.6,2:RTX_3090:24576:8.6,3:RTX_3090:24576:8.6,4:RTX_3090:24576:8.6,5:RTX_3090:24576:8.6,6:RTX_3090:24576:8.6,7:RTX_3090:24576:8.6'
out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6' \
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
@@ -143,6 +144,12 @@ out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:
--no-preflight --no-verify --model qwen3.6-27b --gpus 0,1 --no-projection 2>&1)"
assert_contains "$out" "[launch] Tensor parallel TP=2"
assert_contains "$out" "SWITCHED vllm/dual CUDA=0,1 NVD=0,1 TP=2 PP=1"
selected_count="$(grep -c "\[launch\] selected variant:" <<< "$out" || true)"
if [[ "$selected_count" != "1" ]]; then
echo "ASSERTION FAILED: expected one selected-variant line, got ${selected_count}" >&2
echo "$out" >&2
exit 1
fi
if out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6' \
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
@@ -162,6 +169,26 @@ if out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6
fi
assert_contains "$out" "Valid TP values: 1 2 4"
out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS="${FAKE_8X3090}" \
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
--no-preflight --no-verify --model gemma-4-31b --gpus 0,1,2,3,4,5,6,7 --tp 8 2>&1)"
assert_contains "$out" "[launch] Tensor parallel TP=8"
assert_contains "$out" "[launch] Suggested: vllm/gemma-mtp"
assert_contains "$out" "VRAM budget — per card"
assert_contains "$out" "Note: TP > 4 predictions are extrapolated"
assert_not_contains "$out" "KV projection skipped"
assert_contains "$out" "SWITCHED vllm/gemma-mtp CUDA=0,1,2,3,4,5,6,7 NVD=0,1,2,3,4,5,6,7 TP=8 PP=1"
if out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS="${FAKE_8X3090}" \
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
--no-preflight --no-verify --model qwen3.6-27b --gpus 0,1,2,3,4,5,6,7 --tp 8 --no-projection 2>&1)"; then
echo "ASSERTION FAILED: invalid Qwen TP=8 unexpectedly succeeded" >&2
echo "$out" >&2
exit 1
fi
assert_contains "$out" "num_kv_heads does not divide TP=8"
assert_contains "$out" "Valid TP values: 1 2 4"
# TTY-backed no-arg setup supports the cosmetic but real "Both" choice by
# dispatching through the positional path for both model families.
if ! command -v script >/dev/null 2>&1; then
+30 -5
View File
@@ -1,5 +1,10 @@
#!/usr/bin/env python3
"""kv-calc.py — predict per-card VRAM budget for vLLM composes.
#!/bin/sh
''':'
exec python3 "$0" "$@"
':'''
from __future__ import annotations
__doc__ = """kv-calc.py — predict per-card VRAM budget for vLLM composes.
Predicts (per card, after TP split):
- Model weights
@@ -37,8 +42,6 @@ Usage:
bash tools/kv-calc.py --calibration # both models, grouped per-model
"""
from __future__ import annotations
import argparse
import json
import sys
@@ -60,6 +63,7 @@ QWEN36_27B = {
"num_attn_layers": 16, # full_attention layers
"num_attn_heads": 24,
"num_kv_heads": 4, # GQA
"valid_tp": [1, 2, 4],
"head_dim_attn": 256, # attention head dim
"linear_num_v_heads": 48, # GDN value heads
"linear_num_k_heads": 16, # GDN key heads (GQA-style at the GDN level too)
@@ -86,6 +90,7 @@ GEMMA4_31B = {
"num_sliding_attn_layers": 50, # sliding_attention (fixed window, head_dim=256)
"num_attn_heads": 32,
"num_kv_heads": 16, # GQA 2:1
"valid_tp": [1, 2, 4, 8, 16],
"head_dim_sliding": 256, # sliding_attention head dim
"global_head_dim": 512, # full_attention head dim (asymmetric)
"sliding_window": 1024,
@@ -357,6 +362,16 @@ def cudagraph_overhead_gb(mem_util, tp):
return base + tp_bump
def _validate_tp_for_spec(spec, tp):
valid_tp = spec.get("valid_tp")
if valid_tp and tp not in valid_tp:
raise ValueError(
f"TP={tp} invalid for {spec['model_id']} "
f"(num_kv_heads={spec['num_kv_heads']} cannot be divided across TP cleanly). "
f"Valid TP values: {valid_tp}"
)
def predict(
spec=QWEN36_27B,
kv_format="fp8_e5m2",
@@ -380,6 +395,8 @@ def predict(
drafter_gb: total drafter weight (MTP / DFlash) — split by TP.
dflash_draft_gb: legacy alias — folded into drafter_gb if set.
"""
_validate_tp_for_spec(spec, tp)
weights_gb = _weights_per_card_gb(spec, tp, weights_variant)
growing_b, sliding_b = kv_pool_per_card_bytes(
@@ -441,6 +458,8 @@ def predict(
notes.append("⚠ fp8_e4m3 on Ampere (sm_86): Triton `fp8e4nv` kernel unsupported; use int8_per_token_head instead (PR #40391 via #42102)")
if spec["model_family"] == "gemma4-swa-dense" and tp == 1 and vram_gb < 32:
notes.append("⚠ Gemma 4 31B TP=1 needs ≥32 GB VRAM; 24 GB Ampere boot-OOMs (model weights + drafter + min KV)")
if tp > 4:
notes.append("TP > 4 predictions are extrapolated; report deltas via scripts/report.sh --bench")
return Prediction(
model=spec["model_id"],
@@ -636,7 +655,7 @@ def main():
help="KV cache format. Default: from --compose, or fp8_e5m2.")
p.add_argument("--max-ctx", type=int, help="max_model_len. Default: from --compose, or 180000.")
p.add_argument("--max-num-seqs", type=int, help="max_num_seqs. Default: from --compose, or 1.")
p.add_argument("--tp", type=int, choices=[1, 2, 4], help="tensor_parallel_size. Default: from --compose, or 1.")
p.add_argument("--tp", type=int, choices=[1, 2, 4, 8, 16], help="tensor_parallel_size. Default: from --compose, or 1.")
p.add_argument("--mem-util", type=float, help="gpu_memory_utilization. Default: from --compose, or 0.95.")
p.add_argument("--vram", type=float, default=24, help="VRAM per card in GB. Default 24.")
p.add_argument("--mtp", action="store_true", default=None, help="MTP enabled (Qwen: n=3 built-in; Gemma: external drafter).")
@@ -690,6 +709,12 @@ def main():
weights_variant = args.weights_variant or "default"
header = f"Predicted budget — {model_key} custom config on {args.vram} GB VRAM (kv={kv_format}, ctx={max_ctx:,}, seqs={max_num_seqs}, TP={tp}, mem={mem_util})"
try:
_validate_tp_for_spec(spec, tp)
except ValueError as exc:
print(f"ERROR: {exc}", file=sys.stderr)
return 2
if args.solve_max_ctx:
best = solve_max_ctx(
spec, kv_format=kv_format, max_num_seqs=max_num_seqs,