Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
98f0406d0f | ||
|
|
2a92a199ff |
@@ -16,6 +16,27 @@ history; SemVer takes over from `v0.3.0` onward.
|
||||
|
||||
---
|
||||
|
||||
## v0.6.1 — 2026-05-14
|
||||
|
||||
|
||||
### ✨ Features
|
||||
|
||||
- feat(launch): add hardware-aware launcher ([5882bbe](https://github.com/noonghunna/club-3090/commit/5882bbef6f5ed7e8ceea450fa3a6b167a1bf4926))
|
||||
- feat(tools): extend kv-calc.py to multi-model (Qwen 3.6 + Gemma 4 31B) ([0d48dac](https://github.com/noonghunna/club-3090/commit/0d48dac818ba889f74b74df1c454bb961a123c37))
|
||||
|
||||
|
||||
### 📝 Documentation
|
||||
|
||||
- docs: update launch.sh references for v0.6.1 wizard flow ([e299e70](https://github.com/noonghunna/club-3090/commit/e299e70451c8d146214a6560e582d0e174dd0ebc))
|
||||
|
||||
|
||||
### 🧹 Other
|
||||
|
||||
- Merge codex/v0.6.1-launch into master ([056dcb6](https://github.com/noonghunna/club-3090/commit/056dcb643914fee6169b02b89cb420b038c29b0f))
|
||||
|
||||
|
||||
|
||||
[Pin: `git checkout v0.6.1`] · [Full diff](https://github.com/noonghunna/club-3090/compare/v0.6.0...v0.6.1)
|
||||
## v0.6.0 — 2026-05-13
|
||||
|
||||
|
||||
|
||||
@@ -706,12 +706,6 @@ kv_projection() {
|
||||
echo "[launch] KV projection only available for vLLM variants today." >&2
|
||||
return 0
|
||||
fi
|
||||
if (( TP_VALUE > 4 )); then
|
||||
echo "[launch] KV projection skipped: tools/kv-calc.py currently models TP up to 4." >&2
|
||||
echo "[launch] Proceeding with launch-side head-divisibility validation only." >&2
|
||||
return 0
|
||||
fi
|
||||
|
||||
local kv_model="${mapping%%:*}" kv_compose="${mapping#*:}" kv_json status
|
||||
if kv_json="$("${ROOT_DIR}/tools/kv-calc.py" --model "$kv_model" --compose "$kv_compose" --vram "$MIN_VRAM_GB" --tp "$TP_VALUE" --json 2>&1)"; then
|
||||
status=0
|
||||
@@ -783,7 +777,6 @@ if [[ -z "$VARIANT" ]]; then
|
||||
VARIANT="${CANDIDATE_VARIANTS[0]}"
|
||||
fi
|
||||
echo "[launch] model: $(model_label "$MODEL_NAME")" >&2
|
||||
echo "[launch] selected variant: ${VARIANT}" >&2
|
||||
if (( HET_VRAM_MIXED == 1 && TP_VALUE > 1 )); then
|
||||
echo "[launch] Note: heterogeneous TP is bottlenecked by the smallest selected card (${MIN_VRAM_GB} GB)." >&2
|
||||
fi
|
||||
|
||||
@@ -131,6 +131,7 @@ assert_contains "$out" "[model] SKIP_MODEL=1"
|
||||
# skip prompts, select the expected variant, and export GPU / TP / PP envs.
|
||||
mkdir -p "${TMP_DIR}/models/qwen3.6-27b-autoround-int4" \
|
||||
"${TMP_DIR}/models/gemma-4-31b-autoround-int4"
|
||||
FAKE_8X3090='0:RTX_3090:24576:8.6,1:RTX_3090:24576:8.6,2:RTX_3090:24576:8.6,3:RTX_3090:24576:8.6,4:RTX_3090:24576:8.6,5:RTX_3090:24576:8.6,6:RTX_3090:24576:8.6,7:RTX_3090:24576:8.6'
|
||||
|
||||
out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6' \
|
||||
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
|
||||
@@ -143,6 +144,12 @@ out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:
|
||||
--no-preflight --no-verify --model qwen3.6-27b --gpus 0,1 --no-projection 2>&1)"
|
||||
assert_contains "$out" "[launch] Tensor parallel TP=2"
|
||||
assert_contains "$out" "SWITCHED vllm/dual CUDA=0,1 NVD=0,1 TP=2 PP=1"
|
||||
selected_count="$(grep -c "\[launch\] selected variant:" <<< "$out" || true)"
|
||||
if [[ "$selected_count" != "1" ]]; then
|
||||
echo "ASSERTION FAILED: expected one selected-variant line, got ${selected_count}" >&2
|
||||
echo "$out" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6' \
|
||||
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
|
||||
@@ -162,6 +169,26 @@ if out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6
|
||||
fi
|
||||
assert_contains "$out" "Valid TP values: 1 2 4"
|
||||
|
||||
out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS="${FAKE_8X3090}" \
|
||||
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
|
||||
--no-preflight --no-verify --model gemma-4-31b --gpus 0,1,2,3,4,5,6,7 --tp 8 2>&1)"
|
||||
assert_contains "$out" "[launch] Tensor parallel TP=8"
|
||||
assert_contains "$out" "[launch] Suggested: vllm/gemma-mtp"
|
||||
assert_contains "$out" "VRAM budget — per card"
|
||||
assert_contains "$out" "Note: TP > 4 predictions are extrapolated"
|
||||
assert_not_contains "$out" "KV projection skipped"
|
||||
assert_contains "$out" "SWITCHED vllm/gemma-mtp CUDA=0,1,2,3,4,5,6,7 NVD=0,1,2,3,4,5,6,7 TP=8 PP=1"
|
||||
|
||||
if out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS="${FAKE_8X3090}" \
|
||||
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
|
||||
--no-preflight --no-verify --model qwen3.6-27b --gpus 0,1,2,3,4,5,6,7 --tp 8 --no-projection 2>&1)"; then
|
||||
echo "ASSERTION FAILED: invalid Qwen TP=8 unexpectedly succeeded" >&2
|
||||
echo "$out" >&2
|
||||
exit 1
|
||||
fi
|
||||
assert_contains "$out" "num_kv_heads does not divide TP=8"
|
||||
assert_contains "$out" "Valid TP values: 1 2 4"
|
||||
|
||||
# TTY-backed no-arg setup supports the cosmetic but real "Both" choice by
|
||||
# dispatching through the positional path for both model families.
|
||||
if ! command -v script >/dev/null 2>&1; then
|
||||
|
||||
+30
-5
@@ -1,5 +1,10 @@
|
||||
#!/usr/bin/env python3
|
||||
"""kv-calc.py — predict per-card VRAM budget for vLLM composes.
|
||||
#!/bin/sh
|
||||
''':'
|
||||
exec python3 "$0" "$@"
|
||||
':'''
|
||||
from __future__ import annotations
|
||||
|
||||
__doc__ = """kv-calc.py — predict per-card VRAM budget for vLLM composes.
|
||||
|
||||
Predicts (per card, after TP split):
|
||||
- Model weights
|
||||
@@ -37,8 +42,6 @@ Usage:
|
||||
bash tools/kv-calc.py --calibration # both models, grouped per-model
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
@@ -60,6 +63,7 @@ QWEN36_27B = {
|
||||
"num_attn_layers": 16, # full_attention layers
|
||||
"num_attn_heads": 24,
|
||||
"num_kv_heads": 4, # GQA
|
||||
"valid_tp": [1, 2, 4],
|
||||
"head_dim_attn": 256, # attention head dim
|
||||
"linear_num_v_heads": 48, # GDN value heads
|
||||
"linear_num_k_heads": 16, # GDN key heads (GQA-style at the GDN level too)
|
||||
@@ -86,6 +90,7 @@ GEMMA4_31B = {
|
||||
"num_sliding_attn_layers": 50, # sliding_attention (fixed window, head_dim=256)
|
||||
"num_attn_heads": 32,
|
||||
"num_kv_heads": 16, # GQA 2:1
|
||||
"valid_tp": [1, 2, 4, 8, 16],
|
||||
"head_dim_sliding": 256, # sliding_attention head dim
|
||||
"global_head_dim": 512, # full_attention head dim (asymmetric)
|
||||
"sliding_window": 1024,
|
||||
@@ -357,6 +362,16 @@ def cudagraph_overhead_gb(mem_util, tp):
|
||||
return base + tp_bump
|
||||
|
||||
|
||||
def _validate_tp_for_spec(spec, tp):
|
||||
valid_tp = spec.get("valid_tp")
|
||||
if valid_tp and tp not in valid_tp:
|
||||
raise ValueError(
|
||||
f"TP={tp} invalid for {spec['model_id']} "
|
||||
f"(num_kv_heads={spec['num_kv_heads']} cannot be divided across TP cleanly). "
|
||||
f"Valid TP values: {valid_tp}"
|
||||
)
|
||||
|
||||
|
||||
def predict(
|
||||
spec=QWEN36_27B,
|
||||
kv_format="fp8_e5m2",
|
||||
@@ -380,6 +395,8 @@ def predict(
|
||||
drafter_gb: total drafter weight (MTP / DFlash) — split by TP.
|
||||
dflash_draft_gb: legacy alias — folded into drafter_gb if set.
|
||||
"""
|
||||
_validate_tp_for_spec(spec, tp)
|
||||
|
||||
weights_gb = _weights_per_card_gb(spec, tp, weights_variant)
|
||||
|
||||
growing_b, sliding_b = kv_pool_per_card_bytes(
|
||||
@@ -441,6 +458,8 @@ def predict(
|
||||
notes.append("⚠ fp8_e4m3 on Ampere (sm_86): Triton `fp8e4nv` kernel unsupported; use int8_per_token_head instead (PR #40391 via #42102)")
|
||||
if spec["model_family"] == "gemma4-swa-dense" and tp == 1 and vram_gb < 32:
|
||||
notes.append("⚠ Gemma 4 31B TP=1 needs ≥32 GB VRAM; 24 GB Ampere boot-OOMs (model weights + drafter + min KV)")
|
||||
if tp > 4:
|
||||
notes.append("TP > 4 predictions are extrapolated; report deltas via scripts/report.sh --bench")
|
||||
|
||||
return Prediction(
|
||||
model=spec["model_id"],
|
||||
@@ -636,7 +655,7 @@ def main():
|
||||
help="KV cache format. Default: from --compose, or fp8_e5m2.")
|
||||
p.add_argument("--max-ctx", type=int, help="max_model_len. Default: from --compose, or 180000.")
|
||||
p.add_argument("--max-num-seqs", type=int, help="max_num_seqs. Default: from --compose, or 1.")
|
||||
p.add_argument("--tp", type=int, choices=[1, 2, 4], help="tensor_parallel_size. Default: from --compose, or 1.")
|
||||
p.add_argument("--tp", type=int, choices=[1, 2, 4, 8, 16], help="tensor_parallel_size. Default: from --compose, or 1.")
|
||||
p.add_argument("--mem-util", type=float, help="gpu_memory_utilization. Default: from --compose, or 0.95.")
|
||||
p.add_argument("--vram", type=float, default=24, help="VRAM per card in GB. Default 24.")
|
||||
p.add_argument("--mtp", action="store_true", default=None, help="MTP enabled (Qwen: n=3 built-in; Gemma: external drafter).")
|
||||
@@ -690,6 +709,12 @@ def main():
|
||||
weights_variant = args.weights_variant or "default"
|
||||
header = f"Predicted budget — {model_key} custom config on {args.vram} GB VRAM (kv={kv_format}, ctx={max_ctx:,}, seqs={max_num_seqs}, TP={tp}, mem={mem_util})"
|
||||
|
||||
try:
|
||||
_validate_tp_for_spec(spec, tp)
|
||||
except ValueError as exc:
|
||||
print(f"ERROR: {exc}", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
if args.solve_max_ctx:
|
||||
best = solve_max_ctx(
|
||||
spec, kv_format=kv_format, max_num_seqs=max_num_seqs,
|
||||
|
||||
Reference in New Issue
Block a user