Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
27ca1a40bd | ||
|
|
9a039d8c92 | ||
|
|
39e18733aa | ||
|
|
6dc9a0dce1 | ||
|
|
e1d44bd732 | ||
|
|
273c017646 | ||
|
|
1d7aad112c | ||
|
|
e49c939748 | ||
|
|
92b69bd220 | ||
|
|
0053444e84 | ||
|
|
99328b4cda | ||
|
|
127f4f6d8f | ||
|
|
bdfb939edd | ||
|
|
9cd854dbcb | ||
|
|
8cf38b0590 | ||
|
|
1a233cd9fd | ||
|
|
97195fe443 | ||
|
|
3b2d940d26 | ||
|
|
cf0451a7ad | ||
|
|
2d1b1dc347 | ||
|
|
87f0a0c528 | ||
|
|
f7f6f444b9 | ||
|
|
15eda8a823 | ||
|
|
abf0e327f9 | ||
|
|
937871492a | ||
|
|
b1c68b4fe3 | ||
|
|
54d8bf0b3e | ||
|
|
67de3eca3e | ||
|
|
78f94cacf0 | ||
|
|
3114399983 | ||
|
|
6ec6a6761f | ||
|
|
5cb993be99 | ||
|
|
d116ba9ba3 | ||
|
|
6775d10919 | ||
|
|
2a148d702b | ||
|
|
57eb269cd7 | ||
|
|
ce2617e0bc | ||
|
|
02249ab193 | ||
|
|
6bd8e2420c |
@@ -7,15 +7,6 @@ on:
|
||||
description: "Upstream vLLM image to vendor overlays into"
|
||||
required: false
|
||||
default: "vllm/vllm-openai:nightly-1acd67a795ebccdf9b9db7697ae9082058301657"
|
||||
smoke:
|
||||
description: "GPU smoke behavior"
|
||||
required: false
|
||||
default: "auto"
|
||||
type: choice
|
||||
options:
|
||||
- auto
|
||||
- skip
|
||||
- required
|
||||
schedule:
|
||||
- cron: "0 0 * * 0"
|
||||
push:
|
||||
@@ -36,14 +27,12 @@ concurrency:
|
||||
cancel-in-progress: false
|
||||
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
packages: write
|
||||
|
||||
env:
|
||||
IMAGE_NAME: ghcr.io/noonghunna/vllm-club3090
|
||||
DEFAULT_VLLM_BASE_IMAGE: vllm/vllm-openai:nightly-1acd67a795ebccdf9b9db7697ae9082058301657
|
||||
CANONICAL_COMPOSE: models/qwen3.6-27b/vllm/compose/dual/docker-compose.yml
|
||||
|
||||
jobs:
|
||||
build:
|
||||
@@ -111,80 +100,11 @@ jobs:
|
||||
org.opencontainers.image.version=${{ steps.meta.outputs.image_tag }}
|
||||
club3090.upstream_vllm_image=${{ steps.meta.outputs.upstream_image }}
|
||||
|
||||
detect-smoke-runner:
|
||||
name: Detect self-hosted GPU runner
|
||||
promote-aliases:
|
||||
name: Promote latest and nightly-stable
|
||||
needs: build
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
available: ${{ steps.detect.outputs.available }}
|
||||
smoke_mode: ${{ steps.detect.outputs.smoke_mode }}
|
||||
steps:
|
||||
- name: Detect online gpu-labeled runner
|
||||
id: detect
|
||||
uses: actions/github-script@v7
|
||||
env:
|
||||
SMOKE_MODE: ${{ inputs.smoke || 'auto' }}
|
||||
with:
|
||||
script: |
|
||||
const mode = process.env.SMOKE_MODE || "auto";
|
||||
core.setOutput("smoke_mode", mode);
|
||||
|
||||
if (mode === "skip") {
|
||||
core.notice("Smoke explicitly skipped. Dated image was pushed; aliases will not move.");
|
||||
core.setOutput("available", "false");
|
||||
return;
|
||||
}
|
||||
|
||||
const runners = await github.paginate(
|
||||
github.rest.actions.listSelfHostedRunnersForRepo,
|
||||
{
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
per_page: 100,
|
||||
},
|
||||
);
|
||||
|
||||
const available = runners.some((runner) => {
|
||||
const labels = runner.labels.map((label) => label.name.toLowerCase());
|
||||
return runner.status === "online" &&
|
||||
labels.includes("self-hosted") &&
|
||||
labels.includes("gpu");
|
||||
});
|
||||
|
||||
core.setOutput("available", available ? "true" : "false");
|
||||
if (!available) {
|
||||
const message = "No online self-hosted runner with label 'gpu' was found. Dated image was pushed; latest/nightly-stable were not moved.";
|
||||
if (mode === "required") {
|
||||
core.setFailed(message);
|
||||
} else {
|
||||
core.notice(message);
|
||||
}
|
||||
}
|
||||
|
||||
smoke:
|
||||
name: GPU smoke and alias promotion
|
||||
needs:
|
||||
- build
|
||||
- detect-smoke-runner
|
||||
if: needs.detect-smoke-runner.outputs.available == 'true'
|
||||
runs-on:
|
||||
- self-hosted
|
||||
- linux
|
||||
- x64
|
||||
- gpu
|
||||
timeout-minutes: 120
|
||||
env:
|
||||
IMAGE_REF: ${{ needs.build.outputs.image_ref }}
|
||||
IMAGE_NAME: ghcr.io/noonghunna/vllm-club3090
|
||||
COMPOSE_PROJECT_NAME: club3090-ci-vllm-dual
|
||||
COMPOSE_OVERRIDE: /tmp/club3090-ci-vllm-image.override.yml
|
||||
URL: http://localhost:8010
|
||||
MODEL: qwen3.6-27b-autoround
|
||||
CONTAINER: vllm-qwen36-27b-dual
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Log in to GHCR
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
@@ -192,125 +112,16 @@ jobs:
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Prepare canonical compose override
|
||||
- name: Promote :latest and :nightly-stable to dated tag
|
||||
shell: bash
|
||||
env:
|
||||
IMAGE_REF: ${{ needs.build.outputs.image_ref }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
docker pull "${IMAGE_REF}"
|
||||
cat > "${COMPOSE_OVERRIDE}" <<EOF
|
||||
services:
|
||||
vllm-qwen36-27b-dual:
|
||||
image: ${IMAGE_REF}
|
||||
EOF
|
||||
|
||||
- name: Stop prior club-3090 estate if present
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [[ -f "${HOME}/.club3090/estate.yml" ]]; then
|
||||
bash scripts/launch.sh --down-estate "${HOME}/.club3090/estate.yml" || true
|
||||
fi
|
||||
docker compose \
|
||||
-f "${CANONICAL_COMPOSE}" \
|
||||
-f "${COMPOSE_OVERRIDE}" \
|
||||
-p "${COMPOSE_PROJECT_NAME}" \
|
||||
down --remove-orphans || true
|
||||
|
||||
- name: Boot canonical dual vLLM compose
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
docker compose \
|
||||
-f "${CANONICAL_COMPOSE}" \
|
||||
-f "${COMPOSE_OVERRIDE}" \
|
||||
-p "${COMPOSE_PROJECT_NAME}" \
|
||||
up -d
|
||||
|
||||
- name: Wait for OpenAI endpoint
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
for _ in {1..120}; do
|
||||
if curl -sf -m 5 "${URL}/v1/models" >/dev/null; then
|
||||
exit 0
|
||||
fi
|
||||
sleep 10
|
||||
done
|
||||
docker compose \
|
||||
-f "${CANONICAL_COMPOSE}" \
|
||||
-f "${COMPOSE_OVERRIDE}" \
|
||||
-p "${COMPOSE_PROJECT_NAME}" \
|
||||
logs --tail=200
|
||||
exit 1
|
||||
|
||||
- name: Run verify-full
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
URL="${URL}" MODEL="${MODEL}" CONTAINER="${CONTAINER}" bash scripts/verify-full.sh
|
||||
|
||||
- name: Run 3-prompt smoke bench
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
python3 - <<'PY'
|
||||
import json
|
||||
import time
|
||||
import urllib.request
|
||||
|
||||
url = "http://localhost:8010"
|
||||
model = "qwen3.6-27b-autoround"
|
||||
prompts = [
|
||||
("narrative", "Write a concise paragraph explaining transformer attention.", 96),
|
||||
("code", "Write a small Python function that returns the nth Fibonacci number.", 96),
|
||||
("reasoning", "A train leaves at 08:00 traveling 60 km/h. Another leaves at 09:00 traveling 90 km/h. When does the second catch up?", 96),
|
||||
]
|
||||
|
||||
for label, prompt, max_tokens in prompts:
|
||||
body = json.dumps({
|
||||
"model": model,
|
||||
"messages": [{"role": "user", "content": prompt}],
|
||||
"max_tokens": max_tokens,
|
||||
"temperature": 0.3,
|
||||
"stream": False,
|
||||
"chat_template_kwargs": {"enable_thinking": False},
|
||||
}).encode()
|
||||
req = urllib.request.Request(
|
||||
f"{url}/v1/chat/completions",
|
||||
data=body,
|
||||
headers={"Content-Type": "application/json"},
|
||||
)
|
||||
start = time.time()
|
||||
with urllib.request.urlopen(req, timeout=300) as response:
|
||||
data = json.load(response)
|
||||
wall = time.time() - start
|
||||
usage = data.get("usage") or {}
|
||||
tokens = usage.get("completion_tokens") or 0
|
||||
text = data["choices"][0]["message"].get("content") or ""
|
||||
if not text.strip():
|
||||
raise SystemExit(f"{label}: empty completion")
|
||||
tps = tokens / wall if wall > 0 else 0
|
||||
print(f"{label}: wall={wall:.2f}s completion_tokens={tokens} wall_TPS={tps:.2f}")
|
||||
PY
|
||||
|
||||
- name: Promote latest aliases
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
docker tag "${IMAGE_REF}" "${IMAGE_NAME}:latest"
|
||||
docker tag "${IMAGE_REF}" "${IMAGE_NAME}:nightly-stable"
|
||||
docker push "${IMAGE_NAME}:latest"
|
||||
docker push "${IMAGE_NAME}:nightly-stable"
|
||||
|
||||
- name: Cleanup canonical compose
|
||||
if: always()
|
||||
shell: bash
|
||||
run: |
|
||||
docker compose \
|
||||
-f "${CANONICAL_COMPOSE}" \
|
||||
-f "${COMPOSE_OVERRIDE}" \
|
||||
-p "${COMPOSE_PROJECT_NAME}" \
|
||||
down --remove-orphans || true
|
||||
docker buildx imagetools create \
|
||||
-t "${IMAGE_NAME}:latest" \
|
||||
-t "${IMAGE_NAME}:nightly-stable" \
|
||||
"${IMAGE_REF}"
|
||||
|
||||
retention:
|
||||
name: Retain four weeks of dated nightlies
|
||||
|
||||
+82
-61
@@ -26,6 +26,12 @@ All `Narr / Code TPS` rows come from `bash scripts/bench.sh`, which runs:
|
||||
>
|
||||
> Sampling: `temperature=0.6, top_p=0.95, top_k=20, presence_penalty=0.0, enable_thinking=false`. Three warmups + five measured runs per prompt. Mean wall TPS reported.
|
||||
|
||||
`PP tok/s` is prompt-processing throughput. For vLLM rows, `bench.sh` scrapes
|
||||
the most recent `Avg prompt throughput` lines from container logs. For
|
||||
llama.cpp or engines without that log shape, run `PP=1 bash scripts/bench.sh`
|
||||
to add a single long-prompt fallback probe that computes prompt tokens over
|
||||
TTFT. Existing rows show `—` until re-benched with v0.7.1+ tooling.
|
||||
|
||||
Cross-rig numbers are comparable because the prompt + sampling are pinned. Variations against your rig usually trace back to power caps, PCIe lane counts, or pin (vLLM image SHA / Genesis commit) — see [`scripts/report.sh`](scripts/report.sh) which captures all three.
|
||||
|
||||
## How to add a row for your rig
|
||||
@@ -57,66 +63,66 @@ Primary serving model. Hybrid Qwen3-Next architecture (DeltaNet GDN + standard a
|
||||
|
||||
> ⚠️ **Cliff 2b open on `long-text*` / `long-vision` (2026-05-05)** — Genesis v7.72.2's PN59 streaming-GDN orchestrator doesn't engage on the chunked-prefill path 24 GB single-card configs are forced to take. Single-prompt prefill at >~50K may OOM. Filed at [Sandermage/genesis-vllm-patches#22](https://github.com/Sandermage/genesis-vllm-patches/issues/22). **Safe single-card paths**: `llamacpp/default` (no Cliff 2b) or single-prompt context capped at <50K. **TP=2 paths escape the cliff** entirely (see Dual-card section).
|
||||
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | Peak VRAM | Date | Notes |
|
||||
|---|---|---|---:|---:|---:|---|---|
|
||||
| `minimal.yml` (`mem-util 0.95 max-model-len 65536`) | @noonghunna (1× 3090, x16, 350 W) | TQ3 | 64K | ~32 / ~33 | ~22.4 GB | 2026-05-03 | no MTP. [stiggy2k16](https://github.com/noonghunna/club-3090/issues/43) cross-rig data point — short-prompt vLLM-safe path when llama.cpp is too slow. |
|
||||
| `long-vision.yml` | @noonghunna (1× 3090) | TQ3 | 145K | 50 / 66 | ~23.0 GB | 2026-04-30 | vision + tools + thinking. mem-util 0.95. |
|
||||
| `long-text.yml` ⭐ | @noonghunna (1× 3090) | TQ3 | 180K | 50 / 67 | ~22.3 GB | 2026-04-30 | text-only (vision tower dropped). MTP n=3. mem-util 0.93. **Default for RAG / IDE agents below 25K accumulated ctx**. |
|
||||
| `long-text.yml` | @laurimyllari (1× **4090**, AMD Ryzen 7 7800X3D, 230W cap) | TQ3 | **90K** (forced by KV-pool fit on 4090 — see Notes) | **102.96 / 103.09** | ~23.7 GB | 2026-05-05 | **First 4090 single-card vLLM bench** on club-3090. **Required `max-model-len` drop from 180K→90K** at default mem-util 0.92 (KV cache budget on his 24 GB 4090 is tighter than the 3090s the compose was calibrated against — likely 4090 driver/desktop overhead consumes more idle VRAM). MTP n=3 active, AL 3.34-3.45 narr / per-pos accept 92-95% / 79-84% / 62-67%. CV 2.2%/2.2%. Verify-stress hit Cliff 2b OOM at long-vision 50 MiB (sidesteps via long-text). [Issue #71](https://github.com/noonghunna/club-3090/issues/71) + [disc #62](https://github.com/noonghunna/club-3090/discussions/62#discussioncomment-16821619). |
|
||||
| `long-text-no-mtp.yml` | @noonghunna (1× 3090) | TQ3 | 200K | TBD | ~21.0 GB | — | max-context single-shot, no MTP. Slow decode but biggest ctx window. |
|
||||
| `bounded-thinking.yml` | @noonghunna (1× 3090) | TQ3 | 180K | 50 / 66 | ~21.7 GB | 2026-05-04 | structured-CoT FSM in reasoning channel; **recommended grammar: DeepSeek scratchpad** (PLAN/NOTE×0-15/VERDICT). Phase 3 final: **93.9% HE+ / 66.0% LCB v6** (87.4% combined, +1 net vs the andthattoo G/A/E baseline). Andthattoo G/A/E grammar also works (94.5% HE+ / 62.0% LCB / 86.9% combined, ~4× tighter think budget — pass via `extra_body`). See [STRUCTURED_COT.md](docs/STRUCTURED_COT.md). |
|
||||
| `tools-text.yml` | @noonghunna (1× 3090) | fp8 | 75K | TBD | TBD | — | IDE-agent path that escapes the long-text Cliff 1 mech B leak (see [#16](https://github.com/noonghunna/club-3090/issues/16)). |
|
||||
| `dual-dflash.yml`-shape forced TP=1 (DFlash N=5, fp8 KV, mem-util 0.96, custom_all_reduce disabled) | @efschu (1× **RTX 5090** 32 GB, AMD Ryzen 9 5950X, Debian trixie, PCIe x8, 575 W cap) | fp8 | 49K (KV-fit at 0.96 mem-util) | **126.53 / 200.11** (decode 127.98 / 204.80) | 31.5 GB | 2026-05-07 | **First single-5090 DFlash data point** on club-3090. AutoRound INT4 weights + DFlash N=5 draft. CV 3.0%/2.0%. **Code TPS 200 is the highest single-card number measured on the matrix** — beats single-3090 (50/67 long-text) by ~3× on code, single-4090 (102/103 at 90K) by ~2× code. Trade is ctx ceiling: 49K vs 90K-180K on 24 GB cards, due to KV-pool fit at fp8 + 32 GB total VRAM. vLLM `nightly-01d4d1ad3` (post-v7.72.2 uplift). [Issue #93](https://github.com/noonghunna/club-3090/issues/93). |
|
||||
| `vllm/default` (single, MAX_MODEL_LEN=48000, mem-util 0.92, MTP n=3) | @ygafarov (1× 3090 via **oculink eGPU on PCIe x4**, AMD Ryzen AI MAX+ 395 / Strix Halo miniPC, CachyOS, 124 GB RAM, 290W cap) | TQ3 | 48K | **68.86 / 91.70** (decode 69.27 / 92.76) | 23.6 GB | 2026-05-09 | **Soak: ⚠ borderline** (VRAM grew 240 MiB > 200 MiB threshold, 3 turns >30s; 100% TPS retention + 0 errors + 0 silent-empty turns — x4-PCIe accretion + bus-latency under prefill, not a leak. Threshold may need an "eGPU bus class" allowance.) **First Strix-Halo-miniPC + oculink-eGPU bench** on club-3090. Single 3090 over PCIe x4 (oculink) instead of x8/x16 internal. CV 1.6%/2.7%. MTP AL 3.31, accept 76.9% (per-pos 0.918/0.772/0.616). `scheduler_reserve_full_isl=False`. Driver 595.71.05 (very new, CUDA 13.2). vLLM `nightly-01d4d1ad3`. [Issue #113](https://github.com/noonghunna/club-3090/issues/113). |
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | PP tok/s | Peak VRAM | Date | Notes |
|
||||
| --- | --- | --- | ---: | ---: | ---: | ---: | --- | --- |
|
||||
| `minimal.yml` (`mem-util 0.95 max-model-len 65536`) | @noonghunna (1× 3090, x16, 350 W) | TQ3 | 64K | ~32 / ~33 | — | ~22.4 GB | 2026-05-03 | no MTP. [stiggy2k16](https://github.com/noonghunna/club-3090/issues/43) cross-rig data point — short-prompt vLLM-safe path when llama.cpp is too slow. |
|
||||
| `long-vision.yml` | @noonghunna (1× 3090) | TQ3 | 145K | 50 / 66 | — | ~23.0 GB | 2026-04-30 | vision + tools + thinking. mem-util 0.95. |
|
||||
| `long-text.yml` ⭐ | @noonghunna (1× 3090) | TQ3 | 180K | 50 / 67 | — | ~22.3 GB | 2026-04-30 | text-only (vision tower dropped). MTP n=3. mem-util 0.93. **Default for RAG / IDE agents below 25K accumulated ctx**. |
|
||||
| `long-text.yml` | @laurimyllari (1× **4090**, AMD Ryzen 7 7800X3D, 230W cap) | TQ3 | **90K** (forced by KV-pool fit on 4090 — see Notes) | **102.96 / 103.09** | — | ~23.7 GB | 2026-05-05 | **First 4090 single-card vLLM bench** on club-3090. **Required `max-model-len` drop from 180K→90K** at default mem-util 0.92 (KV cache budget on his 24 GB 4090 is tighter than the 3090s the compose was calibrated against — likely 4090 driver/desktop overhead consumes more idle VRAM). MTP n=3 active, AL 3.34-3.45 narr / per-pos accept 92-95% / 79-84% / 62-67%. CV 2.2%/2.2%. Verify-stress hit Cliff 2b OOM at long-vision 50 MiB (sidesteps via long-text). [Issue #71](https://github.com/noonghunna/club-3090/issues/71) + [disc #62](https://github.com/noonghunna/club-3090/discussions/62#discussioncomment-16821619). |
|
||||
| `long-text-no-mtp.yml` | @noonghunna (1× 3090) | TQ3 | 200K | TBD | — | ~21.0 GB | — | max-context single-shot, no MTP. Slow decode but biggest ctx window. |
|
||||
| `bounded-thinking.yml` | @noonghunna (1× 3090) | TQ3 | 180K | 50 / 66 | — | ~21.7 GB | 2026-05-04 | structured-CoT FSM in reasoning channel; **recommended grammar: DeepSeek scratchpad** (PLAN/NOTE×0-15/VERDICT). Phase 3 final: **93.9% HE+ / 66.0% LCB v6** (87.4% combined, +1 net vs the andthattoo G/A/E baseline). Andthattoo G/A/E grammar also works (94.5% HE+ / 62.0% LCB / 86.9% combined, ~4× tighter think budget — pass via `extra_body`). See [STRUCTURED_COT.md](docs/STRUCTURED_COT.md). |
|
||||
| `tools-text.yml` | @noonghunna (1× 3090) | fp8 | 75K | TBD | — | TBD | — | IDE-agent path that escapes the long-text Cliff 1 mech B leak (see [#16](https://github.com/noonghunna/club-3090/issues/16)). |
|
||||
| `dual-dflash.yml`-shape forced TP=1 (DFlash N=5, fp8 KV, mem-util 0.96, custom_all_reduce disabled) | @efschu (1× **RTX 5090** 32 GB, AMD Ryzen 9 5950X, Debian trixie, PCIe x8, 575 W cap) | fp8 | 49K (KV-fit at 0.96 mem-util) | **126.53 / 200.11** (decode 127.98 / 204.80) | — | 31.5 GB | 2026-05-07 | **First single-5090 DFlash data point** on club-3090. AutoRound INT4 weights + DFlash N=5 draft. CV 3.0%/2.0%. **Code TPS 200 is the highest single-card number measured on the matrix** — beats single-3090 (50/67 long-text) by ~3× on code, single-4090 (102/103 at 90K) by ~2× code. Trade is ctx ceiling: 49K vs 90K-180K on 24 GB cards, due to KV-pool fit at fp8 + 32 GB total VRAM. vLLM `nightly-01d4d1ad3` (post-v7.72.2 uplift). [Issue #93](https://github.com/noonghunna/club-3090/issues/93). |
|
||||
| `vllm/default` (single, MAX_MODEL_LEN=48000, mem-util 0.92, MTP n=3) | @ygafarov (1× 3090 via **oculink eGPU on PCIe x4**, AMD Ryzen AI MAX+ 395 / Strix Halo miniPC, CachyOS, 124 GB RAM, 290W cap) | TQ3 | 48K | **68.86 / 91.70** (decode 69.27 / 92.76) | — | 23.6 GB | 2026-05-09 | **Soak: ⚠ borderline** (VRAM grew 240 MiB > 200 MiB threshold, 3 turns >30s; 100% TPS retention + 0 errors + 0 silent-empty turns — x4-PCIe accretion + bus-latency under prefill, not a leak. Threshold may need an "eGPU bus class" allowance.) **First Strix-Halo-miniPC + oculink-eGPU bench** on club-3090. Single 3090 over PCIe x4 (oculink) instead of x8/x16 internal. CV 1.6%/2.7%. MTP AL 3.31, accept 76.9% (per-pos 0.918/0.772/0.616). `scheduler_reserve_full_isl=False`. Driver 595.71.05 (very new, CUDA 13.2). vLLM `nightly-01d4d1ad3`. [Issue #113](https://github.com/noonghunna/club-3090/issues/113). |
|
||||
|
||||
### Single-card (1× RTX 3090) — llama.cpp
|
||||
|
||||
| Compose | Rig | Quant | Max ctx | Narr / Code TPS | Peak VRAM | Date | Notes |
|
||||
|---|---|---|---:|---:|---:|---|---|
|
||||
| `llamacpp/default` | @noonghunna (1× 3090) | Unsloth Q5_K_XL | 262K | 21 / 21 | ~20 GB | 2026-04-21 | bulletproof — different engine, different memory allocator, no Cliff 1 / Cliff 2. Slow decode but cliff-immune. |
|
||||
| `llamacpp/concurrent` | @noonghunna (1× 3090) | Unsloth Q5_K_XL | 262K | TBD | TBD | — | concurrent-serving variant. |
|
||||
| llama.cpp PR [#22673](https://github.com/ggml-org/llama.cpp/pull/22673) MTP, custom build (`Qwen3.6-27B-MTP-Q4_K_M-GGUF` + `--spec-type mtp --spec-draft-n-max 3`) | @efschu (**2× Tesla V100-SXM2-16GB**, Xeon Gold 6154, Debian 13, custom-built llama-server docker) | Q4_K_M MTP | 100K | **49.96 / 62.46** | 15.6 GB/card (15,596 MiB at 100K ctx) | 2026-05-06 | **First V100 (sm_70 Volta) cross-rig data on the matrix** — only non-3090/4090/5090 GPU class tested. vLLM blocked (V100=CC 7.0, vLLM needs ≥7.5); fell back to llama.cpp via am17an's PR #22673 with a custom-built docker. **All 7 stress checks PASS including 90K NIAH** (Cliff 2 territory). 2× cards via tensor split (`-sm tensor`). MTP n=3, accept rates not in log. ~80 W/card (V100 max 300 W). [Issue #80](https://github.com/noonghunna/club-3090/issues/80). |
|
||||
| llama.cpp PR [#22673](https://github.com/ggml-org/llama.cpp/pull/22673) MTP, host build (`havenoammo/Qwen3.6-27B-MTP-UD-GGUF` + `--spec-type mtp --spec-draft-n-max 3` + q4_0 KV) | @lamentofhighborne (1× RTX 3090, PCIe x8, 350W) | UD-Q4_K_XL + Q8_0 MTP head | **131K** | **47.12 / 60.42** | ~23.1 GiB | 2026-05-07 | **First 1× 3090 llama.cpp MTP data point** on Qwen3.6-27B. Decode 47.60 / 61.71 TPS, TTFT 212 / 194 ms. **`verify-full-mtp.sh` PASS 8/8** (locally-adapted), **`verify-stress-mtp.sh` PASS 7/7 including 91K needle at 131K ctx** — pushes the documented llama.cpp MTP ctx ceiling from ~64-80K (q8_0 KV) to 131K (q4_0 KV). MTP acceptance 78.7%; recurrent 65-layer bug from froggeric's earlier MTP GGUF did **NOT** reproduce on havenoammo's UD GGUF. Native host build (no Docker), surfaced engine-coupling shortcomings in our verify/soak harness — see [Issue #85](https://github.com/noonghunna/club-3090/issues/85). |
|
||||
| llama.cpp PR [#22673](https://github.com/ggml-org/llama.cpp/pull/22673) MTP, host build (`froggeric/Qwen3.6-27B-MTP-GGUF` + `--spec-type mtp --spec-draft-n-max 3` + q4_0 KV) | @lamentofhighborne (1× RTX 3090, PCIe x8, 350 W) | Q4_K_M MTP | **164K** | **47.49 / 55.09** | ~22.2 GiB | 2026-05-07 | **Second 1× 3090 llama.cpp MTP data point on same rig** — froggeric's Q4_K_M MTP GGUF vs havenoammo's UD-Q4_K_XL above. Decode 47.91 / 55.81 TPS, TTFT 96 / 98 ms. `verify-full-mtp.sh` PASS 8/8, `verify-stress-mtp.sh` PASS 7/7 incl. 91K needle at 164K ctx. Functional MTP acceptance **86.7%**; canonical acceptance 55.3% narr / 71.2% code. **Ctx-fit ladder**: 262K OOMed MTP, 229K served without MTP, 196K initialized MTP but daemon died at 90K stress; 164K was the stable stress-passing ceiling on this rig. **Beats havenoammo on narr (47.49 vs 47.12, +0.8%) and ctx ceiling (164K vs 131K) but trails on code (55.09 vs 60.42, −9%)**. Manual long-context needles also passed at **120K** (39.39 decode TPS, 81% MTP accept) and **150K** (35.44 decode TPS, 80% MTP accept). MTP+vision incompat (per froggeric's model card); separate no-MTP+vision path passed 65K and 150K. [Issue #94](https://github.com/noonghunna/club-3090/issues/94). |
|
||||
| Compose | Rig | Quant | Max ctx | Narr / Code TPS | PP tok/s | Peak VRAM | Date | Notes |
|
||||
| --- | --- | --- | ---: | ---: | ---: | ---: | --- | --- |
|
||||
| `llamacpp/default` | @noonghunna (1× 3090) | Unsloth Q5_K_XL | 262K | 21 / 21 | — | ~20 GB | 2026-04-21 | bulletproof — different engine, different memory allocator, no Cliff 1 / Cliff 2. Slow decode but cliff-immune. |
|
||||
| `llamacpp/concurrent` | @noonghunna (1× 3090) | Unsloth Q5_K_XL | 262K | TBD | — | TBD | — | concurrent-serving variant. |
|
||||
| llama.cpp PR [#22673](https://github.com/ggml-org/llama.cpp/pull/22673) MTP, custom build (`Qwen3.6-27B-MTP-Q4_K_M-GGUF` + `--spec-type mtp --spec-draft-n-max 3`) | @efschu (**2× Tesla V100-SXM2-16GB**, Xeon Gold 6154, Debian 13, custom-built llama-server docker) | Q4_K_M MTP | 100K | **49.96 / 62.46** | — | 15.6 GB/card (15,596 MiB at 100K ctx) | 2026-05-06 | **First V100 (sm_70 Volta) cross-rig data on the matrix** — only non-3090/4090/5090 GPU class tested. vLLM blocked (V100=CC 7.0, vLLM needs ≥7.5); fell back to llama.cpp via am17an's PR #22673 with a custom-built docker. **All 7 stress checks PASS including 90K NIAH** (Cliff 2 territory). 2× cards via tensor split (`-sm tensor`). MTP n=3, accept rates not in log. ~80 W/card (V100 max 300 W). [Issue #80](https://github.com/noonghunna/club-3090/issues/80). |
|
||||
| llama.cpp PR [#22673](https://github.com/ggml-org/llama.cpp/pull/22673) MTP, host build (`havenoammo/Qwen3.6-27B-MTP-UD-GGUF` + `--spec-type mtp --spec-draft-n-max 3` + q4_0 KV) | @lamentofhighborne (1× RTX 3090, PCIe x8, 350W) | UD-Q4_K_XL + Q8_0 MTP head | **131K** | **47.12 / 60.42** | — | ~23.1 GiB | 2026-05-07 | **First 1× 3090 llama.cpp MTP data point** on Qwen3.6-27B. Decode 47.60 / 61.71 TPS, TTFT 212 / 194 ms. **`verify-full-mtp.sh` PASS 8/8** (locally-adapted), **`verify-stress-mtp.sh` PASS 7/7 including 91K needle at 131K ctx** — pushes the documented llama.cpp MTP ctx ceiling from ~64-80K (q8_0 KV) to 131K (q4_0 KV). MTP acceptance 78.7%; recurrent 65-layer bug from froggeric's earlier MTP GGUF did **NOT** reproduce on havenoammo's UD GGUF. Native host build (no Docker), surfaced engine-coupling shortcomings in our verify/soak harness — see [Issue #85](https://github.com/noonghunna/club-3090/issues/85). |
|
||||
| llama.cpp PR [#22673](https://github.com/ggml-org/llama.cpp/pull/22673) MTP, host build (`froggeric/Qwen3.6-27B-MTP-GGUF` + `--spec-type mtp --spec-draft-n-max 3` + q4_0 KV) | @lamentofhighborne (1× RTX 3090, PCIe x8, 350 W) | Q4_K_M MTP | **164K** | **47.49 / 55.09** | — | ~22.2 GiB | 2026-05-07 | **Second 1× 3090 llama.cpp MTP data point on same rig** — froggeric's Q4_K_M MTP GGUF vs havenoammo's UD-Q4_K_XL above. Decode 47.91 / 55.81 TPS, TTFT 96 / 98 ms. `verify-full-mtp.sh` PASS 8/8, `verify-stress-mtp.sh` PASS 7/7 incl. 91K needle at 164K ctx. Functional MTP acceptance **86.7%**; canonical acceptance 55.3% narr / 71.2% code. **Ctx-fit ladder**: 262K OOMed MTP, 229K served without MTP, 196K initialized MTP but daemon died at 90K stress; 164K was the stable stress-passing ceiling on this rig. **Beats havenoammo on narr (47.49 vs 47.12, +0.8%) and ctx ceiling (164K vs 131K) but trails on code (55.09 vs 60.42, −9%)**. Manual long-context needles also passed at **120K** (39.39 decode TPS, 81% MTP accept) and **150K** (35.44 decode TPS, 80% MTP accept). MTP+vision incompat (per froggeric's model card); separate no-MTP+vision path passed 65K and 150K. [Issue #94](https://github.com/noonghunna/club-3090/issues/94). |
|
||||
|
||||
### Dual-card (2× RTX 3090, TP=2)
|
||||
|
||||
> NVLink auto-detection: dual-card composes now detect NVLink presence automatically. The `dual-nvlink*.yml` files are deprecated stubs that extend the unified compose with `NVLINK_MODE=force_on`. All NVLink bench rows below were measured with NVLink enabled (either via auto-detection or the deprecated stub). PCIe rows used `NCCL_P2P_DISABLE=1`.
|
||||
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | Peak VRAM | Date | Notes |
|
||||
|---|---|---|---:|---:|---:|---|---|
|
||||
| `dual.yml` ⭐ | @noonghunna (2× 3090 PCIe, no NVLink) | fp8 | 262K (237K single-prompt verified) | 69 / 89 | ~23.6 GB | 2026-04-29 | tested 2-card baseline. fp8 KV, 2 streams, full feature set. **PASSES v2 continuous soak** (Cliff 2b clean). |
|
||||
| `dual-turbo.yml` | @noonghunna (2× 3090 PCIe) | TQ3 | 262K | 58 / 76 per-stream (**269 TPS aggregate at 4 streams**) | ~19.8 GB | 2026-04-29 | TQ3 KV — 4.67× concurrency for multi-tenant agent workloads. |
|
||||
| `dual-turbo.yml` ⭐ | @noonghunna (2× 3090 PCIe) | TQ3 | 262K | **81.21 / 108.20** single-stream | **20.0 GB** | 2026-05-05 | **v7.72.2 uplift**: Genesis pin `7b9fd319` + vLLM `01d4d1ad3` (Sander's PROD pin). 6 redundant local sidecars dropped (PN35/PN30/PN25/P78/PN34 supersede). 5 measured runs each, CV 2.3%/0.9%. AL 3.46. **VRAM −2.1 GB/card vs v7.69 baseline** (PN35 native + PN59 fold value). All 8/8 verify-full checks pass. |
|
||||
| `dual-dflash.yml` | @noonghunna (2× 3090 PCIe) | fp8 | 185K | 82 / **125** | ~23.6 GB | 2026-04-29 | DFlash N=5 + 1.75 GB draft / card. AL ~4.4. Fastest 2-card short-prompt code path. |
|
||||
| `dual-dflash.yml` | @apriori (2× 3090 + EPYC 7302P, Arch Linux, 230 W cap, NODE topology, no NVLink) | fp8 | 185K | **78.44 / 122.71** | ~24.0 GB | 2026-05-05 | **First EPYC + Arch cross-rig data on `dual-dflash`** — matches @noonghunna baseline within run-to-run CV (78/127 reference, narr drift +0.4 / code −3.4%). **PASSES continuous soak** (0 MiB VRAM growth, 0 errors, 0/25 silent-empty, 100% TPS retention) — first independent confirmation `dual-dflash` is Cliff 2b clean cross-rig. 3 turns >30s TTFT warning (informational). [Discussion #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16819551). |
|
||||
| `dual-dflash-noviz.yml` | @noonghunna (2× 3090 PCIe) | fp8 | 200K | 78 / **127** | ~23.8 GB | 2026-04-29 | DFlash + no vision tower. +15K ctx vs `dual-dflash`. |
|
||||
| `dual-dflash-noviz.yml` | @snoby (2× **4090** PCIe — 5-GPU rig, GPUs 2,3, no NVLink, [#46](https://github.com/noonghunna/club-3090/issues/46)) | fp8 | **180K** | 92.55 / **148.99** | ~21.8 GB | 2026-05-04 | First non-3090 cross-rig data. **Required `max-model-len` drop from 200K→180K** vs 3090 baseline (boot OOM at 200K) — 4090 ctx-ceiling gotcha pending investigation. +17% TPS lift vs same compose on 3090 (78→92.55 narr / 127→148.99 code). |
|
||||
| `dual-nvlink.yml` | @JusefPol (2× 3090 PCIe x8 + **NVLink 4× bonded**, i7-11700K, 365 W/card) | fp8 | 262K | **108.81 / 138.55** | ~23.7 GB | 2026-05-04 | First NVLink cross-rig data. **+58% narr / +56% code TPS vs `dual.yml` PCIe-only baseline (69 / 89)** — NVLink reduces the per-token NCCL allreduce latency floor; compounds at multi-stream. verify-stress 7/7 PASS incl. 91K needle. **PASSES v2 continuous soak** (5 sessions × 5 turns, 0 MiB growth, 100% TPS retention). MTP n=3, 65–98% per-position accept. PR [#31](https://github.com/noonghunna/club-3090/pull/31). |
|
||||
| `dual-nvlink-turbo.yml` ⭐ | @danbedford (2× 3090 NVLink, 230W cap) | TQ3 | 262K | **102.34 / 133.98** | ~22.3 GB | 2026-05-05 | **v7.72.2-rebench** (image `nightly-01d4d1ad3`). 4-stream TurboQuant KV + NVLink. **+11% narr / +12% code vs same-rig PCIe `dual-turbo` (#73 below)** — controlled A/B on identical hardware, only `NCCL_P2P_LEVEL` differs. Custom all-reduce ENABLED (disabled on PCIe). CV 3.1% narr / 1.8% code. PR [#56](https://github.com/noonghunna/club-3090/pull/56) + [Issue #69](https://github.com/noonghunna/club-3090/issues/69). |
|
||||
| `dual.yml` | @danbedford (2× 3090 NVLink-cable-attached, run as PCIe via `NCCL_P2P_DISABLE=1`, 230W cap) | fp8 | 262K | **89.24 / 114.57** | ~23.7 GB | 2026-05-06 | **First controlled PCIe-vs-NVLink A/B on same rig** — pair with `dual-nvlink.yml` row immediately above. **+15% narr / +15% code lift from NVLink** (#74 102/132 vs this 89/115). CV 3.8%/2.5%. **Note: this corrects the "+58% narr / +56% code" claim from JusefPol's row** — that comparison conflated NVLink lift with v7.72.2 lift (his baseline was 2026-04-29 dual.yml at 69/89 on the older image). On a strictly v7.72.2-controlled comparison NVLink adds ~15%, not ~58%. [Issue #77](https://github.com/noonghunna/club-3090/issues/77). |
|
||||
| `dual-turbo.yml` | @danbedford (2× 3090 NVLink-cable-attached, run as PCIe via `NCCL_P2P_DISABLE=1`, 230W cap) | TQ3 | 262K | 91.58 / 120.00 | ~22.0 GB | 2026-05-06 | Companion to `dual-nvlink-turbo` row above for the controlled A/B. NVLink lift on TQ3 path: **+11% / +12%**. CV 3.2%/1.9%. [Issue #73](https://github.com/noonghunna/club-3090/issues/73). |
|
||||
| `dual-nvlink.yml` | @danbedford (2× 3090 NVLink, 230W cap) | fp8 | 262K | **102.09 / 131.59** | ~24.0 GB | 2026-05-06 | Second cross-rig data on `dual-nvlink.yml` (vs JusefPol's earlier 108.81/138.55). Lower than JusefPol partly explained by his lower power cap (365 W/card vs 230) — on memory-bandwidth-bound decode, 2 GB/card more thermal headroom doesn't compound much, so close-but-lower at half the wattage is consistent. CV 2.6%/1.4%. [Issue #74](https://github.com/noonghunna/club-3090/issues/74). |
|
||||
| `dual-dflash.yml` | @danbedford (2× 3090 PCIe NVLink-cable-attached but `NCCL_P2P_DISABLE=1`, 230W cap) | FP16 | 185K | 86.62 / **141.02** | ~24.0 GB | 2026-05-06 | Third cross-rig DFlash data point (after @noonghunna 82/125 + @lolren 87/142). **Code TPS 141 ties lolren's 142** as the highest measured on club-3090. CV 2.4%/5.0%. [Issue #75](https://github.com/noonghunna/club-3090/issues/75). |
|
||||
| `dual-dflash-noviz.yml` | @danbedford (2× 3090 PCIe NVLink-cable-attached but `NCCL_P2P_DISABLE=1`, 230W cap) | FP16 | 200K | 88.31 / **142.79** | ~23.9 GB | 2026-05-06 | DFlash + no vision tower. Beats @noonghunna baseline 78/127 (+13%/+12%). CV 2.3%/2.9%. [Issue #76](https://github.com/noonghunna/club-3090/issues/76). |
|
||||
| `dual-nvlink-dflash.yml` ⭐ NEW | @danbedford (2× 3090 NVLink, 230W cap, i9-11900KF) | FP16 | 185K | **101.55 / 163.33** | 24.06 GB/card | 2026-05-07 | **First NVLink-enabled DFlash row.** Mirrors `dual-dflash.yml` shape but enables NCCL P2P over NVLink + custom_all_reduce. **+17% narr / +16% code over his own PCIe `dual-dflash` row above** (86.62 / 141.02 — same rig with `NCCL_P2P_DISABLE=1`). Decode 102.43 / 166.54 TPS, CV 1.8%/1.9%. **PASSES continuous soak** (0 errors, 0 silent-empty, 0 MiB growth, 100% TPS retention, p50 66.71). verify-full 8/8 + verify-stress 7/7 incl. 91K Cliff 2 needle. PR [#92](https://github.com/noonghunna/club-3090/pull/92). |
|
||||
| `dual-nvlink-dflash-noviz.yml` ⭐ NEW | @danbedford (2× 3090 NVLink, 230W cap) | FP16 | **188K** | **103.24 / 167.45** | ~23.97 GB/card | 2026-05-07 | **NVLink + DFlash + no vision** — pushes the with-vision 185K ctx ceiling to **188K** by dropping MoonViT (~0.78 GB freed). Empirically determined: 189K had only 1/3 success rate (flaky on freshly rebooted system), 188K is the stable ceiling. **+17% narr / +17% code over his own PCIe `dual-dflash-noviz` row above** (88.31 / 142.79). Decode 104.07 / 171.01 TPS, CV 2.2%/3.6%. **PASSES continuous soak** (p50 66.75, 100% retention). verify-full 8/8 + verify-stress 7/7. PR [#96](https://github.com/noonghunna/club-3090/pull/96). |
|
||||
| `dual.yml`-shape **+ patched P2P drivers** (no NVLink hardware) | @aaronlockhartdev (2× 3090 PCIe x16, EPYC 7F52, Arch Linux, custom Dockerfile via [Sam McLeod's guide](https://smcleod.net/2026/02/patching-nvidias-driver-and-vllm-to-enable-p2p-on-consumer-gpus/) — patched `aikitoria/open-gpu-kernel-modules` + vLLM `cuda.py` `return True` patch) | fp8 | 262K | **93 / 125** | n/a | 2026-05-07 | **First patched-driver P2P cross-rig data point** — answers the question raised in [disc #70](https://github.com/noonghunna/club-3090/discussions/70). Same-rig controlled A/B: unpatched baseline 91 narr / 114 code → patched P2P 93 / 125 = **+2% narr / +9% code**. Compared to NVLink hardware lift (+15% / +15% per @danbedford's controlled A/B): patched P2P captures **~60% of NVLink's code gain but ~13% of NVLink's narr gain** — code workloads (spec-decode K+1 verify is heavily cross-card matmul) benefit more from cross-card bandwidth than narr decode (more sequential per-token). For ~95% of dual-3090 owners without NVLink, the trade is small TPS lift vs custom kernel module + DKMS maintenance burden. [Issue #91](https://github.com/noonghunna/club-3090/issues/91). |
|
||||
| `dual-dflash-noviz.yml`-shape **+ patched P2P drivers** (no NVLink hardware, custom_all_reduce ENABLED) | @aaronlockhartdev (2× 3090 PCIe x16, EPYC 7F52, Arch Linux, patched kernel module + `NCCL_P2P_LEVEL=PHB`) | fp8 | 200K | **100.47 / 160.15** (decode 101.53 / 164.44) | ~22.2 GB/card | 2026-05-07 | **Second patched-P2P cross-rig data point** — extends [#91 dual.yml result](https://github.com/noonghunna/club-3090/issues/91) to the DFlash + no-vision path. Same-rig controlled A/B: unpatched baseline 82.55 narr / 134.45 code → patched P2P 100.47 / 160.15 = **+22% narr / +19% code**. **Significantly larger lift than `dual.yml`-shape** (+22%/+19% here vs +2%/+9% on `dual.yml`) — DFlash's K+1 cross-card verify pattern stresses peer-bandwidth more than fp8-only `dual.yml`. **Important methodology update**: `NCCL_P2P_LEVEL=PHB` alone with the default vLLM image produced the same lift as the full vLLM `cuda.py` patch — **the in-container vLLM source patch is unnecessary**, only the kernel module patch matters. CV 4.6%/2.4%. custom_all_reduce ENABLED (vs disabled on the `dual.yml` row). [Issue #95](https://github.com/noonghunna/club-3090/issues/95) + [disc #70](https://github.com/noonghunna/club-3090/discussions/70). |
|
||||
| `carnice-bf16mtp.yml` | @noonghunna (2× 3090 PCIe, no NVLink) | fp8 | 262K | **72** / **80** | ~22.25 GB | 2026-05-04 | **Carnice-V2-27B (Hermes agentic fine-tune) + BF16 MTP overlay**. Full 262K context, 2 streams. 71.75 narr / 80.35 code wall TPS (n=5 each, CV ~11%), MTP AL 3.02-3.14, TTFT 141ms. Patched chat template for Hermes JSON tool calls. verify-full 7/8 PASS. soak PASS. |
|
||||
| `dual.yml` ⭐ | @lolren (2× 3090 PCIe + Ryzen 9 5950X, **250W/card cap**) | fp8 | 262K | **89.78 / 117.60** | ~22.3 GB | 2026-05-05 | **First cross-rig data on the v7.72.2 uplift** (image `nightly-01d4d1ad3`, post-PR #59). +30% narr / +32% code over @noonghunna 2026-04-29 baseline (69/89 on older image) — confirms the v7.72.2 dividend cross-rig. CV 3.3%/2.0%. MTP AL ~3.5, per-pos accept 94/84/72%. [Disc #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16820303). |
|
||||
| `dual-dflash.yml` | @lolren (2× 3090 PCIe + Ryzen 9 5950X, 250W cap) | FP16 | 185K | 87.10 / **142.0** | ~22.1 GB | 2026-05-05 | Older image `nightly-7a1eb8ac2`. +6% narr / +14% code over @noonghunna baseline (82/125) — likely Ryzen 5950X advantage on prefill. DFlash AL ~4.5, per-pos accept 93/81/68/56/48%, avg accept 69%. [Disc #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16820303). |
|
||||
| `bounded-thinking.yml` | @lolren (2× 3090 PCIe + Ryzen 9 5950X, 250W cap, **MTP-disabled-suspected**) | TQ3 | 180K | 64.86 / 64.96 (CV **0.1%**) | ~22.3 GB | 2026-05-05 | **Anomaly:** lolren reports "no spec-decode" on this run despite `bounded-thinking.yml` shipping `--speculative-config mtp n=3` by default. Near-identical narr=code TPS + extreme CV stability (0.1%) suggests MTP was inactive — likely because his image was older `nightly-7a1eb8ac2` (pre-v7.72.2 + pre-PN35). Re-test on `nightly-01d4d1ad3` should restore MTP path → expect ~50/66 narr/code with normal CV. Tracked. [Disc #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16820303). |
|
||||
| `dual.yml` | @JDWarner (**Mixed RTX A5000 + RTX 3090**, both **Razer Core X eGPU enclosures over Thunderbolt 3**, Intel NUC11TNH i5-1135G7, **16 GB RAM**, headless, A5000=230W cap / 3090=290W cap, PCIe **x4 Gen 3** per card) | fp8 | 262K | **56.83 / 72.47** (soak p50 93.09) | ~23.6 GB/card | 2026-05-09 | **Soak: ✓ PASS** (5×5, 0 errors, 0 silent-empty, 100% TPS retention, 0 MiB growth). **First TB3 dual-eGPU + mixed-arch cross-rig data**. The setup that "shouldn't work": each card on a separate TB3 controller → ~3.94 GB/s effective per card vs ~32 GB/s on PCIe x16 Gen 4 (~8× cut), mixed Ampere SKUs (workstation A5000 + consumer 3090 with different mem bandwidth + clocks), 16 GB system RAM total. **Result: matches `dual.yml` PCIe x16 baseline within run-to-run noise** — confirms decode on Qwen3.6-27B is per-card-bandwidth bound, cross-card NCCL allreduce is small enough that even an 8× link cut doesn't dominate. Extends @aaronlockhartdev's #91/#95 finding (patched-P2P only +2%/+9% on `dual.yml`) in the opposite direction: even with 8× *less* cross-card bandwidth, decode holds. MTP AL 3.39-3.52, per-pos accept 0.93/0.83/0.70 (89% avg). verify-full + verify-stress all PASS. Genesis pin `7b9fd319` (v7.72.2). [Issue #107](https://github.com/noonghunna/club-3090/issues/107). |
|
||||
| `dual/docker-compose.yml` (default) | @ygafarov (**3090 via USB4 eGPU dock + 5070 Ti via OCuLink** — heterogeneous Ampere + Blackwell consumer dual-eGPU, AMD Ryzen AI MAX+ 395 / Strix Halo miniPC, CachyOS, 123 GB RAM, 290 W cap both cards, PCIe **x4** per card — USB4 ≈ 3.94 GB/s, OCuLink ≈ 7.88 GB/s) | fp8 | 200K | **65.10 / 85.81** | 17.1 / 15.7 GB | 2026-05-12 | **First heterogeneous Ampere + Blackwell consumer dual-eGPU on the matrix.** TP=2 bound by the slower USB4 link in allreduce + sm_86 kernels (5070 Ti spends back-half of step waiting — 91% util but only 125 W out of 290 W cap). KV pool 200K @ 1.00× concurrency — VRAM cap from the 5070 Ti's 16 GiB (model takes 13.8 GiB/card → only ~2.2 GiB left for KV on the smaller card). verify-stress 7/7 incl. **91K needle recall** (Cliff 2 clean). Soak ⚠ borderline (360 MiB > 200 MiB threshold — same eGPU-bus accretion as ygafarov's own #113 single-card row above at 240 MiB; 100% TPS retention + 0 silent-empty + 0 errors so not a leak). MTP AL 3.50, per-pos accept 0.94/0.86/0.70. CV 4.5%/1.8%. **Slower than ygafarov's own single-3090 #113 row** (68.86/91.70 at 48K) — on this rig the single-card path is recommended; the 5070 Ti adds VRAM cap pain without TPS gain. Driver 595.71.05, vLLM `nightly-1acd67a79`, no Genesis (Blackwell consumer not on allowlist). [Issue #120](https://github.com/noonghunna/club-3090/issues/120). |
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | PP tok/s | Peak VRAM | Date | Notes |
|
||||
| --- | --- | --- | ---: | ---: | ---: | ---: | --- | --- |
|
||||
| `dual.yml` ⭐ | @noonghunna (2× 3090 PCIe, no NVLink) | fp8 | 262K (237K single-prompt verified) | 69 / 89 | — | ~23.6 GB | 2026-04-29 | tested 2-card baseline. fp8 KV, 2 streams, full feature set. **PASSES v2 continuous soak** (Cliff 2b clean). |
|
||||
| `dual-turbo.yml` | @noonghunna (2× 3090 PCIe) | TQ3 | 262K | 58 / 76 per-stream (**269 TPS aggregate at 4 streams**) | — | ~19.8 GB | 2026-04-29 | TQ3 KV — 4.67× concurrency for multi-tenant agent workloads. |
|
||||
| `dual-turbo.yml` ⭐ | @noonghunna (2× 3090 PCIe) | TQ3 | 262K | **81.21 / 108.20** single-stream | — | **20.0 GB** | 2026-05-05 | **v7.72.2 uplift**: Genesis pin `7b9fd319` + vLLM `01d4d1ad3` (Sander's PROD pin). 6 redundant local sidecars dropped (PN35/PN30/PN25/P78/PN34 supersede). 5 measured runs each, CV 2.3%/0.9%. AL 3.46. **VRAM −2.1 GB/card vs v7.69 baseline** (PN35 native + PN59 fold value). All 8/8 verify-full checks pass. |
|
||||
| `dual-dflash.yml` | @noonghunna (2× 3090 PCIe) | fp8 | 185K | 82 / **125** | — | ~23.6 GB | 2026-04-29 | DFlash N=5 + 1.75 GB draft / card. AL ~4.4. Fastest 2-card short-prompt code path. |
|
||||
| `dual-dflash.yml` | @apriori (2× 3090 + EPYC 7302P, Arch Linux, 230 W cap, NODE topology, no NVLink) | fp8 | 185K | **78.44 / 122.71** | — | ~24.0 GB | 2026-05-05 | **First EPYC + Arch cross-rig data on `dual-dflash`** — matches @noonghunna baseline within run-to-run CV (78/127 reference, narr drift +0.4 / code −3.4%). **PASSES continuous soak** (0 MiB VRAM growth, 0 errors, 0/25 silent-empty, 100% TPS retention) — first independent confirmation `dual-dflash` is Cliff 2b clean cross-rig. 3 turns >30s TTFT warning (informational). [Discussion #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16819551). |
|
||||
| `dual-dflash-noviz.yml` | @noonghunna (2× 3090 PCIe) | fp8 | 200K | 78 / **127** | — | ~23.8 GB | 2026-04-29 | DFlash + no vision tower. +15K ctx vs `dual-dflash`. |
|
||||
| `dual-dflash-noviz.yml` | @snoby (2× **4090** PCIe — 5-GPU rig, GPUs 2,3, no NVLink, [#46](https://github.com/noonghunna/club-3090/issues/46)) | fp8 | **180K** | 92.55 / **148.99** | — | ~21.8 GB | 2026-05-04 | First non-3090 cross-rig data. **Required `max-model-len` drop from 200K→180K** vs 3090 baseline (boot OOM at 200K) — 4090 ctx-ceiling gotcha pending investigation. +17% TPS lift vs same compose on 3090 (78→92.55 narr / 127→148.99 code). |
|
||||
| `dual-nvlink.yml` | @JusefPol (2× 3090 PCIe x8 + **NVLink 4× bonded**, i7-11700K, 365 W/card) | fp8 | 262K | **108.81 / 138.55** | — | ~23.7 GB | 2026-05-04 | First NVLink cross-rig data. **+58% narr / +56% code TPS vs `dual.yml` PCIe-only baseline (69 / 89)** — NVLink reduces the per-token NCCL allreduce latency floor; compounds at multi-stream. verify-stress 7/7 PASS incl. 91K needle. **PASSES v2 continuous soak** (5 sessions × 5 turns, 0 MiB growth, 100% TPS retention). MTP n=3, 65–98% per-position accept. PR [#31](https://github.com/noonghunna/club-3090/pull/31). |
|
||||
| `dual-nvlink-turbo.yml` ⭐ | @danbedford (2× 3090 NVLink, 230W cap) | TQ3 | 262K | **102.34 / 133.98** | — | ~22.3 GB | 2026-05-05 | **v7.72.2-rebench** (image `nightly-01d4d1ad3`). 4-stream TurboQuant KV + NVLink. **+11% narr / +12% code vs same-rig PCIe `dual-turbo` (#73 below)** — controlled A/B on identical hardware, only `NCCL_P2P_LEVEL` differs. Custom all-reduce ENABLED (disabled on PCIe). CV 3.1% narr / 1.8% code. PR [#56](https://github.com/noonghunna/club-3090/pull/56) + [Issue #69](https://github.com/noonghunna/club-3090/issues/69). |
|
||||
| `dual.yml` | @danbedford (2× 3090 NVLink-cable-attached, run as PCIe via `NCCL_P2P_DISABLE=1`, 230W cap) | fp8 | 262K | **89.24 / 114.57** | — | ~23.7 GB | 2026-05-06 | **First controlled PCIe-vs-NVLink A/B on same rig** — pair with `dual-nvlink.yml` row immediately above. **+15% narr / +15% code lift from NVLink** (#74 102/132 vs this 89/115). CV 3.8%/2.5%. **Note: this corrects the "+58% narr / +56% code" claim from JusefPol's row** — that comparison conflated NVLink lift with v7.72.2 lift (his baseline was 2026-04-29 dual.yml at 69/89 on the older image). On a strictly v7.72.2-controlled comparison NVLink adds ~15%, not ~58%. [Issue #77](https://github.com/noonghunna/club-3090/issues/77). |
|
||||
| `dual-turbo.yml` | @danbedford (2× 3090 NVLink-cable-attached, run as PCIe via `NCCL_P2P_DISABLE=1`, 230W cap) | TQ3 | 262K | 91.58 / 120.00 | — | ~22.0 GB | 2026-05-06 | Companion to `dual-nvlink-turbo` row above for the controlled A/B. NVLink lift on TQ3 path: **+11% / +12%**. CV 3.2%/1.9%. [Issue #73](https://github.com/noonghunna/club-3090/issues/73). |
|
||||
| `dual-nvlink.yml` | @danbedford (2× 3090 NVLink, 230W cap) | fp8 | 262K | **102.09 / 131.59** | — | ~24.0 GB | 2026-05-06 | Second cross-rig data on `dual-nvlink.yml` (vs JusefPol's earlier 108.81/138.55). Lower than JusefPol partly explained by his lower power cap (365 W/card vs 230) — on memory-bandwidth-bound decode, 2 GB/card more thermal headroom doesn't compound much, so close-but-lower at half the wattage is consistent. CV 2.6%/1.4%. [Issue #74](https://github.com/noonghunna/club-3090/issues/74). |
|
||||
| `dual-dflash.yml` | @danbedford (2× 3090 PCIe NVLink-cable-attached but `NCCL_P2P_DISABLE=1`, 230W cap) | FP16 | 185K | 86.62 / **141.02** | — | ~24.0 GB | 2026-05-06 | Third cross-rig DFlash data point (after @noonghunna 82/125 + @lolren 87/142). **Code TPS 141 ties lolren's 142** as the highest measured on club-3090. CV 2.4%/5.0%. [Issue #75](https://github.com/noonghunna/club-3090/issues/75). |
|
||||
| `dual-dflash-noviz.yml` | @danbedford (2× 3090 PCIe NVLink-cable-attached but `NCCL_P2P_DISABLE=1`, 230W cap) | FP16 | 200K | 88.31 / **142.79** | — | ~23.9 GB | 2026-05-06 | DFlash + no vision tower. Beats @noonghunna baseline 78/127 (+13%/+12%). CV 2.3%/2.9%. [Issue #76](https://github.com/noonghunna/club-3090/issues/76). |
|
||||
| `dual-nvlink-dflash.yml` ⭐ NEW | @danbedford (2× 3090 NVLink, 230W cap, i9-11900KF) | FP16 | 185K | **101.55 / 163.33** | — | 24.06 GB/card | 2026-05-07 | **First NVLink-enabled DFlash row.** Mirrors `dual-dflash.yml` shape but enables NCCL P2P over NVLink + custom_all_reduce. **+17% narr / +16% code over his own PCIe `dual-dflash` row above** (86.62 / 141.02 — same rig with `NCCL_P2P_DISABLE=1`). Decode 102.43 / 166.54 TPS, CV 1.8%/1.9%. **PASSES continuous soak** (0 errors, 0 silent-empty, 0 MiB growth, 100% TPS retention, p50 66.71). verify-full 8/8 + verify-stress 7/7 incl. 91K Cliff 2 needle. PR [#92](https://github.com/noonghunna/club-3090/pull/92). |
|
||||
| `dual-nvlink-dflash-noviz.yml` ⭐ NEW | @danbedford (2× 3090 NVLink, 230W cap) | FP16 | **188K** | **103.24 / 167.45** | — | ~23.97 GB/card | 2026-05-07 | **NVLink + DFlash + no vision** — pushes the with-vision 185K ctx ceiling to **188K** by dropping MoonViT (~0.78 GB freed). Empirically determined: 189K had only 1/3 success rate (flaky on freshly rebooted system), 188K is the stable ceiling. **+17% narr / +17% code over his own PCIe `dual-dflash-noviz` row above** (88.31 / 142.79). Decode 104.07 / 171.01 TPS, CV 2.2%/3.6%. **PASSES continuous soak** (p50 66.75, 100% retention). verify-full 8/8 + verify-stress 7/7. PR [#96](https://github.com/noonghunna/club-3090/pull/96). |
|
||||
| `dual.yml`-shape **+ patched P2P drivers** (no NVLink hardware) | @aaronlockhartdev (2× 3090 PCIe x16, EPYC 7F52, Arch Linux, custom Dockerfile via [Sam McLeod's guide](https://smcleod.net/2026/02/patching-nvidias-driver-and-vllm-to-enable-p2p-on-consumer-gpus/) — patched `aikitoria/open-gpu-kernel-modules` + vLLM `cuda.py` `return True` patch) | fp8 | 262K | **93 / 125** | — | n/a | 2026-05-07 | **First patched-driver P2P cross-rig data point** — answers the question raised in [disc #70](https://github.com/noonghunna/club-3090/discussions/70). Same-rig controlled A/B: unpatched baseline 91 narr / 114 code → patched P2P 93 / 125 = **+2% narr / +9% code**. Compared to NVLink hardware lift (+15% / +15% per @danbedford's controlled A/B): patched P2P captures **~60% of NVLink's code gain but ~13% of NVLink's narr gain** — code workloads (spec-decode K+1 verify is heavily cross-card matmul) benefit more from cross-card bandwidth than narr decode (more sequential per-token). For ~95% of dual-3090 owners without NVLink, the trade is small TPS lift vs custom kernel module + DKMS maintenance burden. [Issue #91](https://github.com/noonghunna/club-3090/issues/91). |
|
||||
| `dual-dflash-noviz.yml`-shape **+ patched P2P drivers** (no NVLink hardware, custom_all_reduce ENABLED) | @aaronlockhartdev (2× 3090 PCIe x16, EPYC 7F52, Arch Linux, patched kernel module + `NCCL_P2P_LEVEL=PHB`) | fp8 | 200K | **100.47 / 160.15** (decode 101.53 / 164.44) | — | ~22.2 GB/card | 2026-05-07 | **Second patched-P2P cross-rig data point** — extends [#91 dual.yml result](https://github.com/noonghunna/club-3090/issues/91) to the DFlash + no-vision path. Same-rig controlled A/B: unpatched baseline 82.55 narr / 134.45 code → patched P2P 100.47 / 160.15 = **+22% narr / +19% code**. **Significantly larger lift than `dual.yml`-shape** (+22%/+19% here vs +2%/+9% on `dual.yml`) — DFlash's K+1 cross-card verify pattern stresses peer-bandwidth more than fp8-only `dual.yml`. **Important methodology update**: `NCCL_P2P_LEVEL=PHB` alone with the default vLLM image produced the same lift as the full vLLM `cuda.py` patch — **the in-container vLLM source patch is unnecessary**, only the kernel module patch matters. CV 4.6%/2.4%. custom_all_reduce ENABLED (vs disabled on the `dual.yml` row). [Issue #95](https://github.com/noonghunna/club-3090/issues/95) + [disc #70](https://github.com/noonghunna/club-3090/discussions/70). |
|
||||
| `carnice-bf16mtp.yml` | @noonghunna (2× 3090 PCIe, no NVLink) | fp8 | 262K | **72** / **80** | — | ~22.25 GB | 2026-05-04 | **Carnice-V2-27B (Hermes agentic fine-tune) + BF16 MTP overlay**. Full 262K context, 2 streams. 71.75 narr / 80.35 code wall TPS (n=5 each, CV ~11%), MTP AL 3.02-3.14, TTFT 141ms. Patched chat template for Hermes JSON tool calls. verify-full 7/8 PASS. soak PASS. |
|
||||
| `dual.yml` ⭐ | @lolren (2× 3090 PCIe + Ryzen 9 5950X, **250W/card cap**) | fp8 | 262K | **89.78 / 117.60** | — | ~22.3 GB | 2026-05-05 | **First cross-rig data on the v7.72.2 uplift** (image `nightly-01d4d1ad3`, post-PR #59). +30% narr / +32% code over @noonghunna 2026-04-29 baseline (69/89 on older image) — confirms the v7.72.2 dividend cross-rig. CV 3.3%/2.0%. MTP AL ~3.5, per-pos accept 94/84/72%. [Disc #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16820303). |
|
||||
| `dual-dflash.yml` | @lolren (2× 3090 PCIe + Ryzen 9 5950X, 250W cap) | FP16 | 185K | 87.10 / **142.0** | — | ~22.1 GB | 2026-05-05 | Older image `nightly-7a1eb8ac2`. +6% narr / +14% code over @noonghunna baseline (82/125) — likely Ryzen 5950X advantage on prefill. DFlash AL ~4.5, per-pos accept 93/81/68/56/48%, avg accept 69%. [Disc #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16820303). |
|
||||
| `bounded-thinking.yml` | @lolren (2× 3090 PCIe + Ryzen 9 5950X, 250W cap, **MTP-disabled-suspected**) | TQ3 | 180K | 64.86 / 64.96 (CV **0.1%**) | — | ~22.3 GB | 2026-05-05 | **Anomaly:** lolren reports "no spec-decode" on this run despite `bounded-thinking.yml` shipping `--speculative-config mtp n=3` by default. Near-identical narr=code TPS + extreme CV stability (0.1%) suggests MTP was inactive — likely because his image was older `nightly-7a1eb8ac2` (pre-v7.72.2 + pre-PN35). Re-test on `nightly-01d4d1ad3` should restore MTP path → expect ~50/66 narr/code with normal CV. Tracked. [Disc #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16820303). |
|
||||
| `dual.yml` | @JDWarner (**Mixed RTX A5000 + RTX 3090**, both **Razer Core X eGPU enclosures over Thunderbolt 3**, Intel NUC11TNH i5-1135G7, **16 GB RAM**, headless, A5000=230W cap / 3090=290W cap, PCIe **x4 Gen 3** per card) | fp8 | 262K | **56.83 / 72.47** (soak p50 93.09) | — | ~23.6 GB/card | 2026-05-09 | **Soak: ✓ PASS** (5×5, 0 errors, 0 silent-empty, 100% TPS retention, 0 MiB growth). **First TB3 dual-eGPU + mixed-arch cross-rig data**. The setup that "shouldn't work": each card on a separate TB3 controller → ~3.94 GB/s effective per card vs ~32 GB/s on PCIe x16 Gen 4 (~8× cut), mixed Ampere SKUs (workstation A5000 + consumer 3090 with different mem bandwidth + clocks), 16 GB system RAM total. **Result: matches `dual.yml` PCIe x16 baseline within run-to-run noise** — confirms decode on Qwen3.6-27B is per-card-bandwidth bound, cross-card NCCL allreduce is small enough that even an 8× link cut doesn't dominate. Extends @aaronlockhartdev's #91/#95 finding (patched-P2P only +2%/+9% on `dual.yml`) in the opposite direction: even with 8× *less* cross-card bandwidth, decode holds. MTP AL 3.39-3.52, per-pos accept 0.93/0.83/0.70 (89% avg). verify-full + verify-stress all PASS. Genesis pin `7b9fd319` (v7.72.2). [Issue #107](https://github.com/noonghunna/club-3090/issues/107). |
|
||||
| `dual/docker-compose.yml` (default) | @ygafarov (**3090 via USB4 eGPU dock + 5070 Ti via OCuLink** — heterogeneous Ampere + Blackwell consumer dual-eGPU, AMD Ryzen AI MAX+ 395 / Strix Halo miniPC, CachyOS, 123 GB RAM, 290 W cap both cards, PCIe **x4** per card — USB4 ≈ 3.94 GB/s, OCuLink ≈ 7.88 GB/s) | fp8 | 200K | **65.10 / 85.81** | — | 17.1 / 15.7 GB | 2026-05-12 | **First heterogeneous Ampere + Blackwell consumer dual-eGPU on the matrix.** TP=2 bound by the slower USB4 link in allreduce + sm_86 kernels (5070 Ti spends back-half of step waiting — 91% util but only 125 W out of 290 W cap). KV pool 200K @ 1.00× concurrency — VRAM cap from the 5070 Ti's 16 GiB (model takes 13.8 GiB/card → only ~2.2 GiB left for KV on the smaller card). verify-stress 7/7 incl. **91K needle recall** (Cliff 2 clean). Soak ⚠ borderline (360 MiB > 200 MiB threshold — same eGPU-bus accretion as ygafarov's own #113 single-card row above at 240 MiB; 100% TPS retention + 0 silent-empty + 0 errors so not a leak). MTP AL 3.50, per-pos accept 0.94/0.86/0.70. CV 4.5%/1.8%. **Slower than ygafarov's own single-3090 #113 row** (68.86/91.70 at 48K) — on this rig the single-card path is recommended; the 5070 Ti adds VRAM cap pain without TPS gain. Driver 595.71.05, vLLM `nightly-1acd67a79`, no Genesis (Blackwell consumer not on allowlist). [Issue #120](https://github.com/noonghunna/club-3090/issues/120). |
|
||||
|
||||
### Quad-card (4× RTX 3090, TP=4)
|
||||
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | Peak VRAM | Date | Notes |
|
||||
|---|---|---|---:|---:|---:|---|---|
|
||||
| `multi4.yml` | @whamp (4× 3090 PCIe x4/x16/x8/x16, 300 W cap, no NVLink) | fp8 | 262K | 63 / 76 | ~23.5 GB | 2026-05-03 | TP=4 capacity king. **6.77× concurrency at 262K**. PASSES v2 continuous soak (20 sessions, 0 MiB growth, 90.8% TPS retention). PR [#44](https://github.com/noonghunna/club-3090/pull/44). |
|
||||
| `multi4.yml` | [@alanspires #127](https://github.com/noonghunna/club-3090/issues/127) (**6× 3090 VFIO-passthrough**, AMD EPYC 7313 host, all cards 250W cap, no NVLink) | fp8 | 262K | **74.93 / 92.85** | ~21.7 GB | 2026-05-14 | TP=4 on a 6-card rig — GPUs 4-5 free (Qwen `num_kv_heads=4` doesn't divide 6, so TP=6 invalid). **First VFIO-passthrough + virtualized data point** — scripts (verify-full / verify-stress / soak) all PASS on the virt envelope without modification. KV pool 1,774,963 tokens at 6.77× concurrency. Soak p50 122.84 / p95 161.14 across 5 multi-turn sessions, 0 errors. |
|
||||
| `multi4-dflash.yml` | @whamp (4× 3090 PCIe x4/x16/x8/x16, 300 W cap) | fp8 | 262K | 64 / **104** | ~22.0 GB | 2026-05-03 | TP=4 + DFlash. 2.27× concurrency at 262K. PASSES v2 continuous soak (5 sessions, 0 MiB growth, 100% TPS retention). **Bench-vs-soak inversion**: bench shows DFlash wins by 37% on short-prompt code, soak shows DFlash *loses* by 47% on multi-turn agent — DFlash AL likely collapses on mixed prompts. PR [#44](https://github.com/noonghunna/club-3090/pull/44). |
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | PP tok/s | Peak VRAM | Date | Notes |
|
||||
| --- | --- | --- | ---: | ---: | ---: | ---: | --- | --- |
|
||||
| `multi4.yml` | @whamp (4× 3090 PCIe x4/x16/x8/x16, 300 W cap, no NVLink) | fp8 | 262K | 63 / 76 | — | ~23.5 GB | 2026-05-03 | TP=4 capacity king. **6.77× concurrency at 262K**. PASSES v2 continuous soak (20 sessions, 0 MiB growth, 90.8% TPS retention). PR [#44](https://github.com/noonghunna/club-3090/pull/44). |
|
||||
| `multi4.yml` | [@alanspires #127](https://github.com/noonghunna/club-3090/issues/127) (**6× 3090 VFIO-passthrough**, AMD EPYC 7313 host, all cards 250W cap, no NVLink) | fp8 | 262K | **74.93 / 92.85** | — | ~21.7 GB | 2026-05-14 | TP=4 on a 6-card rig — GPUs 4-5 free (Qwen `num_kv_heads=4` doesn't divide 6, so TP=6 invalid). **First VFIO-passthrough + virtualized data point** — scripts (verify-full / verify-stress / soak) all PASS on the virt envelope without modification. KV pool 1,774,963 tokens at 6.77× concurrency. Soak p50 122.84 / p95 161.14 across 5 multi-turn sessions, 0 errors. |
|
||||
| `multi4-dflash.yml` | @whamp (4× 3090 PCIe x4/x16/x8/x16, 300 W cap) | fp8 | 262K | 64 / **104** | — | ~22.0 GB | 2026-05-03 | TP=4 + DFlash. 2.27× concurrency at 262K. PASSES v2 continuous soak (5 sessions, 0 MiB growth, 100% TPS retention). **Bench-vs-soak inversion**: bench shows DFlash wins by 37% on short-prompt code, soak shows DFlash *loses* by 47% on multi-turn agent — DFlash AL likely collapses on mixed prompts. PR [#44](https://github.com/noonghunna/club-3090/pull/44). |
|
||||
|
||||
### Verify-stress + soak-continuous matrix
|
||||
|
||||
@@ -228,18 +234,18 @@ Tested via `URL=http://localhost:8004 MODEL=luce-dflash bash scripts/verify-stre
|
||||
|
||||
Cross-rig data on Google's official Gemma 4 MTP "assistant" drafter (released 2026-05-05). PR [#41745](https://github.com/vllm-project/vllm/pull/41745) merged 2026-05-06 → today's nightly contains it natively (overlay dropped 2026-05-08). The companion compose [`dual/int8.yml`](models/gemma-4-31b/vllm/compose/dual/int8.yml) (added 2026-05-08) vendors PR [#40391](https://github.com/vllm-project/vllm/pull/40391) (rebased) + PR #42006 + PR #41991 to unlock per-token-head INT8 KV → 8.2× context lift on Ampere (32K → 262K). See announcement [discussion #67](https://github.com/noonghunna/club-3090/discussions/67) for the original Gemma 4 setup story; Phase 2 INT8 PTH validation in progress 2026-05-08.
|
||||
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | AL | Per-pos accept (code) | Peak VRAM | Date | Notes |
|
||||
|---|---|---|---:|---:|---:|---|---:|---|---|
|
||||
| `dual.yml` (TP=2) | @noonghunna (2× 3090 PCIe, no NVLink, 230W cap) | bf16 | 32K | **108.87 / 142.25** | **3.94-4.04** | 92 / 79 / 68 / 59 % | 22.5 GB/card | 2026-05-05 | First Ampere consumer cross-rig data on Google MTP drafters. **+1.79× narr / +2.31× code** over baseline (61 TPS no-spec-decode same TP). **PASSES continuous soak** (100 turns, 0 errors / 0 silent-empty / 0 MiB growth, 98.3% TPS retention). bf16 KV (fp8 blocked on Ampere — see TP=1 row). PR [#41745](https://github.com/vllm-project/vllm/pull/41745) overlay + transformers 5.8.0 entrypoint. |
|
||||
| `dual.yml` (TP=2) re-bench post-#41745 merge | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | 32K | **105.91 / 141.11** | 3.94 | (warming) | 21.5 GB/card | 2026-05-08 | Re-validated on post-merge nightly `1acd67a795...` (PR #41745 overlay dropped, transformers entrypoint upgrade dropped). Within CV of 109/142 baseline → cleanup is parity-clean. KV pool 99K tokens, 3.03× concurrency at 32K. |
|
||||
| **`dual-int8.yml` (TP=2, max-num-seqs=4)** ⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | **int8_per_token_head** | **98K** | **96.16 / 127.11** | 3.79 | (warming) | 22.2 GB/card | 2026-05-08 | **3.07× context lift over bf16 ceiling on Ampere — INT8 PTH KV unblocks Gemma 4 long-context.** Vendors PR #40391 rebased + PR #42006 + PR #41991 stacked (see `models/gemma-4-31b/vllm/patches/`). KV pool **354K tokens, 3.6× concurrency**. ~10% TPS cost vs bf16 / 32K. **PASSES verify-stress 7/7** incl. 91K Cliff-2 needle. PR #40391's per-token-head page-size fix routes via `get_padded_attention_kv_cache_shape()`; INT8 (not fp8) is the right Ampere dtype because Triton `fp8e4nv` kernel is not supported on sm_86 (Ada/Blackwell only). |
|
||||
| **`dual-int8.yml` (TP=2, max-num-seqs=1, MAX_MODEL_LEN=262144)** ⭐⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | int8_per_token_head | **262K (model native max)** | **95.27 / 125.93** | 3.93 | (warming) | 22.1 GB/card | 2026-05-08 | **8.2× context lift vs dual.yml — full Gemma 4 native context (262144) unblocked on dual 3090 Ampere.** KV pool 455K tokens, 1.74× concurrency at full 262K. **PASSES verify-stress 7/7** + **137K NIAH PASS** (correctly recalled needle from 137,557-token prompt, 5min wall, ~458 prefill TPS). Per-token TPS preserved at full max-model-len (95/126 at 262K vs 96/127 at 98K — bench prompt size dominates, not max-model-len). Override `MAX_MODEL_LEN=262144 MAX_NUM_SEQS=1`. |
|
||||
| `dual-dflash.yml` (TP=2, n=7) | @noonghunna (2× 3090 PCIe, no NVLink, 230W cap) | bf16 | 32K | **95.16 / 167.55** | **~3.0 narr / 5.23 code** | 89 / 78 / 66 / 57 / 50 / 43 / 39 % | 22.7 GB/card | 2026-05-06 | First Ampere consumer cross-rig data on **z-lab Gemma 4 DFlash** block-diffusion drafter (vLLM PR [#41703](https://github.com/vllm-project/vllm/pull/41703) — Codex-rebased onto upstream/main). **+2.74× code / +1.56× narr** over baseline. **PASSES continuous soak** (100 turns, 0 errors / 0 silent-empty / 0 MiB growth, 98.6% TPS retention, p50 55.8 TPS — 5.8% higher than n=5). DFlash dominates MTP on **code (+18%)**; MTP wins on narrative (+15%). n-sweep: n=5 109/141 (best narr) → n=6 99/161 (knee) → **n=7 95/168 (code-optimal default)** → n=8 91/167 (dominated) → n=15 82/172 (past knee). PR #41703 overlay (12 RO-mounted files) + transformers 5.8.0 + nightly `e47c98ef`. |
|
||||
| `dual-dflash.yml` (TP=2, n=7) re-bench | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | 32K | **104.48 / 176.66** (CV 2.1% / 3.6%) | 2.85 narr / 4.11-4.94 code | (warm) | 22.3 GB/card | 2026-05-08 | Re-validated after Phase 2 INT8 PTH session. Same overlay + same `e47c98ef` pin — modest uplift over 2026-05-06 (warm-cache + ambient variance — CV ranges overlap at +1σ). KV pool 42,848 tokens, 1.31× concurrency at 32K. Ampere upper-bound for Gemma 4 + DFlash: code-optimal at 177 TPS. Combining DFlash drafter with PR #40391 INT8 PTH KV (Phase 3, dual-dflash-int8.yml) is the next structural step — would unlock long-context code-optimal. |
|
||||
| **`dual-dflash-int8.yml` (TP=2, n=7, MAX_MODEL_LEN=262144, MAX_NUM_SEQS=1)** ⭐⭐⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | int8_per_token_head + drafter bf16 | **262K (model native max)** | **86.86 / 145.96** (CV 0.9% / 2.2%) | 5.0-5.3 long-ctx code | (warm) | 22.0 GB/card | 2026-05-08 | **8.2× context lift over `dual-dflash.yml` 32K bf16 baseline — DFlash + INT8 PTH KV unblocked on Ampere via [vLLM PR #42102](https://github.com/vllm-project/vllm/pull/42102) (our patch).** Matches the 32K bf16 baseline's code TPS within CV at 8× more context (146 vs 168 = -13% perf cost for 8× ctx). KV pool 168,178 tokens, 0.64× concurrency at full 262K — effective single-stream serving ceiling ~168K. **NIAH PASS at 98,444 tokens** (`bronze octopus 17` recalled cleanly, 157s wall = ~625 effective prefill TPS). DFlash drafter uses BF16 KV in independent pool (target uses INT8 PTH); the patch partitions them at unify-time, drafter cache_dtype overridden to "auto" in qwen3_dflash.py, FA metadata scheduler reads per-spec dtype. **n-sweep at 262K config 2026-05-08**: n=5 81.91/138.87 (-6/-5%), n=7 86.86/145.96 (default), n=8 86.63/152.06 (+0/+4% but CV 5.7% — within noise). n=7 retained as default — sweet spot didn't shift meaningfully from the 32K bf16 baseline. **DFlash code-optimal advantage preserved at long ctx**: 146 code TPS vs dual-int8.yml's 126 at 262K = +16% code (offset: -10% narr). Pin nightly `e47c98ef`; needs `dual-dflash-int8.yml` compose (still ⚠️ flagged DOES NOT BOOT until PR #42102 lands — currently requires the vllm-src patch mounts). |
|
||||
| `single.yml` (TP=1) | @noonghunna (1× 3090) | bf16 / fp8 | — | **boot OOM** | — | — | — | 2026-05-05 | **Upstream-blocked on Ampere consumer.** bf16 KV: weights+drafter+profiling at 8K ctx + mem-util 0.95 leaves zero KV pool ("No available memory for the cache blocks"). fp8 KV: Triton `fp8e4nv not supported in this architecture` on sm_86 (Ampere supports `fp8e4b15`/`fp8e5` only); but `fp8_e5m2` is rejected by `gemma4_mm.py:1336` allowlist. Compose preserved for re-test when (a) vLLM adds Ampere-aware fp8 dispatch OR (b) PR #41745 relaxes the assert. Gemma 4 26B-A4B MoE single-card is the obvious follow-up. |
|
||||
| **`dual-awq.yml` (TP=2, MAX_NUM_SEQS=4, MTP n=4)** ⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | **65K** | **104.59 / 130.56** (CV 1.7% / 0.4%) | 3.07 narr / 3.55-3.88 code | 0.79 / 0.59 / 0.43 / 0.31 | 19.8 GB/card | 2026-05-08 | **Cross-rig reproducer of @3dluvr's [#103 bench](https://github.com/noonghunna/club-3090/issues/103) — AWQ-4bit weights instead of AutoRound INT4.** Bypasses PR #40391 (per-token-head bug) entirely because AWQ doesn't use FP8 KV — trades weight quant precision for ctx instead of trading KV precision. Vendor: `cyankiwi/gemma-4-31B-it-AWQ-4bit` (~17 GB on disk, AWQ-pack-quantized group_size=32, asymmetric, MSE observer; routed via vLLM compressed-tensors loader → Marlin kernel). KV pool 89,228 tokens, 1.36× concurrency at 65K. Numbers comparable to dual.yml (BF16 INT4 AutoRound, 105.91/141.11) on narrative; slightly slower code at this n. Default config — multi-stream agent / RAG. |
|
||||
| **`dual-awq.yml` (TP=2, MAX_NUM_SEQS=1, MTP n=8, MAX_MODEL_LEN=118304)** ⭐⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | **118K** | **101.16 / 141.90** (CV 2.5% / 0.4%) | 3.7 narr / 5.13 code | 0.89 / 0.77 / 0.64 / 0.54 / 0.45 / 0.37 / 0.26 / 0.21 | 19.8 GB/card | 2026-05-08 | **3.7× context lift over dual.yml's BF16 32K ceiling.** Closer match to @3dluvr's #103 anchor (113/163 at 195K with `--dtype half --async-scheduling --cudagraph_capture_sizes [9]`). Our config: `--dtype bfloat16`, default cudagraph capture sizes `[1,2,4,8,16]` (auto-clamped to single-stream). vLLM auto-estimated 195K won't fit at 0.85 mem-util (10.18 GiB KV needed vs 7.25 GiB available); 118K is the achievable ceiling at this mem-util / cudagraph budget. **NIAH PASS at 88K** (recalled "bronze octopus 17" from 88,527-token prompt, 11-tok completion in 127.1s wall, ~696 effective prefill TPS). KV pool 118,304 tokens, 1.00× concurrency. Code AL 5.13 (n=8 saturates well on code, less so on narr where AL is 3.7). **n=8 vs n=4 trade**: code +9% (130 → 142), narrative -3% (104 → 101) — n=8 dominates n=4 for code. Override `MAX_MODEL_LEN=118304 MAX_NUM_SEQS=1 MTP_N=8`. **A/B finding 2026-05-08** — three tuning flags tested individually on this rig (with bench n=3 each, sync-baseline 101.16/141.90):
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | PP tok/s | AL | Per-pos accept (code) | Peak VRAM | Date | Notes |
|
||||
| --- | --- | --- | ---: | ---: | ---: | ---: | --- | ---: | --- | --- |
|
||||
| `dual.yml` (TP=2) | @noonghunna (2× 3090 PCIe, no NVLink, 230W cap) | bf16 | 32K | **108.87 / 142.25** | — | **3.94-4.04** | 92 / 79 / 68 / 59 % | 22.5 GB/card | 2026-05-05 | First Ampere consumer cross-rig data on Google MTP drafters. **+1.79× narr / +2.31× code** over baseline (61 TPS no-spec-decode same TP). **PASSES continuous soak** (100 turns, 0 errors / 0 silent-empty / 0 MiB growth, 98.3% TPS retention). bf16 KV (fp8 blocked on Ampere — see TP=1 row). PR [#41745](https://github.com/vllm-project/vllm/pull/41745) overlay + transformers 5.8.0 entrypoint. |
|
||||
| `dual.yml` (TP=2) re-bench post-#41745 merge | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | 32K | **105.91 / 141.11** | — | 3.94 | (warming) | 21.5 GB/card | 2026-05-08 | Re-validated on post-merge nightly `1acd67a795...` (PR #41745 overlay dropped, transformers entrypoint upgrade dropped). Within CV of 109/142 baseline → cleanup is parity-clean. KV pool 99K tokens, 3.03× concurrency at 32K. |
|
||||
| **`dual-int8.yml` (TP=2, max-num-seqs=4)** ⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | **int8_per_token_head** | **98K** | **96.16 / 127.11** | — | 3.79 | (warming) | 22.2 GB/card | 2026-05-08 | **3.07× context lift over bf16 ceiling on Ampere — INT8 PTH KV unblocks Gemma 4 long-context.** Vendors PR #40391 rebased + PR #42006 + PR #41991 stacked (see `models/gemma-4-31b/vllm/patches/`). KV pool **354K tokens, 3.6× concurrency**. ~10% TPS cost vs bf16 / 32K. **PASSES verify-stress 7/7** incl. 91K Cliff-2 needle. PR #40391's per-token-head page-size fix routes via `get_padded_attention_kv_cache_shape()`; INT8 (not fp8) is the right Ampere dtype because Triton `fp8e4nv` kernel is not supported on sm_86 (Ada/Blackwell only). |
|
||||
| **`dual-int8.yml` (TP=2, max-num-seqs=1, MAX_MODEL_LEN=262144)** ⭐⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | int8_per_token_head | **262K (model native max)** | **95.27 / 125.93** | — | 3.93 | (warming) | 22.1 GB/card | 2026-05-08 | **8.2× context lift vs dual.yml — full Gemma 4 native context (262144) unblocked on dual 3090 Ampere.** KV pool 455K tokens, 1.74× concurrency at full 262K. **PASSES verify-stress 7/7** + **137K NIAH PASS** (correctly recalled needle from 137,557-token prompt, 5min wall, ~458 prefill TPS). Per-token TPS preserved at full max-model-len (95/126 at 262K vs 96/127 at 98K — bench prompt size dominates, not max-model-len). Override `MAX_MODEL_LEN=262144 MAX_NUM_SEQS=1`. |
|
||||
| `dual-dflash.yml` (TP=2, n=7) | @noonghunna (2× 3090 PCIe, no NVLink, 230W cap) | bf16 | 32K | **95.16 / 167.55** | — | **~3.0 narr / 5.23 code** | 89 / 78 / 66 / 57 / 50 / 43 / 39 % | 22.7 GB/card | 2026-05-06 | First Ampere consumer cross-rig data on **z-lab Gemma 4 DFlash** block-diffusion drafter (vLLM PR [#41703](https://github.com/vllm-project/vllm/pull/41703) — Codex-rebased onto upstream/main). **+2.74× code / +1.56× narr** over baseline. **PASSES continuous soak** (100 turns, 0 errors / 0 silent-empty / 0 MiB growth, 98.6% TPS retention, p50 55.8 TPS — 5.8% higher than n=5). DFlash dominates MTP on **code (+18%)**; MTP wins on narrative (+15%). n-sweep: n=5 109/141 (best narr) → n=6 99/161 (knee) → **n=7 95/168 (code-optimal default)** → n=8 91/167 (dominated) → n=15 82/172 (past knee). PR #41703 overlay (12 RO-mounted files) + transformers 5.8.0 + nightly `e47c98ef`. |
|
||||
| `dual-dflash.yml` (TP=2, n=7) re-bench | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | 32K | **104.48 / 176.66** (CV 2.1% / 3.6%) | — | 2.85 narr / 4.11-4.94 code | (warm) | 22.3 GB/card | 2026-05-08 | Re-validated after Phase 2 INT8 PTH session. Same overlay + same `e47c98ef` pin — modest uplift over 2026-05-06 (warm-cache + ambient variance — CV ranges overlap at +1σ). KV pool 42,848 tokens, 1.31× concurrency at 32K. Ampere upper-bound for Gemma 4 + DFlash: code-optimal at 177 TPS. Combining DFlash drafter with PR #40391 INT8 PTH KV (Phase 3, dual-dflash-int8.yml) is the next structural step — would unlock long-context code-optimal. |
|
||||
| **`dual-dflash-int8.yml` (TP=2, n=7, MAX_MODEL_LEN=262144, MAX_NUM_SEQS=1)** ⭐⭐⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | int8_per_token_head + drafter bf16 | **262K (model native max)** | **86.86 / 145.96** (CV 0.9% / 2.2%) | — | 5.0-5.3 long-ctx code | (warm) | 22.0 GB/card | 2026-05-08 | **8.2× context lift over `dual-dflash.yml` 32K bf16 baseline — DFlash + INT8 PTH KV unblocked on Ampere via [vLLM PR #42102](https://github.com/vllm-project/vllm/pull/42102) (our patch).** Matches the 32K bf16 baseline's code TPS within CV at 8× more context (146 vs 168 = -13% perf cost for 8× ctx). KV pool 168,178 tokens, 0.64× concurrency at full 262K — effective single-stream serving ceiling ~168K. **NIAH PASS at 98,444 tokens** (`bronze octopus 17` recalled cleanly, 157s wall = ~625 effective prefill TPS). DFlash drafter uses BF16 KV in independent pool (target uses INT8 PTH); the patch partitions them at unify-time, drafter cache_dtype overridden to "auto" in qwen3_dflash.py, FA metadata scheduler reads per-spec dtype. **n-sweep at 262K config 2026-05-08**: n=5 81.91/138.87 (-6/-5%), n=7 86.86/145.96 (default), n=8 86.63/152.06 (+0/+4% but CV 5.7% — within noise). n=7 retained as default — sweet spot didn't shift meaningfully from the 32K bf16 baseline. **DFlash code-optimal advantage preserved at long ctx**: 146 code TPS vs dual-int8.yml's 126 at 262K = +16% code (offset: -10% narr). Pin nightly `e47c98ef`; needs `dual-dflash-int8.yml` compose (still ⚠️ flagged DOES NOT BOOT until PR #42102 lands — currently requires the vllm-src patch mounts). |
|
||||
| `single.yml` (TP=1) | @noonghunna (1× 3090) | bf16 / fp8 | — | **boot OOM** | — | — | — | — | 2026-05-05 | **Upstream-blocked on Ampere consumer.** bf16 KV: weights+drafter+profiling at 8K ctx + mem-util 0.95 leaves zero KV pool ("No available memory for the cache blocks"). fp8 KV: Triton `fp8e4nv not supported in this architecture` on sm_86 (Ampere supports `fp8e4b15`/`fp8e5` only); but `fp8_e5m2` is rejected by `gemma4_mm.py:1336` allowlist. Compose preserved for re-test when (a) vLLM adds Ampere-aware fp8 dispatch OR (b) PR #41745 relaxes the assert. Gemma 4 26B-A4B MoE single-card is the obvious follow-up. |
|
||||
| **`dual-awq.yml` (TP=2, MAX_NUM_SEQS=4, MTP n=4)** ⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | **65K** | **104.59 / 130.56** (CV 1.7% / 0.4%) | — | 3.07 narr / 3.55-3.88 code | 0.79 / 0.59 / 0.43 / 0.31 | 19.8 GB/card | 2026-05-08 | **Cross-rig reproducer of @3dluvr's [#103 bench](https://github.com/noonghunna/club-3090/issues/103) — AWQ-4bit weights instead of AutoRound INT4.** Bypasses PR #40391 (per-token-head bug) entirely because AWQ doesn't use FP8 KV — trades weight quant precision for ctx instead of trading KV precision. Vendor: `cyankiwi/gemma-4-31B-it-AWQ-4bit` (~17 GB on disk, AWQ-pack-quantized group_size=32, asymmetric, MSE observer; routed via vLLM compressed-tensors loader → Marlin kernel). KV pool 89,228 tokens, 1.36× concurrency at 65K. Numbers comparable to dual.yml (BF16 INT4 AutoRound, 105.91/141.11) on narrative; slightly slower code at this n. Default config — multi-stream agent / RAG. |
|
||||
| **`dual-awq.yml` (TP=2, MAX_NUM_SEQS=1, MTP n=8, MAX_MODEL_LEN=118304)** ⭐⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | **118K** | **101.16 / 141.90** (CV 2.5% / 0.4%) | — | 3.7 narr / 5.13 code | 0.89 / 0.77 / 0.64 / 0.54 / 0.45 / 0.37 / 0.26 / 0.21 | 19.8 GB/card | 2026-05-08 | **3.7× context lift over dual.yml's BF16 32K ceiling.** Closer match to @3dluvr's #103 anchor (113/163 at 195K with `--dtype half --async-scheduling --cudagraph_capture_sizes [9]`). Our config: `--dtype bfloat16`, default cudagraph capture sizes `[1,2,4,8,16]` (auto-clamped to single-stream). vLLM auto-estimated 195K won't fit at 0.85 mem-util (10.18 GiB KV needed vs 7.25 GiB available); 118K is the achievable ceiling at this mem-util / cudagraph budget. **NIAH PASS at 88K** (recalled "bronze octopus 17" from 88,527-token prompt, 11-tok completion in 127.1s wall, ~696 effective prefill TPS). KV pool 118,304 tokens, 1.00× concurrency. Code AL 5.13 (n=8 saturates well on code, less so on narr where AL is 3.7). **n=8 vs n=4 trade**: code +9% (130 → 142), narrative -3% (104 → 101) — n=8 dominates n=4 for code. Override `MAX_MODEL_LEN=118304 MAX_NUM_SEQS=1 MTP_N=8`. **A/B finding 2026-05-08** — three tuning flags tested individually on this rig (with bench n=3 each, sync-baseline 101.16/141.90): |
|
||||
| Flag | Narr Δ | Code Δ | Verdict |
|
||||
|---|---:|---:|---|
|
||||
| `--dtype half` (vs bfloat16) | -3% | -2% | Ampere has NO fp16 hardware accel on sm_86 — bfloat16 is canonical |
|
||||
@@ -250,6 +256,21 @@ None close the **-13% narr / -11% code gap to 3dluvr's anchor**. Remaining gap l
|
||||
| `dual-dflash.yml`-shape forced TP=1 (mem-util 0.96, max-model-len 12000) | @apnar (1× **RTX 5090** 32 GB, air-cooled, 600 W) | bf16 | **12K** | **150.40 / 261.06** (decode 151.16 / 264.62) | 28.8 GB | 2026-05-07 | **First single-5090 Gemma 4 DFlash data point.** Trade vs MTP row above: ~6% narr loss, **+21% code lift** (215→261). 1st-warmup TTFT outlier (73 s) suggests cudagraph warmup taking longer on first request; subsequent warmups stable at <40 ms. CV 3.6%/2.8%, peak 440 W. **Required mem-util 0.96 + max-model-len 12K** to fit BF16 weights + DFlash N=5 drafter on 32 GB — DFlash drafter footprint pushes out ctx ceiling vs MTP's 32K. [Disc #67](https://github.com/noonghunna/club-3090/discussions/67#discussioncomment-16832042). |
|
||||
|
||||
|
||||
---
|
||||
|
||||
## MoE models (v0.7.3 — preview track)
|
||||
|
||||
First-pass numbers for the two MoE models onboarded in v0.7.3. **Preview** track = `vllm-nightly-clean` engine (no Genesis patches, no TQ3 KV, no MTP yet) — exercises the upstream MoE loader [PR #42521](https://github.com/vllm-project/vllm/pull/42521) (qwen3_5_moe weight loading, merged 2026-05-14) without other overlays. Production track (Genesis-anchored, MTP, longer context) lands in v0.7.4 after Genesis v7.73.x re-anchors on a post-#42521 nightly.
|
||||
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | PP tok/s | AL | Per-pos accept | Peak VRAM | Date | Notes |
|
||||
| --- | --- | --- | ---: | ---: | ---: | ---: | --- | ---: | --- | --- |
|
||||
| **`qwen3.6-35b-a3b/dual/preview.yml` (TP=2)** ⭐ | @noonghunna (2× 3090 PCIe, 230 W cap) | fp8_e5m2 | 16K | **182.68 / 177.45** (decode 186.98 / 186.90) | — | n/a (no drafter) | n/a | **21.94 GB/card** | 2026-05-15 | **First v0.7.3 MoE preview row on the matrix.** Engine `vllm-nightly-clean` (nightly `bf610c2f`, post-#42521). No Genesis, no MTP — exercises the upstream qwen3_5_moe loader cleanly. CV **0.3% / 0.9%** (very stable). TTFT 126 ms. GPU 0 at 99% util / 292 W, GPU 1 at 85% / 244 W. Decode TPS basically identical narr vs code (187 / 187) — characteristic of MoE memory-bandwidth-bound decode (3 B active params per forward, weights fit cache). **~2× the Qwen 3.6 27B dense `dual.yml` baseline** (89/118 wall) on the same hardware — MoE's active-params advantage on Ampere. Next steps: TQ3 KV + MTP (built-in head) after Genesis v7.73.x lands; longer ctx after upstream MoE expert dispatch overhead is measured. Compose: `models/qwen3.6-35b-a3b/vllm/compose/dual/preview.yml`. |
|
||||
| `qwen3.6-35b-a3b/dual/preview-mtp.yml` (TP=2, MTP n=3, built-in head) ⚠️ | @noonghunna (2× 3090 PCIe, 230 W cap) | fp8_e5m2 | 16K | **90.36 / 115.33** (decode 91.48 / 119.65) | — | **3.44** (narr) | 0.927 / 0.810 / 0.698 | 22.72 GB/card | 2026-05-15 | **MTP MAKES THINGS SLOWER on Qwen MoE preview path.** A/B vs preview.yml above (same config, same nightly, same hardware): **−51% narr / −35% code wall TPS** despite 81.2% avg draft acceptance and AL 3.44. Per-position accept (0.927 / 0.810 / 0.698) is healthy — the draft head works correctly. Surfacing cause: asymmetric GPU util (GPU 0 at **39%** / 233 W vs GPU 1 at 79% / 191 W; non-MTP run had both at 85-99%) indicates the draft forward pass on MoE imposes inter-GPU sync overhead the acceptance gain can't amortize. Vendor warning at boot: `max_num_scheduled_tokens=4096 ... suboptimal performance ... increase max_num_batched_tokens to accommodate the additional draft token slots`. **CV 5.4% / 2.0%** (less stable than no-MTP). VRAM +0.78 GB/card vs no-MTP (MTP head workspace). **Practical implication**: for v0.7.3, route Qwen 35B-A3B users to `preview.yml` (no MTP) as the default — MTP gives worse latency despite high acceptance. Re-test after (a) Genesis v7.73.x re-anchors on a post-#42521 nightly (Cliff 2 mitigations may unblock the underlying scheduler overhead), (b) max_num_batched_tokens raised to ~12K to satisfy the vLLM warning, or (c) MTP n=2 to see if smaller spec depth changes the calculus. Compose: `models/qwen3.6-35b-a3b/vllm/compose/dual/preview-mtp.yml`. |
|
||||
| **`gemma-4-26b-a4b/dual/awq.yml` (TP=2)** ⭐ | @noonghunna (2× 3090 PCIe, 230 W cap) | bf16 | 32K | **138.88 / 138.67** (decode 139.92 / 139.98) | — | n/a (no drafter) | n/a | **23.45 GB/card** | 2026-05-15 | **First v0.7.3 Gemma MoE production-track row.** Engine `vllm-nightly-clean` (nightly `bf610c2f`) + vendored [vLLM PR #40886](https://github.com/vllm-project/vllm/pull/40886) overlay (compressed-tensors AWQ MoE key remapping — applied at boot via anchor-based Python patcher in `models/gemma-4-26b-a4b/vllm/patches/vllm-pr40886-awq-moe-keys/install.sh`). Weights: `cyankiwi/gemma-4-26B-A4B-it-AWQ-4bit` (17 GB). CV **0.2% / 0.0%** — extraordinarily stable, characteristic of MoE memory-bandwidth-bound decode. TTFT 53 ms. GPU 0 at 98% util / 360 W, GPU 1 at 59% / 299 W. **Identical narr/code wall TPS (139 / 139)** — same MoE signature as Qwen 3.6 35B-A3B preview above (per-token weight reads dominate, prompt content distribution doesn't matter). ~76% of Qwen 35B-A3B preview's narr TPS — consistent with 4 B vs 3 B active params per forward. Compose: `models/gemma-4-26b-a4b/vllm/compose/dual/awq.yml`. |
|
||||
| **`gemma-4-26b-a4b/dual/awq-mtp.yml` (TP=2, MTP n=4)** ⭐⭐ | @noonghunna (2× 3090 PCIe, 230 W cap) | bf16 | 32K | **155.05 / 207.02** (decode 156.68 / 210.61) | — | **3.04 narr / 3.79 code** | 0.77 / 0.55 / 0.40 / 0.29 (narr); 50.9% avg (narr), 73.5% avg (code) | 23.50 GB/card | 2026-05-15 | **MTP boosts Gemma MoE: +12% narr / +49% code over awq.yml**. Engine `vllm-nightly-clean` + PR #40886 overlay + external `google/gemma-4-26B-A4B-it-assistant` drafter (832 MB BF16). Both GPUs symmetric at 98% util — **the external-drafter path doesn't pay the inter-GPU sync penalty that the Qwen 35B-A3B built-in MTP head pays** (see `preview-mtp.yml` row above where MTP made things 50% slower). Mechanism: external drafter is a small dense model (`Gemma4AssistantForCausalLM`, ~0.5 B params) — the draft forward bypasses MoE expert routing entirely. The 32-49% code lift confirms MTP is structurally compatible with Gemma's hybrid-SWA attention on the AWQ compressed-tensors path. **Recommended default for Gemma 26B-A4B going forward; awq.yml stays as the no-drafter A/B reference.** Compose: `models/gemma-4-26b-a4b/vllm/compose/dual/awq-mtp.yml`. |
|
||||
| `gemma-4-26b-a4b/dual/docker-compose.yml` (TP=2, Intel AutoRound INT4) | @noonghunna (2× 3090 PCIe) | — | — | **boot fail (SM86)** | — | — | — | — | 2026-05-15 | **Ampere-blocked on Intel AutoRound INT4.** `moe_intermediate_size=704` is not a multiple of `group_size=128` (5.5×); Marlin K-dim alignment fails. SM86 has no WNA16 kernel for unaligned K-dim — only SM90+ Cutlass W4A8 / Machete handle arbitrary shapes. Same failure on TP=1 (no split) and TP=2 (split to 352 per rank). Both Intel quant variants (`int4-mixed-AutoRound` and `int4-AutoRound`) hit the same error. **AWQ is the Ampere path** (row above). AutoRound compose preserved here as a documented-blocker for SM90+ rigs (RTX 5090 / Pro 6000 should boot it). |
|
||||
|
||||
|
||||
---
|
||||
|
||||
## Quality benches — Aider Polyglot 30
|
||||
|
||||
@@ -16,6 +16,83 @@ history; SemVer takes over from `v0.3.0` onward.
|
||||
|
||||
---
|
||||
|
||||
## v0.7.2 — 2026-05-15
|
||||
|
||||
|
||||
### ✨ Features
|
||||
|
||||
- feat(launch): add hardware topology advisor ([d116ba9](https://github.com/noonghunna/club-3090/commit/d116ba9ba3442482e0005bab9067d304b40e5e24))
|
||||
|
||||
|
||||
|
||||
[Pin: `git checkout v0.7.2`] · [Full diff](https://github.com/noonghunna/club-3090/compare/v0.7.1...v0.7.2)
|
||||
## v0.7.1 — 2026-05-15
|
||||
|
||||
|
||||
### ✨ Features
|
||||
|
||||
- feat(bench): surface prompt processing throughput ([2a148d7](https://github.com/noonghunna/club-3090/commit/2a148d702b9415129d4c4ec9d3e7d30765927aa4))
|
||||
- feat(llamacpp): expose batch tuning knobs ([02249ab](https://github.com/noonghunna/club-3090/commit/02249ab1939f354ac062d343efefe32677203174))
|
||||
|
||||
|
||||
### 🐛 Bug fixes
|
||||
|
||||
- fix(ci): simplify vllm image workflow, drop smoke-gate (#135) ([ce2617e](https://github.com/noonghunna/club-3090/commit/ce2617e0bc0f56d42caf64e96847d966380be80c))
|
||||
|
||||
|
||||
### 📝 Documentation
|
||||
|
||||
- docs(upstream): PR #42102 closed-as-slop; local overlay permanent ([57eb269](https://github.com/noonghunna/club-3090/commit/57eb269cd70935fc3069b85e46ead8f0f0af13dc))
|
||||
|
||||
|
||||
|
||||
[Pin: `git checkout v0.7.1`] · [Full diff](https://github.com/noonghunna/club-3090/compare/v0.7.0...v0.7.1)
|
||||
## v0.7.0 — 2026-05-14
|
||||
|
||||
|
||||
### ✨ Features
|
||||
|
||||
- feat(scripts): add diagnose-profile triage ([c2adb39](https://github.com/noonghunna/club-3090/commit/c2adb3970dbdcb3bf227cfc7a1ac58a2de3930a4))
|
||||
- feat(compose): use profile-sourced vllm image pins ([e6e33ab](https://github.com/noonghunna/club-3090/commit/e6e33ab37235cdfa2d47987e16e00d617564e55d))
|
||||
- feat(launch): export profile vllm pins ([c306383](https://github.com/noonghunna/club-3090/commit/c3063838bfde3f7811db3edbb9541814aea8f09d))
|
||||
- feat(profiles): resolve vllm nightly pins ([40f1ef7](https://github.com/noonghunna/club-3090/commit/40f1ef78f8e3ab80a5424ce196c7715df85a0d1a))
|
||||
- feat(launch): add estate planner orchestration ([c9b153f](https://github.com/noonghunna/club-3090/commit/c9b153f91f3dd4a076395008ad3bfecdd226a452))
|
||||
- feat(launch): validate single-model profiles ([a142b1c](https://github.com/noonghunna/club-3090/commit/a142b1ce1085c78e8752fa52718af66f7a49f237))
|
||||
- feat(compat): add profile validator and estate self-test ([6581ccc](https://github.com/noonghunna/club-3090/commit/6581ccca9e8613594dba876ba2472e78d48c9eed))
|
||||
- feat(compose): accept ESTATE_GPUS and ESTATE_PORT overrides ([a57596e](https://github.com/noonghunna/club-3090/commit/a57596e0764babc6fe91e96af8b6dd746da45e09))
|
||||
- feat(profiles): ship v0.7.0 data layer ([69825d7](https://github.com/noonghunna/club-3090/commit/69825d7dae0fe1387a04879b31e0d225143b3684))
|
||||
|
||||
|
||||
### 🐛 Bug fixes
|
||||
|
||||
- fix(tools): resolve profile image pins in audit ([98535dc](https://github.com/noonghunna/club-3090/commit/98535dc81453e54be03a1f5915bb2367ebab5c27))
|
||||
- fix(tools): bump engine nightly profiles ([d1acde0](https://github.com/noonghunna/club-3090/commit/d1acde0b28eccb123338d2a9d94e4ef6ff75cc4b))
|
||||
- fix(ci): keep vllm base arg in image metadata ([1abe65f](https://github.com/noonghunna/club-3090/commit/1abe65fe2810be594d6890ce0f26a3f0a554075f))
|
||||
- fix(launch): persist estate source of truth ([52e4347](https://github.com/noonghunna/club-3090/commit/52e43470c6ee111dfa97be81e13e229ede715073))
|
||||
|
||||
|
||||
### 📝 Documentation
|
||||
|
||||
- docs: document profile-sourced vllm pins ([86445be](https://github.com/noonghunna/club-3090/commit/86445be3e8c24ad829d45c4f1702a0bccf27e4dc))
|
||||
- docs: document club vllm image pin ([2ae8303](https://github.com/noonghunna/club-3090/commit/2ae8303833497f20d4079a12226ca23c83f71337))
|
||||
- docs: expand KV_MATH + add ADDING_MODELS workflow ([1f8aaa2](https://github.com/noonghunna/club-3090/commit/1f8aaa2acc72a7cff763fa81ae938d552fffffb2))
|
||||
- docs(hardware): clarify 3090 stock TDP varies by board SKU ([0d59f94](https://github.com/noonghunna/club-3090/commit/0d59f949e472095e3ecb83ce133eb103d10588d9))
|
||||
|
||||
|
||||
### 🧹 Maintenance
|
||||
|
||||
- chore(vllm): use club3090 image in composes ([aebc4f3](https://github.com/noonghunna/club-3090/commit/aebc4f321c3536943bfbd554d2c21ed6824742f9))
|
||||
- chore(ci): build club vllm image ([e88a2a8](https://github.com/noonghunna/club-3090/commit/e88a2a8d21efd3556739fcb0fb1a26327b7d38cd))
|
||||
- refactor(kv-calc): consume profile data ([9ccde62](https://github.com/noonghunna/club-3090/commit/9ccde62abe360bfb9170fc88102623dc5e87597e))
|
||||
|
||||
|
||||
### 🧹 Other
|
||||
|
||||
- Revert "chore(vllm): use club3090 image in composes" ([c7c40bd](https://github.com/noonghunna/club-3090/commit/c7c40bdf1232ec2a2f8e5b1d98249df33026f46b))
|
||||
|
||||
|
||||
|
||||
[Pin: `git checkout v0.7.0`] · [Full diff](https://github.com/noonghunna/club-3090/compare/v0.6.3...v0.7.0)
|
||||
## v0.6.3 — 2026-05-14
|
||||
|
||||
|
||||
|
||||
+2
-2
@@ -93,7 +93,7 @@ For contributors who know the `BENCHMARKS.md` section structure and want to prop
|
||||
```bash
|
||||
bash scripts/bench.sh
|
||||
```
|
||||
Drop the run-by-run output in the PR — `wall_TPS`, `decode_TPS`, `TTFT`, MTP `AL` (where applicable). Mean + CV + n=5 minimum.
|
||||
Drop the run-by-run output in the PR — `wall_TPS`, `decode_TPS`, `PP tok/s`, `TTFT`, MTP `AL` (where applicable). Mean + CV + n=5 minimum.
|
||||
5. **Open the PR with a description that answers four questions:**
|
||||
- What problem does this solve? (One paragraph.)
|
||||
- What's the measured impact? (Numbers.)
|
||||
@@ -127,7 +127,7 @@ Or run the steps individually if you'd rather:
|
||||
CONTAINER=<container-name> ENDPOINT=<http://localhost:port> \
|
||||
bash scripts/soak-test.sh
|
||||
```
|
||||
5. **`bench.sh` run** — 3 warmups + 5 measured runs of narrative + code prompts. Report `wall_TPS`, `decode_TPS`, `TTFT`, peak VRAM/card per run, MTP/DFlash AL where applicable.
|
||||
5. **`bench.sh` run** — 3 warmups + 5 measured runs of narrative + code prompts. Report `wall_TPS`, `decode_TPS`, `PP tok/s`, `TTFT`, peak VRAM/card per run, MTP/DFlash AL where applicable. For llama.cpp or other engines without vLLM prompt-throughput logs, include `PP=1 bash scripts/bench.sh` so the long-prompt fallback captures prompt-processing throughput.
|
||||
6. **BENCHMARKS row** — under the appropriate model section, mirroring existing column shape (incl. `Rig` column). Attribution is automatic.
|
||||
7. **CHANGELOG entry** — in `models/<model>/CHANGELOG.md`.
|
||||
|
||||
|
||||
@@ -44,6 +44,8 @@ Each hardware page lists every supported model with the working composes for tha
|
||||
|---|---|---|---|---|
|
||||
| **[Qwen3.6-27B](models/qwen3.6-27b/)** | Production-ready ⭐ | 1× / 2× 3090 | vLLM ✅ · llama.cpp ✅ · SGLang ❌ blocked | Vision · tools · MTP n=3 · up to 262K ctx · vLLM dual = 89/127 TPS · llama.cpp single = full 262K, no prefill cliffs |
|
||||
| **[Gemma 4 31B](models/gemma-4-31b/)** | Production-ready (dual-card only on Ampere 24 GB) | 2× 3090 only ¹ | vLLM ✅ · llama.cpp ❌ · SGLang ❌ | Vision · tools · MTP n=3 (Google official drafter) **OR** DFlash n=7 (z-lab drafter) · up to 262K ctx via INT8 PTH KV (PR [#40391](https://github.com/vllm-project/vllm/pull/40391) vendored) · MTP dual = 106/141 TPS at 32K, 95/126 at 262K · DFlash dual = 105/177 TPS at 32K (code-optimal) |
|
||||
| **[Qwen3.6 35B-A3B](models/qwen3.6-35b-a3b/)** ⭐ NEW v0.7.3 | Preview (production-track blocked on Genesis v7.73.x) | 2× 3090 | vLLM ✅ (preview) · SGLang ❌ · llama.cpp ❌ | **MoE (256 experts × 8 active, ~3 B active params)** · vision · tools · upstream native loader via [vLLM PR #42521](https://github.com/vllm-project/vllm/pull/42521) · preview dual = **182/177 TPS at 16K** (no MTP, no TQ3, no Genesis) · ~2× the Qwen 3.6-27B dense baseline on the same hardware · production path (Genesis + TQ3 + MTP) pending upstream Genesis re-anchor |
|
||||
| **[Gemma 4 26B-A4B](models/gemma-4-26b-a4b/)** ⭐ NEW v0.7.3 | Production via AWQ (Intel AutoRound INT4 blocked on Ampere) | 2× 3090 | vLLM ✅ (AWQ overlay) · SGLang ❌ · llama.cpp ❌ | **MoE (128 experts × 8 active, ~4 B active params)** · vision · tools · cyankiwi AWQ-4bit weights via vendored [vLLM PR #40886](https://github.com/vllm-project/vllm/pull/40886) (compressed-tensors MoE key remapping) · AWQ dual = **139/139 TPS at 32K**, CV 0.2% / 0.0% · Intel AutoRound variants Ampere-blocked (Marlin K-dim alignment — moe_intermediate_size=704 not aligned to group_size=128) — AutoRound works on SM90+ |
|
||||
|
||||
¹ Single-card boot OOMs on Ampere 24 GB regardless of KV format (weights + drafter + profiling at 8K ctx leaves no KV pool). Single-card Gemma 4 is feasible on 32 GB+ GPUs (validated on RTX 5090 32 GB by [@apnar](https://github.com/noonghunna/club-3090/discussions/67#discussioncomment-16832042)).
|
||||
|
||||
@@ -55,7 +57,7 @@ More models coming. The repo structure scales — when we add Qwen3.5-27B / GLM-
|
||||
|
||||

|
||||
|
||||
Bench protocol: 3 warm + 5 measured runs of the canonical narrative + code prompts. Substrate: vLLM nightly `0.20.1rc1.dev16+g7a1eb8ac2` + Genesis v7.69 dev tip (commit `2db18df`), with local backports `patch_inputs_embeds_optional.py` (vllm#35975) and `patch_tolist_cudagraph.py`. llama.cpp mainline `0d0764dfd`, RTX 3090 sm_86 PCIe-only at 230 W. Per-config details + run-by-run numbers + VRAM + AL/accept rates: [models/qwen3.6-27b/CHANGELOG.md](models/qwen3.6-27b/CHANGELOG.md) (per-model history) and [scripts/bench.sh](scripts/bench.sh) (canonical bench).
|
||||
Bench protocol: 3 warm + 5 measured runs of the canonical narrative + code prompts. `scripts/bench.sh` reports wall TPS, decode TPS, TTFT, and prompt-processing throughput (`PP tok/s`; vLLM log scrape, `PP=1` long-prompt fallback for llama.cpp). Substrate: vLLM nightly `0.20.1rc1.dev16+g7a1eb8ac2` + Genesis v7.69 dev tip (commit `2db18df`), with local backports `patch_inputs_embeds_optional.py` (vllm#35975) and `patch_tolist_cudagraph.py`. llama.cpp mainline `0d0764dfd`, RTX 3090 sm_86 PCIe-only at 230 W. Per-config details + run-by-run numbers + VRAM + AL/accept rates: [models/qwen3.6-27b/CHANGELOG.md](models/qwen3.6-27b/CHANGELOG.md) (per-model history) and [scripts/bench.sh](scripts/bench.sh) (canonical bench).
|
||||
|
||||
---
|
||||
|
||||
|
||||
+88
-99
@@ -1,127 +1,116 @@
|
||||
# Club-3090 CI GPU Runner Setup
|
||||
# Club-3090 vLLM Image — Build + Distribution
|
||||
|
||||
The vLLM image workflow builds on GitHub-hosted Ubuntu, then optionally smokes the
|
||||
fresh image on a self-hosted GPU runner. If no runner is registered, the workflow
|
||||
still pushes the dated `nightly-YYYYMMDD-clubXXXX` image and leaves `latest` /
|
||||
`nightly-stable` untouched.
|
||||
The vLLM image workflow (`.github/workflows/build-vllm-image.yml`) builds on
|
||||
GitHub-hosted Ubuntu, applies vendored overlays via Dockerfile, pushes a dated
|
||||
nightly tag to GHCR, then promotes the `:latest` and `:nightly-stable` aliases
|
||||
to point at the just-built dated tag. No self-hosted runner required.
|
||||
|
||||
## Runner Requirements
|
||||
The release-pinned tag `:club-vX.Y.Z` is created automatically when the workflow
|
||||
runs on a Git tag push (e.g. `v0.7.0`).
|
||||
|
||||
- Linux x86_64 host with Docker Engine and Docker Compose v2.
|
||||
- NVIDIA driver and NVIDIA Container Toolkit installed.
|
||||
- At least two 24 GB NVIDIA GPUs for the canonical `qwen3.6-27b/vllm/dual`
|
||||
smoke. The production validation target is 2x RTX 3090.
|
||||
- Enough local storage for the vLLM image, model cache, Docker layers, and
|
||||
compile caches. Plan for at least 250 GB free.
|
||||
- A dedicated runner host. Do not run untrusted pull-request jobs on this
|
||||
machine.
|
||||
## Tag conventions
|
||||
|
||||
## Labels
|
||||
| Tag | Mutable? | What it represents | Recommended for |
|
||||
|---|---|---|---|
|
||||
| `nightly-YYYYMMDD-clubNNNN` | Immutable | A specific build of upstream nightly + our overlays | Reproducibility; pin in `VLLM_IMAGE` to lock to a known state |
|
||||
| `:latest` | Mutable | The most-recent dated nightly | Users who want bleeding-edge; expect occasional breakage |
|
||||
| `:nightly-stable` | Mutable | Same target as `:latest` today; reserved for future divergence | Same as `:latest` for now |
|
||||
| `:club-vX.Y.Z` | Immutable (per release) | Built on the v0.7.0 tag push and never moves | Users who want verified releases; this is the recommended path |
|
||||
|
||||
Register the runner with the normal self-hosted labels plus `gpu`:
|
||||
## Why no smoke-gating?
|
||||
|
||||
```text
|
||||
self-hosted
|
||||
linux
|
||||
x64
|
||||
gpu
|
||||
```
|
||||
The workflow originally tried to gate `:latest` promotion on a self-hosted GPU
|
||||
runner running `verify-full.sh` + a 3-prompt smoke. That added:
|
||||
|
||||
The workflow checks for an online runner with `self-hosted` and `gpu`; the smoke
|
||||
job itself targets `[self-hosted, linux, x64, gpu]`.
|
||||
- A dependency on infra (the runner) we don't maintain.
|
||||
- A permission requirement (`administration: read` to list runners) the default `GITHUB_TOKEN` lacks.
|
||||
- A failure mode where `:latest` never gets promoted if no runner is registered.
|
||||
|
||||
## Registration
|
||||
We dropped smoke-gating in v0.7.1 (issue #135) because:
|
||||
|
||||
1. Open the GitHub repository.
|
||||
2. Go to **Settings -> Actions -> Runners -> New self-hosted runner**.
|
||||
3. Choose Linux x64 and follow GitHub's generated commands.
|
||||
4. Add the `gpu` label during configuration, or add it later from the runner UI.
|
||||
5. Install the runner as a service:
|
||||
1. The build step itself catches the most common failure modes (missing patch
|
||||
source, invalid Dockerfile, overlay path mismatches).
|
||||
2. Users who want verified images use `:club-vX.Y.Z` release tags, not `:latest`.
|
||||
3. The Docker Hub convention is that `:latest` = "most recent, no guarantees".
|
||||
|
||||
If a self-hosted GPU runner ever becomes available, smoke-gating can be layered
|
||||
on top of this workflow as a separate post-build job — the simpler design today
|
||||
doesn't preclude it.
|
||||
|
||||
## Using the GHCR image
|
||||
|
||||
The pre-built GHCR image is **opt-in**. Default launches use the upstream vLLM
|
||||
nightly SHA resolved from `scripts/lib/profiles/engines/<engine-id>.yml →
|
||||
install.spec` (the standard `vllm/vllm-openai:nightly-<sha>` ref).
|
||||
|
||||
To use the verified release image:
|
||||
|
||||
```bash
|
||||
sudo ./svc.sh install
|
||||
sudo ./svc.sh start
|
||||
VLLM_IMAGE=ghcr.io/noonghunna/vllm-club3090:club-v0.7.0 \
|
||||
bash scripts/launch.sh --variant vllm/dual
|
||||
```
|
||||
|
||||
The runner user must be able to run Docker commands. On a typical Ubuntu host:
|
||||
To use the bleeding-edge `:latest` (rebuilt on every overlay change + weekly):
|
||||
|
||||
```bash
|
||||
sudo usermod -aG docker "$USER"
|
||||
newgrp docker
|
||||
VLLM_IMAGE=ghcr.io/noonghunna/vllm-club3090:latest \
|
||||
bash scripts/launch.sh --variant vllm/dual
|
||||
```
|
||||
|
||||
Restart the runner service after changing group membership.
|
||||
|
||||
## Host Preflight
|
||||
|
||||
Run these on the runner host before enabling the smoke job:
|
||||
|
||||
```bash
|
||||
nvidia-smi
|
||||
docker compose version
|
||||
docker run --rm --gpus all nvidia/cuda:12.8.0-base-ubuntu24.04 nvidia-smi
|
||||
```
|
||||
|
||||
Then clone this repository at the path used by the runner workspace once and
|
||||
make sure the model cache is present or mounted at the compose default:
|
||||
|
||||
```bash
|
||||
ls -ld models-cache
|
||||
```
|
||||
|
||||
If your cache lives elsewhere, set `MODEL_DIR` in the runner service
|
||||
environment. The canonical compose reads `${MODEL_DIR:-../../../../../models-cache}`.
|
||||
|
||||
## What The Smoke Job Does
|
||||
|
||||
On a green build, the workflow:
|
||||
|
||||
1. Pulls `ghcr.io/noonghunna/vllm-club3090:nightly-YYYYMMDD-clubXXXX`.
|
||||
2. Boots `models/qwen3.6-27b/vllm/compose/dual/docker-compose.yml` with a
|
||||
temporary compose override that points at the dated image.
|
||||
3. Waits for `http://localhost:8010/v1/models`.
|
||||
4. Runs `bash scripts/verify-full.sh`.
|
||||
5. Runs a three-prompt OpenAI-compatible smoke bench.
|
||||
6. Only then retags the image as `latest` and `nightly-stable`.
|
||||
|
||||
If any smoke step fails, the dated image remains available for debugging and the
|
||||
rolling aliases do not move.
|
||||
|
||||
## Using The GHCR Image
|
||||
|
||||
The pre-built GHCR image is opt-in. Normal launches use the upstream vLLM
|
||||
nightly SHA resolved from `scripts/lib/profiles/engines/<engine-id>.yml`.
|
||||
|
||||
To force a verified club image after this workflow has moved `latest` forward:
|
||||
|
||||
```bash
|
||||
VLLM_IMAGE=ghcr.io/noonghunna/vllm-club3090:latest bash scripts/launch.sh --variant vllm/dual
|
||||
```
|
||||
|
||||
`VLLM_IMAGE` is a full image reference override. The launcher still exports
|
||||
`VLLM_IMAGE` is a full image-reference override. The launcher still exports
|
||||
`VLLM_NIGHTLY_SHA` from the matching EngineProfile, but Docker Compose uses
|
||||
`VLLM_IMAGE` first.
|
||||
`VLLM_IMAGE` first when present.
|
||||
|
||||
## Existing Containers
|
||||
## Workflow triggers
|
||||
|
||||
The runner should be dedicated to CI. Before booting the canonical compose, the
|
||||
workflow tears down the default club-3090 estate if `~/.club3090/estate.yml`
|
||||
exists, then runs `docker compose down` for the CI project name. Avoid running
|
||||
manual workloads on the same host while the workflow is active.
|
||||
The workflow runs on:
|
||||
|
||||
## Registry Permissions
|
||||
- **`workflow_dispatch`** — manual `gh workflow run build-vllm-image.yml` for ad-hoc rebuilds
|
||||
- **`schedule`** — weekly, Sunday 00:00 UTC
|
||||
- **`push`** to master when `.github/workflows/build-vllm-image.yml`, `docker/vllm-club3090/**`, or `models/*/vllm/patches/**` change
|
||||
- **`push`** of any tag matching `v0.7.*`, `v0.[8-9].*`, or `v[1-9]*` — adds the `:club-vX.Y.Z` release tag
|
||||
|
||||
The workflow uses `GITHUB_TOKEN` with `packages: write` to push GHCR images and
|
||||
move aliases. No personal access token is required for the repository-owned
|
||||
package.
|
||||
## Workflow permissions
|
||||
|
||||
`GITHUB_TOKEN` with default `contents: read` + `packages: write` is sufficient.
|
||||
No `administration: read` is needed (we removed the runner detection that
|
||||
required it). No personal access tokens required.
|
||||
|
||||
## Manual `:latest` bootstrap (recovery)
|
||||
|
||||
If `:latest` ever gets out of sync with the most-recent dated nightly (e.g.
|
||||
during this workflow's redesign migration), you can manually re-tag a dated
|
||||
nightly as `:latest` without re-building:
|
||||
|
||||
```bash
|
||||
docker login ghcr.io --username "${GITHUB_USERNAME}" --password-stdin <<< "${GITHUB_TOKEN}"
|
||||
|
||||
docker buildx imagetools create \
|
||||
-t ghcr.io/noonghunna/vllm-club3090:latest \
|
||||
-t ghcr.io/noonghunna/vllm-club3090:nightly-stable \
|
||||
ghcr.io/noonghunna/vllm-club3090:nightly-YYYYMMDD-clubXXXX
|
||||
```
|
||||
|
||||
Replace `nightly-YYYYMMDD-clubXXXX` with the dated tag you want to bless as
|
||||
`:latest`. List recent tags via:
|
||||
|
||||
```bash
|
||||
gh api -H 'Accept: application/vnd.github+json' \
|
||||
/users/noonghunna/packages/container/vllm-club3090/versions | \
|
||||
jq -r '.[] | .metadata.container.tags[]' | head -20
|
||||
```
|
||||
|
||||
(This requires `read:packages` scope on the token. Without it, list tags via the
|
||||
GHCR web UI: https://github.com/noonghunna/club-3090/pkgs/container/vllm-club3090.)
|
||||
|
||||
## Retention
|
||||
|
||||
The scheduled workflow keeps:
|
||||
|
||||
- `latest`
|
||||
- `nightly-stable`
|
||||
- every `club-v*` release tag
|
||||
- dated `nightly-YYYYMMDD-clubXXXX` tags from the last four weeks
|
||||
- `:latest`
|
||||
- `:nightly-stable`
|
||||
- every `:club-v*` release tag
|
||||
- dated `nightly-YYYYMMDD-clubNNNN` tags from the last four weeks
|
||||
|
||||
Older dated nightly package versions are deleted by the retention job.
|
||||
Older dated nightly versions are deleted by the retention job (runs only on the
|
||||
weekly schedule + `workflow_dispatch`, not on every push).
|
||||
|
||||
@@ -45,6 +45,8 @@ Plus per-card model weights drop from ~14 GB to ~7 GB (sharded), KV cache from f
|
||||
|
||||
Cost paid for this: NCCL allreduce per layer between cards (~30-50µs/token on PCIe), ~10-20% TPS overhead vs single-card if single-card actually worked.
|
||||
|
||||
> **Reading soak-test results for TP=2 / llama.cpp configs.** A clean `verdict PASS` on `dual.yml` / `dual-turbo.yml` / `llamacpp/default` does NOT mean the Cliff 2 mitigation patches in the compose's overlay set (PN-* sidecars, FLA chunked-prefill stabilizers, etc.) are doing the work — the topology alone takes that failure mode off the table. PASS on TP=2 reflects "the configuration is stable end-to-end at this depth," not "patches X/Y are load-bearing here." For per-patch attribution, run the same soak with overlays stripped and compare. See `scripts/soak-test.sh --help` ("PASS VERDICT" block) and [#140](https://github.com/noonghunna/club-3090/issues/140).
|
||||
|
||||
## Why llama.cpp escapes Cliff 2b on a single card
|
||||
|
||||
llama.cpp uses **different kernels and a different memory allocator** than vLLM. Three concrete differences:
|
||||
|
||||
@@ -390,6 +390,25 @@ If you're on **SM89+ hardware (RTX 4090 / 5090, A6000 Ada / Blackwell)**, the pe
|
||||
|
||||
---
|
||||
|
||||
## Note for older host platforms (PCIe Gen 3 + older CPUs)
|
||||
|
||||
If your rig is on **PCIe Gen 3** (rather than Gen 4) **and/or paired with a pre-Zen3 / pre-2018 CPU** (e.g. Xeon Gold 61xx Skylake, Xeon E5 v4 Broadwell), TP=2 paths take a 30-40% throughput hit vs the Gen 4 / Ryzen 5950X / EPYC rigs in `BENCHMARKS.md`. Two compounding causes:
|
||||
|
||||
1. **PCIe Gen 3 x16 ≈ 15.75 GB/s** per direction vs Gen 4 x16 ≈ 31.5 GB/s. TP=2 all-reduce on the residual stream every layer is GB/s-class traffic — halving interconnect bandwidth roughly halves the all-reduce wall time, and decode-TPS is sensitive to that.
|
||||
2. **Older Xeon / Broadwell CPUs** have lower per-core clock and IPC than current Ryzen / EPYC parts. Affects prefill throughput, TTFT, and host-side coordination between the two GPUs.
|
||||
|
||||
**Symptom**: GPU utilization asymmetry during decode (e.g. `GPU 0: 28% util / 174W` vs `GPU 1: 85% util / 254W`) — communication-starved TP=2, where one card finishes its half-step and stalls waiting on all-reduce.
|
||||
|
||||
**Mitigation on Gen 3 rigs**:
|
||||
|
||||
- **Enable persistence mode** (`sudo nvidia-smi -pm 1`) — common to find this off on KVM/VM hosts; with it disabled the driver tears down between idle periods and adds per-request init latency.
|
||||
- **Prefer single-card paths**: with interconnect being the bottleneck, `vllm/minimal` (single-card fp8 KV, no MTP) or `vllm/long-text-no-mtp` (single-card TQ3 KV) often beats `dual.yml` on these rigs. You give up max context ceiling but get back the decode TPS the interconnect was eating.
|
||||
- **More host RAM** if VM-passthrough: 32+ GB recommended; vLLM uses host RAM for tokenizer staging, paged weight loading, and IPC buffers — VMs with 15 GB total tend to thrash.
|
||||
|
||||
See [issue #137](https://github.com/noonghunna/club-3090/issues/137) for a worked example: Xeon Gold 6138 + PCIe Gen 3 x16 + 2× 3090 (KVM passthrough) → 32 / 41 TPS on `dual.yml`, vs Ryzen 5950X + Gen 4 + same KV config → 89 / 117 TPS ([@lolren disc #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16820303)).
|
||||
|
||||
---
|
||||
|
||||
## Note for WSL2 / Windows users
|
||||
|
||||
### GPU memory budget on WSL2
|
||||
|
||||
@@ -0,0 +1,71 @@
|
||||
# Kernel & Attention Backend Matrix (Mid-2026)
|
||||
|
||||
**Last updated:** 2026-05-15
|
||||
**Focus:** Consumer / Prosumer GPUs (RTX 3090 → 5090) + hybrid/MoE models (Qwen3.6, Gemma4)
|
||||
|
||||
This matrix helps decide which inference engine + kernel combination to route composes to.
|
||||
|
||||
## Core Modern Kernels
|
||||
|
||||
| Kernel / Backend | Primary Purpose | Best Hardware | Key Strengths | Maturity on Ampere (3090) |
|
||||
|-------------------------------|----------------------------------------|----------------------------|--------------------------------------------|---------------------------|
|
||||
| **FlashAttention-2** | Standard attention | Ampere+ | Excellent balance, wide compatibility | Very High |
|
||||
| **FlashAttention-3** | Hopper/Blackwell optimized | H100/B200+ | FP8, asynchrony, TMA/WGMMA | Medium (falls back) |
|
||||
| **FlashInfer** | Flexible paged / custom attention | All NVIDIA | High performance, Triton-based, very tunable | Very High |
|
||||
| **RadixAttention** | Prefix sharing (radix tree) | All | Best for chat, RAG, agents | High |
|
||||
| **PagedAttention** (vLLM) | Memory-efficient KV management | All | Low fragmentation, high concurrency | Very High |
|
||||
| **Triton Custom Kernels** | Rapid prototyping / specialized ops | All | Easy to write, high flexibility | High |
|
||||
| **TensorRT-LLM Kernels** | Deep fusion + hardware-specific | NVIDIA (best on Ada+) | Highest raw speed on supported hardware | High |
|
||||
| **MLA / FlashMLA** | Multi-head Latent Attention (DeepSeek/Qwen) | All | Optimized for compressed KV | High |
|
||||
|
||||
## KV Cache Impact
|
||||
|
||||
Different kernels affect KV cache size/efficiency differently. Some only speed up attention computation (no KV size change); others fundamentally change how the cache is laid out, paged, or shared across requests.
|
||||
|
||||
| Kernel / System | Impacts KV Cache Size? | Main Impact on KV Cache | Best For | Notes for 3090 / Consumer |
|
||||
|-----------------------------------|------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------|---------------------------------------|------------------------------------------|
|
||||
| **FlashAttention-2 / -3** | No | Reduces HBM traffic during attention computation (IO-aware tiling). Makes attention faster without materializing full attention matrix. | Speed (especially prefill) | FA2 is excellent on Ampere. FA3 falls back. |
|
||||
| **FlashInfer** | No | Highly optimized paged + custom attention kernels. Excellent at handling non-contiguous KV blocks with minimal overhead. | Flexibility + speed on all NVIDIA | Often the fastest backend on 3090. |
|
||||
| **PagedAttention** (vLLM) | Yes (effective) | Breaks KV cache into small pages/blocks. Dramatically reduces fragmentation → much higher real-world memory utilization. | High concurrency + long context | Core reason vLLM can serve many users. |
|
||||
| **RadixAttention** (SGLang) | Yes (very strong) | Stores KV cache in a prefix tree (radix trie). Shares common prefixes across requests → massive memory savings in chat/RAG/agent workloads. | Prefix-heavy workloads | Biggest win for multi-turn / agents. |
|
||||
| **TensorRT-LLM kernels** | No | Highly fused, hardware-specific kernels. Excellent with FP8/TQ3 but same base KV size. | Raw speed on new NVIDIA cards | Less flexible on 3090. |
|
||||
| **Block Diffusion** (DFlash/Zaya) | Indirect but big | Generates multiple tokens per forward pass → fewer total KV updates per output token. KV cache grows slower in practice. | Throughput on bandwidth-limited cards| Very promising for 3090. |
|
||||
|
||||
## Engine Support Matrix
|
||||
|
||||
| Feature / Kernel | **vLLM** | **SGLang** | **TensorRT-LLM** | **llama.cpp** | Notes for 3090 / Consumer |
|
||||
|-----------------------------------|-----------------------------------|-------------------------------------|------------------------------------|--------------------------------|---------------------------|
|
||||
| **Default Attention** | FlashAttention / FlashInfer | FlashInfer (preferred) | Custom TRT kernels + FA3 | Basic FA2 / custom | FlashInfer best on 3090 |
|
||||
| **RadixAttention / Prefix Cache**| APC (Automatic Prefix Caching) | **Native RadixAttention** (best) | Limited | Basic | SGLang wins for agents/RAG |
|
||||
| **Paged KV Cache** | **Native PagedAttention** | Yes | Yes | Basic | vLLM strongest |
|
||||
| **FlashAttention-3** | Good (Hopper+) | Good | **Best** | No | Falls back on Ampere |
|
||||
| **DFlash (Block Diffusion)** | Good | **Excellent** (early + deep) | Emerging | Limited | SGLang currently strongest |
|
||||
| **MTP / Speculative Decoding** | Strong (but TQ3 issues) | Strong | Excellent | Good | vLLM + MTP works well |
|
||||
| **TQ3 / Advanced KV Quant** | Yes (but MTP broken) | Partial / WIP | Excellent | Good (GGUF) | vLLM best but buggy |
|
||||
| **MoE / Hybrid Models** | Very Good | Excellent (esp. Qwen/DeepSeek) | Excellent | Decent | SGLang & TRT-LLM shine |
|
||||
| **Structured Output** | Good | **Best-in-class** | Good | Basic | SGLang wins |
|
||||
| **3090 / Ampere Optimization** | Solid | **Strong** | Good | Very Good | SGLang + FlashInfer often fastest |
|
||||
|
||||
## Recommendations for club-3090 / Composes
|
||||
|
||||
### Primary Routing Guidance
|
||||
|
||||
- **Interactive / Chat / Agents / RAG** → **SGLang** (RadixAttention + DFlash)
|
||||
- **Max raw throughput on new cards (Ada/Blackwell)** → **TensorRT-LLM**
|
||||
- **Broad compatibility + stability on 3090** → **vLLM** (with FlashInfer backend)
|
||||
- **Low VRAM / single card / maximum compression** → **llama.cpp** (accept GGUF tradeoffs)
|
||||
- **Hybrid MoE models (Qwen3.6 35B-A3B, Gemma4 26B)** → **SGLang** or **vLLM** with FlashInfer
|
||||
|
||||
### 3090-Specific Tips
|
||||
- Use **FlashInfer** backend wherever possible (`--attention-backend flashinfer` in vLLM/SGLang).
|
||||
- TQ3 + MTP is currently unstable in vLLM → prefer FP8/INT8 KV or DFlash.
|
||||
- SGLang + RadixAttention + DFlash often gives the best real-world experience on consumer hardware.
|
||||
|
||||
---
|
||||
|
||||
**Cross-references**:
|
||||
- See [DTYPE_MATRIX.md](DTYPE_MATRIX.md) for quantization + hardware accelerators
|
||||
- See [KV_MATH.md](KV_MATH.md) for cache calculations per model
|
||||
- See [INFERENCE_ENGINES.md](INFERENCE_ENGINES.md) for high-level engine comparison & setup
|
||||
|
||||
Contributions & updates welcome — this space moves fast!
|
||||
+189
-72
@@ -8,10 +8,10 @@ Four model families are documented:
|
||||
|---|---|---|
|
||||
| **Qwen 3.6 27B** (dense) | Calibrated 11/11 on this stack | Qwen3-Next hybrid: 16 full-attention + 48 GDN (Gated DeltaNet) layers |
|
||||
| **Gemma 4 31B** (dense) | Calibrated 7/7 on this stack | Sliding-window + dense MLP: 50 SWA + 10 full-attention layers |
|
||||
| **Qwen 3.6 35B-A3B** (MoE) | Math-ready, calibration pending | Qwen3-Next hybrid + MoE: 30 GDN + 10 gated-attention layers |
|
||||
| **Gemma 4 26B-A4B** (MoE) | Math-ready, calibration pending | Sliding-window + dense MoE: SWA + occasional global layers |
|
||||
| **Qwen 3.6 35B-A3B** (MoE) | **Config-verified, calibration pending** | Qwen3-Next hybrid + MoE: 30 GDN + 10 gated-attention layers. Confirmed from `config.json` 2026-05-15 — see [Qwen section](#qwen-36-35b-a3b-moe--per-card-budget-components). |
|
||||
| **Gemma 4 26B-A4B** (MoE) | **Config-verified, calibration pending** | Sliding-window + dense MoE: 25 SWA + 5 full-attention layers. **Asymmetric KV heads** (8 sliding / 2 global). Confirmed from `config.json` 2026-05-15 — see [Gemma section](#gemma-4-26b-a4b-moe--per-card-budget-components). |
|
||||
|
||||
For models marked **calibration pending**, the architectural math derivations below are anchored to published config + model card material; absolute numbers ship as estimates until we measure them on the stack. See [Sources of Error & Accuracy](#sources-of-error--accuracy) at the end.
|
||||
For models marked **config-verified, calibration pending**: architectural facts (layer counts, head dims, K=V tying, MoE expert counts, layer-type pattern) are sourced directly from the on-disk `config.json` and `layer_types` arrays — not estimates. What remains pending is the **empirical activation-peak coefficient** for each (model, KV-format) pair, which needs ≥4 measured BENCHMARKS rows per model. See [Sources of Error & Accuracy](#sources-of-error--accuracy) at the end.
|
||||
|
||||
## TL;DR
|
||||
|
||||
@@ -84,16 +84,70 @@ peak ≈ weights/TP ← exact, from checkpoint
|
||||
+ drafter_overhead/TP ← speculative-decoding drafter weights, if any
|
||||
```
|
||||
|
||||
### DeltaNet recurrent state (DeltaNet-family models only)
|
||||
|
||||
Hybrid DeltaNet models (Qwen3-Next family) maintain a fixed-size recurrent state between tokens, separate from the per-token growing KV. This state is tiny but worth noting for completeness:
|
||||
|
||||
```
|
||||
delta_state_bytes ≈ num_gdn_layers
|
||||
× (linear_num_k_heads × linear_k_head_dim
|
||||
+ linear_num_v_heads × linear_v_head_dim
|
||||
+ linear_conv_kernel_dim × (linear_num_k_heads × linear_k_head_dim
|
||||
+ linear_num_v_heads × linear_v_head_dim))
|
||||
× 4 (fp32, mamba_ssm_dtype)
|
||||
× max_num_seqs
|
||||
```
|
||||
|
||||
Three components per layer: K state + V state + conv1d kernel state. All four `linear_*` fields are in `config.json → text_config`. Concrete sizes for our models in per-model §"DeltaNet recurrent state" subsections — typically single-digit MB total, negligible vs activation peak.
|
||||
|
||||
**Worked example — Qwen 3.6 35B-A3B at `max_num_seqs=1`:**
|
||||
|
||||
```
|
||||
linear_num_k_heads = 16, linear_k_head_dim = 128 → 2,048 elements per layer
|
||||
linear_num_v_heads = 32, linear_v_head_dim = 128 → 4,096 elements per layer
|
||||
linear_conv_kernel_dim = 4 → conv state = 4 × (2,048 + 4,096) = 24,576 elements
|
||||
num_gdn_layers = 30, fp32 (4 bytes), max_num_seqs = 1
|
||||
|
||||
delta_state_bytes = 30 × (2,048 + 4,096 + 24,576) × 4 × 1
|
||||
= 30 × 30,720 × 4
|
||||
= 3,686,400 bytes
|
||||
≈ 3.5 MB
|
||||
```
|
||||
|
||||
Plug `max_num_seqs = 4` → ~14 MB. Both well below activation-peak scale.
|
||||
|
||||
Per-model sections below derive each term concretely.
|
||||
|
||||
## Quick reference: per-token growing-KV bytes
|
||||
|
||||
Headline numbers for the four shipping models, computed from each per-model formula in the deep sections. Useful for at-a-glance capacity planning.
|
||||
|
||||
| Model | bf16 (TP=1 / TP=2) | fp8_e5m2 (TP=1 / TP=2) | INT8 PTH or TQ3 (TP=1 / TP=2) | Vs Qwen 27B† |
|
||||
|---|---:|---:|---:|---:|
|
||||
| Qwen 3.6 27B | 65,536 B / 32,768 B | 32,768 B / 16,384 B | **13,927 B / 6,963 B** (TQ3) | **1.00×** (baseline) |
|
||||
| Qwen 3.6 35B-A3B (MoE) | 20,480 B / 10,240 B | 10,240 B / 5,120 B | **4,352 B / 2,176 B** (TQ3) | **0.31×** (~3.2× lighter) |
|
||||
| Gemma 4 31B | 163,840 B / 81,920 B | 81,920 B / 40,960 B | ~82,700 B / ~41,400 B (INT8 PTH) | **2.50×** (~2.5× heavier) |
|
||||
| Gemma 4 26B-A4B (MoE) | **10,240 B / 5,120 B** | 5,120 B / 2,560 B | ~5,170 B / ~2,585 B (INT8 PTH) | **0.16×** (~6.4× lighter) |
|
||||
|
||||
† Ratio at fp8_e5m2 TP=2 — pick this as the comparison anchor because it's a common production config. Ratios shift slightly under other formats but the family hierarchy is stable.
|
||||
|
||||
**What jumps out:**
|
||||
|
||||
- **Gemma 4 26B-A4B vs 31B**: ~16× smaller per-token growing KV thanks to asymmetric KV head counts (2 global vs 16). At 200K context + fp8 + TP=2, growing KV per card is ~512 MB for the MoE vs ~8 GB for the 31B. Long-context serving on 24 GB Ampere is dramatically cheaper.
|
||||
- **Qwen 3.6 35B-A3B vs 27B**: ~3.2× smaller per token (10 growing layers × 2 KV heads vs 16 × 4). The MoE shifts the bottleneck from KV to weights + activation.
|
||||
- **Sliding-window KV** for Gemma models is **fixed** (not per-token): ~50 MB total (26B-A4B) / ~200 MB total (31B) at bf16. Excluded from per-token math but included in the per-model deep sections.
|
||||
- **TQ3 (Genesis) only applies to Qwen-family** (DeltaNet kernel dependency); **INT8 PTH (PR #40391/#42102) is the long-context unlock for Gemma family on Ampere**.
|
||||
|
||||
## Model architecture summary
|
||||
|
||||
| Model | Total layers | Growing layers | Sliding / fixed | KV heads | Head dim | K=V tied | MoE | Special notes |
|
||||
|---|---:|---:|---:|---:|---:|:---:|:---:|---|
|
||||
| **Qwen 3.6 27B** | 64 | 16 (full-attention) | 48 (GDN recurrent) | 4 | 256 | No (×2) | No | DeltaNet block-wise activation peak (Cliff 2). `linear_attn` in-proj stays fp16 even under INT4 quant. |
|
||||
| **Qwen 3.6 35B-A3B** | 40 | 10 (gated attention) | 30 (Gated DeltaNet) | 2 | 256 | No (×2) | Yes | Pattern: `10 × (3× GDN → MoE → 1× Gated Attn → MoE)`. Active params ~3B, total 35B. MoE experts gate decode FLOPs but not KV size. |
|
||||
| **Qwen 3.6 35B-A3B** | 40 | **10** (gated attention at idx 3,7,11,15,19,23,27,31,35,39) | 30 (Gated DeltaNet) | **2** | 256 | No (×2) | **Yes (256×8)** | `full_attention_interval=4`: every 4th layer is attention. Built-in MTP (`mtp_num_hidden_layers=1`). `attn_output_gate=True` (gated attention). Vision-capable. Active params ~3B, total 35B. |
|
||||
| **Gemma 4 31B** | 60 | 10 (full-attention) | 50 (SWA, window=1024) | 16 | 256 sliding / **512 global** | Yes (×1) | No | Global layers use 2× head_dim of sliding layers. K=V tying confirmed empirically against boot-log KV cache reports. |
|
||||
| **Gemma 4 26B-A4B** | TBD (likely ~40-48) | TBD (likely sparse global) | Majority SWA (window=1024) | TBD | TBD | Likely yes (×1) | Yes | A4B = 4B active params from 26B total. Layer pattern from model card README. **Numbers pending config.json + first boot.** |
|
||||
| **Gemma 4 26B-A4B** | **30** | **5** (full-attention at idx 5,11,17,23,29) | 25 (SWA, window=1024) | **8 sliding / 2 global** (asymmetric) | 256 sliding / **512 global** | **Yes (×1)** | **Yes (128×8)** | Asymmetric KV-head split per layer type. Every 6th layer is global, last layer always global. Per-token growing KV is **~16× smaller** than Gemma 4 31B (see [Gemma section](#gemma-4-26b-a4b-moe--per-card-budget-components)). Vision + audio support. **No Genesis required.** |
|
||||
|
||||
> **MoE column format**: `N×K` = `num_experts × num_experts_per_tok` (e.g. "256×8" = 256 experts, 8 active per token).
|
||||
|
||||
**Hybrid quirks to internalize:**
|
||||
|
||||
@@ -208,7 +262,7 @@ In the Qwen3-Next hybrid architecture, **only the 16 full_attention layers contr
|
||||
Applying the general formula:
|
||||
|
||||
```
|
||||
per_token_bytes = 16 (growing layers) × 4 (kv_heads) × 256 (head_dim) × 2 (no K=V tie) × bpe
|
||||
per_token_bytes = 16 (growing layers) × 4 (kv_heads) × 256 (head_dim) × k_v_tensors=2 × bpe
|
||||
= 32,768 × bpe bytes
|
||||
```
|
||||
|
||||
@@ -268,36 +322,54 @@ overhead = 0.5 + 1.0 × mem_util + 0.3 × (TP - 1) # GB
|
||||
|
||||
This is rough — actual overhead depends on how many graphs vLLM captures, which depends on `max_num_seqs`, `compile_sizes`, and other internals.
|
||||
|
||||
### 5. DFlash draft model
|
||||
### 5. DeltaNet recurrent state (per-stream, constant)
|
||||
|
||||
The 48 GDN layers maintain a fixed-size recurrent state between tokens (separate from the block-wise intermediate during forward — that's the activation peak in §3). Concrete size for Qwen 3.6 27B:
|
||||
|
||||
- K state: `16 × 128 × fp32 = 8 KB` per layer
|
||||
- V state: `48 × 128 × fp32 = 24 KB` per layer
|
||||
- Conv state: `4 × (16×128 + 48×128) × fp32 = ~128 KB` per layer
|
||||
- **Total per layer: ~160 KB** × 48 layers × `max_num_seqs` streams
|
||||
|
||||
At `max_num_seqs=1`: ~7.5 MB total per card. At `max_num_seqs=4`: ~30 MB. Negligible vs activation peak (GB-scale) and KV pool (sub-GB). Listed for completeness; don't model in budget projections.
|
||||
|
||||
### 6. DFlash draft model
|
||||
|
||||
Only present on `dual-dflash*.yml` composes. `z-lab/Qwen3.6-27B-DFlash` is a ~1.75 GB draft model (per card, FP16). With TP > 1, the draft itself is sharded.
|
||||
|
||||
## Qwen 3.6 35B-A3B (MoE) — per-card budget components
|
||||
|
||||
**Status**: math-ready, **calibration pending** (not yet served on this stack). Numerical values below are derived from the architecture pattern in the model card; expect re-calibration once we measure boot peaks.
|
||||
**Status**: **config-verified** (architecture confirmed from on-disk `config.json` 2026-05-15), **calibration pending** (not yet served on this stack — activation coefficients TBD). All architectural numbers below are sourced from the model checkpoint, not estimates.
|
||||
|
||||
### Architecture summary
|
||||
|
||||
Qwen 3.6 35B-A3B is a Qwen3-Next hybrid MoE:
|
||||
Qwen 3.6 35B-A3B is a Qwen3-Next hybrid MoE (`model_type: qwen3_5_moe`, `architectures: Qwen3_5MoeForConditionalGeneration`):
|
||||
|
||||
- 40 transformer layers
|
||||
- Pattern: `10 × (3× Gated DeltaNet → MoE → 1× Gated Attention → MoE)`
|
||||
- **10 growing-attention layers** (gated attention with KV cache)
|
||||
- **30 GDN layers** (recurrent state, fixed size)
|
||||
- MoE: 128 experts, 8 active per token (typical Qwen3-Next MoE config — verify from `config.json`)
|
||||
- **40 transformer layers**
|
||||
- `full_attention_interval: 4` → every 4th layer is full attention; the other 3 are Gated DeltaNet
|
||||
- `layer_types` array confirms **10 full_attention layers at indices [3, 7, 11, 15, 19, 23, 27, 31, 35, 39]** + **30 linear_attention (GDN) layers**
|
||||
- **2 KV heads** (`num_key_value_heads: 2`) — caps `valid_tp` at `[1, 2]`
|
||||
- **16 attention heads**, **head_dim: 256**
|
||||
- **MoE: 256 experts, 8 active per token** (was estimated as 128 — real config has 2× more experts)
|
||||
- `moe_intermediate_size: 512`, `shared_expert_intermediate_size: 512`
|
||||
- Built-in MTP drafter (`mtp_num_hidden_layers: 1`) — same pattern as Qwen 3.6 27B
|
||||
- `attn_output_gate: True` — gated attention
|
||||
- Vision-capable (`vision_config` + image/video token IDs present)
|
||||
- Active params: ~3B; total params: 35B
|
||||
|
||||
### 1. Model weights
|
||||
|
||||
MoE weights are dominated by the expert FFNs. Per-card budget under TP:
|
||||
MoE weights are dominated by the expert FFNs. **5 quant variants on disk** as of 2026-05-15:
|
||||
|
||||
| Quant (planned) | On-disk estimate | Per-card at TP=2 |
|
||||
|---|---:|---:|
|
||||
| AutoRound INT4 (when available) | ~22-25 GB | 11-12 GB |
|
||||
| AWQ-4bit (community) | ~22-25 GB | 11-12 GB |
|
||||
| BF16 (unquantized) | ~70 GB | 35 GB (does not fit on 24 GB) |
|
||||
| Quant | On-disk | Per-card at TP=2 | Notes |
|
||||
|---|---:|---:|---|
|
||||
| AutoRound INT4 (`qwen3.6-35b-a3b-autoround-int4`) | 20 GB | 10 GB | Production; matches our Qwen 3.6 27B AutoRound pipeline |
|
||||
| GPTQ INT4 (`qwen3.6-35b-a3b-gptq-int4`) | 22 GB | 11 GB | Experimental |
|
||||
| GGUF (`qwen3.6-35b-a3b-gguf`) | 90 GB | n/a (llama.cpp single-card path) | Multi-bit-depth |
|
||||
| DFlash variants (`*-dflash`, `*-dflash-gguf`) | variable | n/a | Experimental (z-lab) |
|
||||
| BF16 unquantized | ~70 GB | 35 GB | Does not fit on 24 GB |
|
||||
|
||||
Like the dense Qwen 3.6 27B, DeltaNet `linear_attn` in-projection layers will likely stay at fp16 even under INT4 quantization. The byte count will be included in the total checkpoint size.
|
||||
Like the dense Qwen 3.6 27B, DeltaNet `linear_attn` in-projection layers stay at fp16 even under INT4 quantization. The byte count is included in the total checkpoint size.
|
||||
|
||||
**Note**: MoE expert weights all live in VRAM (they're sparse-activated at FLOPs level, not at memory level). Don't confuse "active params" with "loaded params" — the budget is for the full 35B.
|
||||
|
||||
@@ -306,7 +378,7 @@ Like the dense Qwen 3.6 27B, DeltaNet `linear_attn` in-projection layers will li
|
||||
Applying the general formula:
|
||||
|
||||
```
|
||||
per_token_bytes = 10 (growing layers) × 2 (kv_heads) × 256 (head_dim) × 2 (no K=V tie) × bpe
|
||||
per_token_bytes = 10 (growing layers) × 2 (kv_heads) × 256 (head_dim) × k_v_tensors=2 × bpe
|
||||
= 10,240 × bpe bytes
|
||||
```
|
||||
|
||||
@@ -338,11 +410,22 @@ The activation peak should be **~60-70% of dense Qwen 3.6 27B's** (30/48 layers
|
||||
|
||||
MoE introduces a few new accounting items:
|
||||
|
||||
- **Router workspace**: small (`hidden_size × num_experts` weights, ~100-200 MB). One-time cost, not per-token.
|
||||
- **Expert dispatch buffers**: vLLM allocates buffers for top-k expert routing. Empirical ~200-400 MB per card.
|
||||
- **Router workspace**: `hidden_size × num_experts × bf16_bytes = 2048 × 256 × 2 = ~1 MB` per router. Across 40 layers ≈ 40 MB. Tiny one-time cost.
|
||||
- **Expert dispatch buffers**: vLLM allocates buffers for top-k expert routing across all 256 experts. Empirical ~200-400 MB per card.
|
||||
- **No KV-side impact**: MoE only gates FFN compute. The KV cache for the gated-attention layers is unaffected.
|
||||
|
||||
### 5. Cudagraph + workspace overhead
|
||||
### 5. DeltaNet recurrent state (per-stream, constant)
|
||||
|
||||
The 30 GDN layers maintain a fixed-size recurrent state between tokens (separate from the block-wise intermediate during forward, which is the activation peak). Concrete size:
|
||||
|
||||
- K state: `linear_num_k_heads × linear_k_head_dim × fp32 = 16 × 128 × 4 = 8 KB` per layer
|
||||
- V state: `linear_num_v_heads × linear_v_head_dim × fp32 = 32 × 128 × 4 = 16 KB` per layer
|
||||
- Conv state: `linear_conv_kernel_dim × (16×128 + 32×128) × fp32 = ~96 KB` per layer
|
||||
- **Total per layer: ~120 KB** × 30 layers × `max_num_seqs` streams
|
||||
|
||||
At `max_num_seqs=1`: ~3.5 MB total per card. At `max_num_seqs=4`: ~14 MB. **Negligible** vs activation peak (which is GB-scale) and KV pool (sub-GB). Listed here for completeness; don't bother modelling in budget projections.
|
||||
|
||||
### 6. Cudagraph + workspace overhead
|
||||
|
||||
Same form as dense models:
|
||||
|
||||
@@ -394,7 +477,7 @@ Two shipped quants on this stack: AutoRound INT4 (default) and AWQ-4bit (Tier 2
|
||||
Each stores K and V at `global_head_dim=512`, with K==V tying meaning a single store per element:
|
||||
|
||||
```
|
||||
per_token_bytes_growing = 10 (growing layers) × 16 (kv_heads) × 512 (global_head_dim) × 1 (K=V tied) × bpe
|
||||
per_token_bytes_growing = 10 (growing layers) × 16 (kv_heads) × 512 (global_head_dim) × k_v_tensors=1 × bpe
|
||||
= 81,920 × bpe bytes
|
||||
```
|
||||
|
||||
@@ -419,7 +502,7 @@ Total growing-KV pool per card = `per_token_bytes_growing / TP × max_ctx × max
|
||||
The 50 sliding-attention layers maintain a fixed-size KV window (`sliding_window=1024`). K==V tying applies here too:
|
||||
|
||||
```
|
||||
sliding_kv_bytes_total = 50 (sliding layers) × 16 (kv_heads) × 256 (head_dim) × 1 (K=V tied) × bpe × 1024 (window)
|
||||
sliding_kv_bytes_total = 50 (sliding layers) × 16 (kv_heads) × 256 (head_dim) × k_v_tensors=1 × bpe × 1024 (window)
|
||||
= 209,715,200 × bpe bytes
|
||||
≈ 200 MB × bpe
|
||||
```
|
||||
@@ -463,84 +546,118 @@ At TP > 1, drafter weights shard across cards (`drafter_gb / TP`).
|
||||
|
||||
### Architecture summary
|
||||
|
||||
Gemma 4 26B-A4B is a Gemma 4 MoE with sliding-window attention:
|
||||
Gemma 4 26B-A4B is a Gemma 4 MoE (`model_type: gemma4`, `architectures: Gemma4ForConditionalGeneration`):
|
||||
|
||||
- Likely ~40-48 transformer layers (smaller than Gemma 4 31B's 60)
|
||||
- Sliding-window + occasional global layers (Gemma 4 family pattern); final layer typically global
|
||||
- Sliding window 1024 (same as 31B)
|
||||
- K=V tying expected (Gemma 4 family convention)
|
||||
- MoE: number of experts + active-per-token TBD from `config.json`
|
||||
- **30 transformer layers** (notably smaller than Gemma 4 31B's 60)
|
||||
- `layer_types` array confirms **5 full_attention layers at indices [5, 11, 17, 23, 29]** + **25 sliding_attention layers**
|
||||
- Pattern: every 6th layer is global; **last layer is always global** (per Gemma 4 family convention)
|
||||
- `sliding_window: 1024`
|
||||
- **`attention_k_eq_v: True`** — K and V share storage (×1)
|
||||
- **Asymmetric KV head counts** (the big architectural surprise vs Gemma 4 31B):
|
||||
- `num_key_value_heads: 8` — for sliding-attention layers
|
||||
- `num_global_key_value_heads: 2` — for full-attention layers
|
||||
- `head_dim: 256` (sliding), `global_head_dim: 512` (global)
|
||||
- **MoE: 128 experts, 8 active per token** (`top_k_experts: 8` in config)
|
||||
- `moe_intermediate_size: 704`
|
||||
- Multimodal: `vision_config` + `audio_config` token IDs + image/video token IDs present
|
||||
- **Does NOT require Genesis** (Gemma 4 family has no DeltaNet quirks)
|
||||
- Active params: ~4B; total params: 26B
|
||||
|
||||
**Exact layer counts pending the model's actual config.json on disk.** The placeholders below show the math shape — substitute real values once measured.
|
||||
|
||||
### 1. Model weights
|
||||
|
||||
| Quant (planned) | On-disk estimate | Per-card at TP=2 |
|
||||
|---|---:|---:|
|
||||
| AutoRound INT4 (when available) | ~16-18 GB | 8-9 GB |
|
||||
| BF16 (unquantized) | ~52 GB | 26 GB (does not fit on 24 GB) |
|
||||
| Quant | On-disk | Per-card at TP=2 | Notes |
|
||||
|---|---:|---:|---|
|
||||
| **Intel AutoRound INT4 mixed** (`gemma-4-26b-a4b-autoround-int4-mixed`) | ~14-15 GB | 7-8 GB | Production target. Mixed precision protects routing-critical layers; matches our AutoRound pipeline. |
|
||||
| Intel AutoRound INT4 (pure) | ~13 GB | 6.5 GB | Alternative; slightly worse routing quality than mixed. |
|
||||
| Community AWQ-4bit (cyankiwi) | ~13-14 GB | 6.5-7 GB | Different quant pipeline → activation coefficients don't transfer from our AutoRound calibration. |
|
||||
| BF16 (unquantized) | ~52 GB | 26 GB | Does not fit on 24 GB. |
|
||||
|
||||
MoE expert weights all live in VRAM (sparse-activation at FLOPs, not at memory). Active-params count (4B) doesn't reduce the loaded budget.
|
||||
MoE expert weights all live in VRAM (sparse-activation at FLOPs level, not at memory). Active-params count (4B) doesn't reduce the loaded budget.
|
||||
|
||||
### 2. KV pool — growing portion (global layers only)
|
||||
### 2. KV pool — growing portion (5 full_attention layers)
|
||||
|
||||
Estimated `N_global` global layers (TBD from README; Gemma 4 family typically uses 1 global per 5 sliding):
|
||||
The asymmetric KV head count dramatically reduces per-token growing KV vs Gemma 4 31B:
|
||||
|
||||
```
|
||||
per_token_bytes_growing = N_global × num_kv_heads × global_head_dim × 1 (K=V tied) × bpe
|
||||
per_token_bytes_growing = num_full_attn_layers × num_global_kv_heads × global_head_dim × k_v_tensors=1 × bpe
|
||||
= 5 × 2 × 512 × 1 × bpe
|
||||
= 5,120 × bpe bytes
|
||||
```
|
||||
|
||||
If `N_global = 8` and shape matches 31B family (`num_kv_heads=16`, `global_head_dim=512`):
|
||||
**Compare to Gemma 4 31B's growing KV** = `10 × 16 × 512 × 1 × bpe = 81,920 × bpe bytes` per token. The 26B-A4B is **~16× lighter per token**:
|
||||
|
||||
- Fewer full-attention layers: 5 vs 10
|
||||
- Fewer KV heads on global layers: 2 vs 16
|
||||
- Same head_dim and K=V tying
|
||||
|
||||
| KV format | bpe | per-token growing KV (TP=1) | per-token (TP=2) |
|
||||
|---|---:|---:|---:|
|
||||
| `bf16` / `fp16` | 2.0 | 10,240 B (~10 KB) | 5,120 B |
|
||||
| `fp8_e5m2` / `fp8_e4m3` | 1.0 | 5,120 B (~5 KB) | 2,560 B |
|
||||
| `int8_per_token_head` | ~1.01 | ~5,170 B | ~2,585 B |
|
||||
| `q4_0` | ~0.56 | ~2,867 B | ~1,434 B |
|
||||
|
||||
**Implication**: at 200K context, growing KV pool per card at TP=2 + fp8 = `2,560 × 200,000 = ~512 MB`. **The 26B-A4B is extremely KV-light** — even at full 262K context, growing KV per card is under 700 MB at fp8. The constraint shifts decisively to weights + activation peak, NOT to KV.
|
||||
|
||||
This means BF16 KV becomes viable at 262K on Ampere consumer cards (~1.3 GB growing KV per card) — a contrast to Gemma 4 31B where INT8 PTH was the unlock for long context.
|
||||
|
||||
### 3. KV pool — fixed sliding portion (25 sliding_attention layers)
|
||||
|
||||
The 25 SWA layers maintain a fixed-size KV window (`sliding_window: 1024`):
|
||||
|
||||
```
|
||||
per_token_bytes_growing = 8 × 16 × 512 × 1 × bpe = 65,536 × bpe bytes
|
||||
sliding_kv_bytes_total = num_sliding_layers × num_kv_heads × head_dim × k_v_tensors=1 × bpe × sliding_window
|
||||
= 25 × 8 × 256 × 1 × bpe × 1024
|
||||
= 52,428,800 × bpe bytes
|
||||
≈ 50 MB × bpe
|
||||
```
|
||||
|
||||
That's ~80% of Gemma 4 31B's growing KV per token. INT8 KV remains the right format for 24 GB Ampere at long contexts.
|
||||
**Constant** — doesn't scale with `max_ctx` or `max_num_seqs`. At fp8 KV: ~50 MB per card (TP=1) or ~25 MB at TP=2. Negligible.
|
||||
|
||||
### 3. KV pool — fixed sliding portion
|
||||
Note: this is dramatically smaller than Gemma 4 31B's sliding portion (`50 × 16 × 256 × 1 × bpe × 1024 ≈ 200 MB × bpe`) due to fewer sliding layers (25 vs 50) and fewer KV heads (8 vs 16).
|
||||
|
||||
```
|
||||
sliding_kv_bytes_total = N_sliding × num_kv_heads × head_dim × 1 (K=V tied) × bpe × 1024
|
||||
```
|
||||
### 4. Activation peak (SWA prefill + dense MoE intermediate buffer)
|
||||
|
||||
If `N_sliding = 32` and shape matches 31B (`head_dim=256`):
|
||||
Same mechanism as Gemma 4 31B (SWA prefill + dense MoE intermediate buffer). MoE adds small per-expert routing overhead but **shouldn't dominate**.
|
||||
|
||||
```
|
||||
sliding_kv_bytes_total = 32 × 16 × 256 × 1 × bpe × 1024 = 134,217,728 × bpe bytes ≈ 128 MB × bpe
|
||||
```
|
||||
Projected coefficient (calibration pending; expect ≥4 BENCHMARKS rows before locking in):
|
||||
|
||||
Constant; doesn't scale with `max_ctx`.
|
||||
| KV format | Projected bytes/layer/token | Reasoning |
|
||||
|---|---:|---|
|
||||
| `bf16` / `fp16` | ~1.0-1.5 KB | Smaller than Gemma 4 31B due to fewer total layers (30 vs 60) and smaller `hidden_size` (2816 vs 5376) |
|
||||
| `fp8_e5m2` / `int8_per_token_head` | ~1.0-1.5 KB | Similar to BF16; minimal dequant overhead |
|
||||
|
||||
### 4. Activation peak
|
||||
Expected activation peak: ~1-2 GB at TP=2 dual-card configs, but **calibration TBD**.
|
||||
|
||||
Same mechanism as Gemma 4 31B (SWA prefill + dense MoE intermediate buffer). MoE may add a small per-expert routing overhead but **shouldn't dominate**. Empirical coefficient TBD; expected ~1-2 GB at TP=2.
|
||||
### 5. MoE-specific considerations
|
||||
|
||||
### 5. MoE considerations
|
||||
Same accounting as Qwen 3.6 35B-A3B:
|
||||
|
||||
Same as Qwen 3.6 35B-A3B MoE:
|
||||
|
||||
- Router workspace (~100-200 MB, one-time)
|
||||
- Expert dispatch buffers (~200-400 MB per card)
|
||||
- No KV-side impact from MoE
|
||||
- **Router workspace**: `hidden_size × num_experts` = `2816 × 128` ≈ 360 K weights. Tiny (~700 KB at BF16). One-time cost.
|
||||
- **Expert dispatch buffers**: vLLM allocates buffers for top-k expert routing. Empirical ~200-400 MB per card.
|
||||
- **No KV-side impact**: MoE only gates FFN compute. KV cache for full-attention layers is unaffected.
|
||||
|
||||
### 6. Cudagraph + workspace overhead + drafter
|
||||
|
||||
Same empirical form as 31B; drafter family TBD (Google may release a Gemma 4 26B MTP assistant similar to the 31B-it-assistant).
|
||||
Same empirical form as Gemma 4 31B; standard `0.5 + 1.0 × mem_util + 0.3 × (TP - 1) GB`.
|
||||
|
||||
**Drafter family**:
|
||||
- `google/gemma-4-26B-A4B-it-assistant` released as MTP drafter (~0.5-1 GB, FP16). Same pattern as our existing `gemma-4-31b-it-assistant` drafter.
|
||||
- `z-lab/gemma-4-26B-A4B-it-DFlash` released as DFlash drafter (community).
|
||||
|
||||
### Estimated per-card budget at TP=2, 24 GB VRAM
|
||||
|
||||
| Term | Value (INT8 KV, 100K ctx, seqs=1) | Notes |
|
||||
| Term | Value (fp8 KV, 200K ctx, seqs=1) | Notes |
|
||||
|---|---:|---|
|
||||
| Weights / 2 | ~8-9 GB | INT4 quant |
|
||||
| KV pool growing | ~3-4 GB | 100K × 32 KB/tok = ~3.2 GB |
|
||||
| KV pool sliding | ~0.13 GB | Constant |
|
||||
| Activation peak | ~1-2 GB | Smaller than 31B if fewer total layers |
|
||||
| Cudagraph + overhead | ~1.2 GB | |
|
||||
| **Predicted peak** | **~13-17 GB** | Comfortable headroom on 24 GB, fits 20 GB at lower ctx |
|
||||
| Weights / 2 | ~7-8 GB | AutoRound INT4 mixed (~14-15 GB on-disk) |
|
||||
| KV pool growing | ~0.5 GB | Asymmetric KV heads + few global layers |
|
||||
| KV pool sliding | ~0.05 GB | Constant; trivially small |
|
||||
| Activation peak | ~1-2 GB | Smaller than Gemma 4 31B |
|
||||
| Cudagraph + overhead | ~1.2 GB | Empirical fit |
|
||||
| MoE expert dispatch buffers | ~0.3 GB | Per-card |
|
||||
| **Predicted peak** | **~10-12 GB** | Massive headroom on 24 GB; could likely run at higher mem_util or push to BF16 KV at full 262K |
|
||||
|
||||
**Calibration pending**. Numbers will shift once real `config.json` values replace estimates.
|
||||
**Calibration pending**. The headline finding to verify on first boot: Gemma 4 26B-A4B at full 262K context should fit on a single 3090 with INT4 weights — single-card serving may be the right default for this model.
|
||||
|
||||
## Best practices for building a KV calculator
|
||||
|
||||
|
||||
@@ -42,6 +42,57 @@ isn't your topology.
|
||||
|
||||
---
|
||||
|
||||
## Topology classification
|
||||
|
||||
The launcher classifies your selected hardware and emits strategy guidance
|
||||
when the cards are not matched. You can run the classifier without booting
|
||||
anything:
|
||||
|
||||
```bash
|
||||
bash scripts/launch.sh --topology
|
||||
```
|
||||
|
||||
Use `--gpus 0,1` or `--cards 2` with `--topology` if you only want advice
|
||||
for a subset.
|
||||
|
||||
| Class | What it means | Example | Recommended |
|
||||
|---|---|---|---|
|
||||
| `single_card` | 1 GPU detected | 1x RTX 3090 | Use the largest single-card compose that fits (`vllm/default`, `vllm/long-text`, `llamacpp/default`). |
|
||||
| `homogeneous` | All cards have matched VRAM and matched SM | 2x RTX 3090 | TP=N is the optimal default; use the shipped `vllm/dual*` or `vllm/dual4*` composes. |
|
||||
| `vram_matched_compute_mismatched` | Same VRAM, different compute tier | RTX 3090 + RTX 4090 | TP=N works correctly, but faster cards wait at NCCL allreduce. Estate planner is better for multi-model workloads. |
|
||||
| `vram_mismatched` | Different VRAM sizes | RTX 3060 12 GB + RTX 3090 24 GB | Prefer llama.cpp `--tensor-split`, manual PP=N experiments, or estate planner. Avoid TP=N across the full mismatched set. |
|
||||
| `heterogeneous_mixed` | Multiple VRAM and compute tiers | RTX 3060 + RTX 3090 + RTX 4090 | Manual selection. Run one model on the largest matched subset or use estate planner for separate endpoints. |
|
||||
|
||||
### Why TP=N is poor on VRAM-mismatched cards
|
||||
|
||||
Tensor parallelism splits weights evenly across cards. If one card has 24 GB
|
||||
and another has 12 GB, TP=2 still puts roughly half the model on each card.
|
||||
The smaller card becomes the hard ceiling for weights, KV cache, activations,
|
||||
and fragmentation. For Qwen 3.6 27B INT4, that usually leaves too little KV
|
||||
headroom to be useful.
|
||||
|
||||
For mismatched VRAM, the practical paths are:
|
||||
|
||||
- llama.cpp `--tensor-split` for weighted layer placement.
|
||||
- PP=N as a manual vLLM flag flip (`--pipeline-parallel-size N`) when you are
|
||||
deliberately experimenting. club-3090 does not ship a PP compose today.
|
||||
- Estate planner: `bash scripts/launch.sh --estate` runs different models on
|
||||
different card subsets without forcing one model across uneven VRAM.
|
||||
|
||||
### When compute-mismatched TP is fine
|
||||
|
||||
Matched VRAM with different SM, such as RTX 3090 + RTX 4090, is a different
|
||||
trade-off. TP=2 works because both cards have enough memory for the same model
|
||||
shard and KV budget. The cost is throughput: the faster card waits at NCCL
|
||||
allreduce barriers, so effective pair speed caps near the slower card. You
|
||||
preserve per-card VRAM capacity, but waste some compute on the faster card.
|
||||
|
||||
That is acceptable for one-model serving. If your goal is maximum aggregate
|
||||
throughput from two different cards, estate planner usually wins because each
|
||||
card runs its own model at full speed.
|
||||
|
||||
---
|
||||
|
||||
## Valid TP values for Qwen3.6-27B
|
||||
|
||||
vLLM's tensor parallelism splits attention heads across cards. The TP
|
||||
|
||||
+6
-4
@@ -57,14 +57,15 @@ image: ${VLLM_IMAGE:-vllm/vllm-openai:nightly-${VLLM_NIGHTLY_SHA}}
|
||||
`scripts/launch.sh`, `scripts/switch.sh`, and estate boot resolve
|
||||
`VLLM_NIGHTLY_SHA` from `scripts/lib/profiles/engines/<engine-id>.yml →
|
||||
install.spec`. `VLLM_IMAGE` is a full-image override for users who want to opt
|
||||
into the pre-built GHCR image after the CI smoke has moved `latest` forward.
|
||||
into the pre-built GHCR image. The `:club-vX.Y.Z` release tags are the
|
||||
recommended pinned target; `:latest` follows the most-recent dated nightly.
|
||||
|
||||
| Pin source | Composes using it | Reason for pin | Retirement candidate? |
|
||||
|---|---|---|---|
|
||||
| `scripts/lib/profiles/engines/vllm-nightly-mtp.yml` → `vllm/vllm-openai:nightly-1acd67a7...` | MTP vLLM composes | Post-#41745 nightly for Qwen and Gemma MTP paths. | Bump this YAML when a newer upstream nightly absorbs the required local fixes. |
|
||||
| `scripts/lib/profiles/engines/vllm-nightly-dflash.yml` → `vllm/vllm-openai:nightly-e47c98ef...` | DFlash vLLM composes | DFlash overlay baseline. | Bump this YAML after DFlash overlay drift is revalidated. |
|
||||
| `scripts/lib/profiles/engines/vllm-nightly-full.yml` → `vllm/vllm-openai:nightly-e47c98ef...` | full-overlay vLLM composes | INT8 PTH + DFlash coexistence overlay baseline. | Bump this YAML when #42102/#41703/#35936/#40361 land and the overlay surface shrinks. |
|
||||
| `VLLM_IMAGE=ghcr.io/noonghunna/vllm-club3090:latest` | opt-in override for any vLLM compose | v0.7.0 pre-built image path. CI vendors the overlay set and only moves `latest` / `nightly-stable` after GPU smoke passes. | Optional. It is not the default boot path. |
|
||||
| `scripts/lib/profiles/engines/vllm-nightly-full.yml` → `vllm/vllm-openai:nightly-e47c98ef...` | full-overlay vLLM composes | INT8 PTH + DFlash coexistence overlay baseline. **#42102 closed-as-slop 2026-05-15 → DFlash+quant-KV overlay is now permanent.** | Bump this YAML when #41703 / #35936 / #40361 land and the overlay surface shrinks. The #42102/#41559 coexistence patch stays vendored indefinitely. |
|
||||
| `VLLM_IMAGE=ghcr.io/noonghunna/vllm-club3090:club-vX.Y.Z` (or `:latest`) | opt-in override for any vLLM compose | Pre-built image with vendored overlays baked in. Release tags (`:club-v0.7.0`, etc.) are immutable; `:latest` follows the most-recent dated nightly. See `docs/CI_RUNNER_SETUP.md`. | Optional. It is not the default boot path. |
|
||||
| `ghcr.io/ggml-org/llama.cpp:server-cuda` | 2 (Qwen 3.6-27B llama-cpp) | Stable tag, no hash drift on upstream side. No patches mounted. | Not a retirement candidate — drift-free. Capture digest if reproducibility matters. |
|
||||
|
||||
**Retirement workflow:** see [`NIGHTLY_BUMP_RUNBOOK.md`](./NIGHTLY_BUMP_RUNBOOK.md).
|
||||
@@ -82,6 +83,7 @@ into the pre-built GHCR image after the CI smoke has moved `latest` forward.
|
||||
| Issue / PR | Status | Why it matters | Workaround |
|
||||
|---|---|---|---|
|
||||
| [#35936](https://github.com/vllm-project/vllm/pull/35936) — `tool_choice="required"` falls back to configured tool parser | 🟡 Open / **local overlay active** | Qwen3-Coder with `--tool-call-parser qwen3_coder` emits XML-style tool calls. On pinned nightly `1acd67a79`, non-streaming `tool_choice="required"` validates JSON only, bypasses the configured parser, and returns `tool_calls=[]`. MLS-Bench hits this when `thinking.enabled=false`. | Vendored overlay: [`models/qwen3.6-27b/vllm/patches/vllm-pr35936-required-fallback/README.md`](../models/qwen3.6-27b/vllm/patches/vllm-pr35936-required-fallback/README.md). Drop when #35936 or equivalent lands in our pinned image. |
|
||||
| **[#41800](https://github.com/vllm-project/vllm/pull/41800)** — `truncate_prompt_tokens` kwarg on `get_max_tokens()` | ✅ **Merged upstream 2026-05-06 at `d5b31c95`** / **local overlay active on pre-fix engine pins** | opencode (and other agentic clients sending `truncate_prompt_tokens`) fail with HTTP 400 `get_max_tokens() got an unexpected keyword argument` on engines pinned to `01d4d1ad` (Genesis MTP), `e47c98ef` (DFlash, full). All three SHAs predate `d5b31c95`. `vllm-nightly-clean` (`bf610c2f`, post-fix) doesn't need the overlay. Tracking issue: club-3090 #139. Triggered by club-3090 #138 (SEVENID's opencode failure). | Vendored overlay: [`models/qwen3.6-27b/vllm/patches/vllm-pr41800-truncate-prompt-tokens/README.md`](../models/qwen3.6-27b/vllm/patches/vllm-pr41800-truncate-prompt-tokens/README.md). Wired into 18 affected composes (every compose routing through `vllm-nightly-(mtp\|dflash\|full)`). Install script has upstream-fix detection: no-ops cleanly when run against a post-`d5b31c95` nightly. **Drop trigger per engine**: bump each affected engine's pin past `d5b31c95`. For `vllm-nightly-mtp` that requires Genesis v7.73.x; for `vllm-nightly-dflash` and `vllm-nightly-full` it requires re-validating PR #41703 and PR #42102 overlays on a newer base. |
|
||||
| [#40361](https://github.com/vllm-project/vllm/pull/40361) — Marlin pad-sub-tile-n | 🟡 Open, mergeable, **stale 13d** (last update 2026-04-20) | All 4 dual-card composes + `dual-nvlink.yml` + `dual-nvlink-turbo.yml` mount the patched files vendored in-repo at `models/qwen3.6-27b/vllm/patches/vllm-marlin-pad/`. Drops out as a setup dependency when this merges + propagates. | Vendored mount: see [`models/qwen3.6-27b/vllm/patches/vllm-marlin-pad/README.md`](../models/qwen3.6-27b/vllm/patches/vllm-marlin-pad/README.md). Queued for rebase + ping next week (see "Active follow-ups" table above). |
|
||||
| [#40807](https://github.com/vllm-project/vllm/issues/40807) — `.tolist()` cudagraph crash on continuation-prefill | ✅ **Retired locally** (2026-05-05 Genesis v7.72.2 bump) — Genesis ships [P78 `TOLIST_CAPTURE_GUARD`](../models/qwen3.6-27b/vllm/compose/dual/tq3-mtp-genesis.yml) as the equivalent fix. Currently disabled (`=0`) on `tq3-mtp-genesis.yml` after rebench-full leg 6 (2026-05-11) passed clean with it off — apparent root cause is now covered by Genesis PN34 (workspace-lock relax) + post-#41434 attention rework. Non-Genesis composes on `1acd67a79` pin run without any guard for this bug; unvalidated at long-context TurboQuant chunked-prefill (worth testing per cferra's vllm#41403 validation pass — see vllm#40798 row below). | None active. Drop the Genesis env var permanently if a future v7.73.x rebench leaves it OFF without regression. |
|
||||
| [#40798](https://github.com/vllm-project/vllm/pull/40798) + [#42215](https://github.com/vllm-project/vllm/pull/42215) — share decode scratch workspace pre-CUDA-graph + decode-kernel warmup | 🟡 Open, validated cross-rig | Pair closes the `AssertionError: Workspace is locked but allocation requires NMB` crash that fires at `turboquant_attn.py:_continuation_prefill` for ≥48K-token chunked-prefill with TurboQuant KV. Independently validated on 2× 3090 sm_86 by cferra (vllm#41403 [comment](https://github.com/vllm-project/vllm/issues/41403#issuecomment-4435164709), 2026-05-12). | Genesis [PN34 `WORKSPACE_LOCK_RELAX`](../models/qwen3.6-27b/vllm/compose/dual/tq3-mtp-genesis.yml) addresses the same symptom via a different mechanism (relax-lock vs reserve-before-capture). On non-Genesis composes (`1acd67a79` pin) we currently have no guard — re-validate against this PR pair once they propagate to a nightly we pin to, then A/B PN34 vs upstream. |
|
||||
@@ -89,7 +91,7 @@ into the pre-built GHCR image after the CI smoke has moved `latest` forward.
|
||||
| [#40914](https://github.com/vllm-project/vllm/pull/40914) — Sandermage K+1 verify routing | 🟡 Open, ❌ negative on our Qwen3.6-27B stack | **Reframed 2026-05-11:** the synthetic `seq_lens` K+1 route is not the P67-equivalent we need here. Local rebase on post-#41434 nightly made MTP acceptance look perfect (AL=4.0 / ~100%) but produced `!`-flood needle corruption plus tool/multi-turn timeouts. Dropping it improved verify-stress from 3/7 to 5/7, but TQ3/TQ4/k8v4 + MTP still fail long-context needles. | Do not ship Genesis-free TQ+MTP on #40914 alone. Use `dual/tq3-nomtp.yml` without Genesis, or `dual/tq3-mtp-genesis.yml` with Genesis P67/P67b. |
|
||||
| [#40334](https://github.com/vllm-project/vllm/pull/40334) — DFlash `combine_hidden_states` dtype mismatch | 🟡 Open | All `dual-dflash*.yml` need `--dtype bfloat16` flag to work around. | Composes set `--dtype bfloat16`. Drop when this lands. |
|
||||
| [#40382](https://github.com/vllm-project/vllm/issues/40382) — Gemma-4 + DFlash unservable on Ampere | 🟠 Open, no fix in progress | Blocks DFlash on Gemma-4 family. Not directly our problem (we serve Qwen3.6) but tracked because future model adds may hit it. | None — different attention backend selection. |
|
||||
| **[#41559](https://github.com/vllm-project/vllm/issues/41559) — DFlash spec-decode incompatible with all KV cache quantization** (seantechco, filed 2026-05-03) | 🟢 **OUR FIX PR OPEN: [#42102](https://github.com/vllm-project/vllm/pull/42102)** (filed 2026-05-08) | **REFRAMED 2026-05-08 PM via Codex investigation**: original allowlist-gating framing was partially outdated on current main. Current state: FLASH_ATTN gates dynamically via `flash_attn_supports_fp8()` (FA3-only); FLEX_ATTENTION raises `NotImplementedError` on quantized KV at impl construction; TRITON_ATTN remains causal-only via `assert causal` at `triton_unified_attention.py:542`. The KV-quant write path itself (`triton_reshape_and_cache_flash_per_token_head_quant`) IS causal-mask-independent — but no current backend actually executes both quantized KV AND non-causal attention. **Sharper framing for the common case (BF16 DFlash drafter alongside quantized target KV)**: don't need any backend to "support quantized KV in non-causal mode" — just need the engine to stop forcing target+drafter to share a single page-size unify pass. Three-layer local fix at `/opt/ai/engines/vllm/primary` branch `dflash-noncausal-kv-quant` (commit `cfb8f711`, 4 files, +333/-35): (1) `vllm/v1/core/kv_cache_utils.py` partition DFlash drafter specs into independent KV groups before unify, allocator extended to size isolated tensors by their own page_size; (2) `vllm/model_executor/models/qwen3_dflash.py` override drafter cache_dtype to "auto" when engine global is quantized; (3) `vllm/v1/attention/backends/flash_attn.py` FA metadata scheduler uses per-spec dtype when spec's kv_quant_mode is NONE. **Validated end-to-end on dual 3090 Ampere**: Gemma 4 + z-lab DFlash drafter + INT8 PTH KV target boots HEALTHY at 65K, Paris smoke clean, narrative 95.89 / code 168.09 TPS (matches bf16 32K baseline within CV — long-context unlocked at zero perf cost), AL 5.0-5.3 long-ctx code preserved, NIAH PASS at 32K prompt, KV pool 149,345 tokens (4× lift over baseline). | Local commit `cfb8f711` ready for review + push to `noonghunna/vllm` fork + upstream PR submission. PR description draft at `/tmp/dflash-int8-pr-description.md` (covers non-duplication checks, AI-assistance disclosure, validation matrix). Forensic Phase 3a/3b stacks remain at `models/gemma-4-31b/vllm/patches/vllm-gemma4-dflash-int8/` as historical record (the wrong-fix path that helped diagnose). Container artifacts cleaned up; Qwen production restored. |
|
||||
| **[#41559](https://github.com/vllm-project/vllm/issues/41559) — DFlash spec-decode incompatible with all KV cache quantization** (seantechco, filed 2026-05-03) | ❌ **OUR FIX PR #42102 CLOSED AS SLOP** by @benchislett on 2026-05-15 (no comment, just `closed-as-slop` label). Issue #41559 still OPEN upstream. | Local fix preserved: three-layer patch (4 files, +333/-35) on branch `dflash-noncausal-kv-quant` (commits `cfb8f711` + `5cb61c60`). (1) `vllm/v1/core/kv_cache_utils.py` partitions DFlash drafter specs into independent KV groups before unify, allocator extended to size isolated tensors by their own page_size; (2) `vllm/model_executor/models/qwen3_dflash.py` overrides drafter cache_dtype to "auto" when engine global is quantized; (3) `vllm/v1/attention/backends/flash_attn.py` FA metadata scheduler uses per-spec dtype when spec's kv_quant_mode is NONE. **Validated locally on dual 3090 Ampere**: Gemma 4 + z-lab DFlash drafter + INT8 PTH KV target boots HEALTHY at 65K, narrative 95.89 / code 168.09 TPS, AL 5.0-5.3 preserved, NIAH PASS at 32K, KV pool 149,345 tokens (4× lift). | **Vendor permanently.** Patch lives at `models/gemma-4-31b/vllm/patches/vllm-gemma4-dflash-int8/` and is baked into `vllm-nightly-full` + `vllm-nightly-dflash` EngineProfiles. Re-engagement with upstream NOT recommended (vLLM has hardened anti-AI-PR policy). Watch issue #41559 for any newer maintainer-blessed PR; drop our overlay then. |
|
||||
| [#40354](https://github.com/vllm-project/vllm/issues/40354) — Marlin TP=2 W4A16 < 64 | ✅ Same root-cause as #40361 | Our PR #40361 resolves this. | See #40361 row. |
|
||||
| [#39931](https://github.com/vllm-project/vllm/issues/39931) — DeltaNet rollback support | 🔴 Open, architectural | Blocks **all** spec-decode (EAGLE / DFlash) on Qwen3-Next family across engines. The reason "speculative decoding doesn't work" on this stack. | Use MTP (no rollback needed) until this lands. |
|
||||
| [#40124](https://github.com/vllm-project/vllm/issues/40124) — related architectural | 🔴 Open | Pairs with #39931 for DeltaNet rollback. | Same as above. |
|
||||
|
||||
@@ -0,0 +1,110 @@
|
||||
# ===========================================================================
|
||||
# Profile (at-a-glance):
|
||||
# Model: Gemma 4 26B-A4B MoE (cyankiwi AWQ-4bit, compressed-tensors)
|
||||
# Topology: Dual 3090 (TP=2)
|
||||
# Drafter: MTP n=4 — google/gemma-4-26B-A4B-it-assistant (external)
|
||||
# KV: bfloat16 (sidesteps Ampere fp8 dispatch issues)
|
||||
# Vision: off (limit-mm-per-prompt image=0 audio=0)
|
||||
# Max ctx: 32K (matches awq.yml for direct A/B)
|
||||
# Genesis: none (Gemma 4 family doesn't need Genesis)
|
||||
# Status: 🔵 v0.7.3 PRIMARY — AWQ + MTP, ships alongside awq.yml
|
||||
# ---------------------------------------------------------------------------
|
||||
# Same shape as awq.yml plus the external MTP assistant drafter wired via
|
||||
# `--speculative-config`. n=4 per the gemma-26b-it-assistant DrafterProfile.
|
||||
#
|
||||
# Direct A/B target: awq.yml gives 138.88 / 138.67 wall TPS (2026-05-15, no
|
||||
# spec-decode). This compose measures MTP's contribution. Per the Qwen
|
||||
# 35B-A3B preview-MTP finding (slower despite high acceptance), it's worth
|
||||
# being skeptical — measure both before declaring a default.
|
||||
#
|
||||
# Why a separate drafter path (not built-in MTP head):
|
||||
# Gemma 4 26B-A4B doesn't ship a built-in MTP head — `mtp_num_hidden_layers`
|
||||
# is null in the model YAML. Google released an external assistant model
|
||||
# (`google/gemma-4-26B-A4B-it-assistant`, ~0.97 GB BF16) for this purpose.
|
||||
# Same family pattern as the 31B assistant we already use for `gemma-mtp`.
|
||||
#
|
||||
# PR #40886 overlay still applied (compressed-tensors AWQ MoE key remap).
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 2
|
||||
# Tensor-parallel: 2
|
||||
# Requires-sm: 7.5+
|
||||
services:
|
||||
vllm-gemma-4-26b-a4b-awq-mtp-tp2:
|
||||
image: ${VLLM_IMAGE:-vllm/vllm-openai:nightly-${VLLM_NIGHTLY_SHA}}
|
||||
container_name: "${ESTATE_CONTAINER:-vllm-gemma-4-26b-a4b-awq-mtp-tp2}"
|
||||
restart: "no"
|
||||
ports:
|
||||
- "${ESTATE_PORT:-${PORT:-8043}}:8000"
|
||||
volumes:
|
||||
- ${MODEL_DIR:-../../../../../models-cache}:/root/.cache/huggingface
|
||||
- ../../cache/torch_compile:/root/.cache/vllm/torch_compile_cache
|
||||
- ../../cache/triton:/root/.triton/cache
|
||||
# vLLM PR #40886 overlay — AWQ compressed-tensors MoE key remapping.
|
||||
# See ../../patches/vllm-pr40886-awq-moe-keys/README.md.
|
||||
- ../../patches/vllm-pr40886-awq-moe-keys/install.sh:/etc/club3090/install-pr40886.sh:ro
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=${ESTATE_GPUS:-${NVIDIA_VISIBLE_DEVICES:-all}}
|
||||
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-}
|
||||
- VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||
- NCCL_CUMEM_ENABLE=0
|
||||
- NCCL_P2P_DISABLE=1
|
||||
- VLLM_NO_USAGE_STATS=1
|
||||
- OMP_NUM_THREADS=1
|
||||
- PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True,max_split_size_mb:512
|
||||
- VLLM_ALLOW_LONG_MAX_MODEL_LEN=1
|
||||
- TRITON_CACHE_DIR=/root/.triton/cache
|
||||
shm_size: "16gb"
|
||||
ipc: host
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
entrypoint:
|
||||
- /bin/bash
|
||||
- -c
|
||||
- |
|
||||
bash /etc/club3090/install-pr40886.sh
|
||||
exec vllm serve ${VLLM_ENFORCE_EAGER:+--enforce-eager} "$@"
|
||||
- --
|
||||
command:
|
||||
- --host
|
||||
- 0.0.0.0
|
||||
- --port
|
||||
- "8000"
|
||||
- --model
|
||||
- /root/.cache/huggingface/gemma-4-26b-a4b-awq-4bit
|
||||
- --served-model-name
|
||||
- gemma-4-26b-a4b-awq
|
||||
- --tensor-parallel-size
|
||||
- "${TP:-2}"
|
||||
- --pipeline-parallel-size
|
||||
- "${PP:-1}"
|
||||
- --max-model-len
|
||||
- "${MAX_MODEL_LEN:-32768}"
|
||||
- --gpu-memory-utilization
|
||||
- "${GPU_MEMORY_UTILIZATION:-0.92}"
|
||||
- --max-num-seqs
|
||||
- "256"
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --limit-mm-per-prompt
|
||||
- '{"image":0,"audio":0}'
|
||||
- --kv-cache-dtype
|
||||
- auto
|
||||
- --trust-remote-code
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- gemma4
|
||||
- --chat-template
|
||||
- /vllm-workspace/examples/tool_chat_template_gemma4.jinja
|
||||
- --enable-prefix-caching
|
||||
- --enable-chunked-prefill
|
||||
# External MTP assistant drafter — n=4 per gemma-26b-it-assistant DrafterProfile.
|
||||
- --speculative-config
|
||||
- '{"model":"/root/.cache/huggingface/gemma-4-26b-a4b-it-assistant","num_speculative_tokens":4}'
|
||||
@@ -0,0 +1,117 @@
|
||||
# ===========================================================================
|
||||
# Profile (at-a-glance):
|
||||
# Model: Gemma 4 26B-A4B MoE (cyankiwi AWQ-4bit, compressed-tensors)
|
||||
# Topology: Dual 3090 (TP=2)
|
||||
# Drafter: none (base smoke compose — MTP via gemma-26b-it-assistant
|
||||
# can be added in a follow-up compose)
|
||||
# KV: bfloat16 (sidesteps Ampere fp8 dispatch issues — Triton
|
||||
# fp8e4nv not supported on sm_86)
|
||||
# Vision: off (limit-mm-per-prompt image=0 audio=0) for first boot
|
||||
# Max ctx: 32K (first-boot smoke; can extend after validation)
|
||||
# Genesis: none (Gemma 4 family doesn't need Genesis)
|
||||
# Status: 🔵 v0.7.3 PRIMARY — AWQ path with PR #40886 overlay
|
||||
# Active params: ~4B (128 experts × 8 active = ~4B routed)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Gemma 4 26B-A4B-it (cyankiwi AWQ-4bit, compressed-tensors format) — the
|
||||
# v0.7.3 production Ampere path for this model, replacing the Intel
|
||||
# AutoRound INT4 attempt which is structurally blocked on SM86 by Marlin
|
||||
# K-dim alignment (moe_intermediate_size=704 not aligned to group_size=128).
|
||||
#
|
||||
# Why this works on Ampere where AutoRound INT4 doesn't:
|
||||
# - AWQ via compressed-tensors routes through a different vLLM kernel
|
||||
# path that handles arbitrary K shapes (unlike Marlin)
|
||||
# - PR #40886 overlay (mounted below) fixes the KeyError on the
|
||||
# `_packed` / `_scale` suffixed MoE expert weights — without this,
|
||||
# vLLM's `gemma4.py::_weight_iterator` doesn't know how to remap
|
||||
# the compressed-tensors per-expert keys into the FusedMoE format
|
||||
#
|
||||
# Overlay drops when: vLLM PR #40886 merges upstream AND vllm-nightly-clean
|
||||
# bumps past the merge commit. Track in docs/UPSTREAM.md.
|
||||
#
|
||||
# Vendor: cyankiwi/gemma-4-26B-A4B-it-AWQ-4bit (~17 GB on disk, 4 shards).
|
||||
# Format: compressed-tensors pack-quantized, group-quantized scales.
|
||||
# PR #40886 head: tajwali/vllm @ 652819dad0bf9bbb0436d6660822e7aff30c3ff0.
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 2
|
||||
# Tensor-parallel: 2
|
||||
# Requires-sm: 7.5+
|
||||
services:
|
||||
vllm-gemma-4-26b-a4b-awq-tp2:
|
||||
image: ${VLLM_IMAGE:-vllm/vllm-openai:nightly-${VLLM_NIGHTLY_SHA}}
|
||||
container_name: "${ESTATE_CONTAINER:-vllm-gemma-4-26b-a4b-awq-tp2}"
|
||||
restart: "no"
|
||||
ports:
|
||||
- "${ESTATE_PORT:-${PORT:-8042}}:8000"
|
||||
volumes:
|
||||
- ${MODEL_DIR:-../../../../../models-cache}:/root/.cache/huggingface
|
||||
- ../../cache/torch_compile:/root/.cache/vllm/torch_compile_cache
|
||||
- ../../cache/triton:/root/.triton/cache
|
||||
# vLLM PR #40886 overlay — AWQ compressed-tensors MoE key remapping.
|
||||
# See ../../patches/vllm-pr40886-awq-moe-keys/README.md.
|
||||
- ../../patches/vllm-pr40886-awq-moe-keys/install.sh:/etc/club3090/install-pr40886.sh:ro
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=${ESTATE_GPUS:-${NVIDIA_VISIBLE_DEVICES:-all}}
|
||||
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-}
|
||||
- VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||
- NCCL_CUMEM_ENABLE=0
|
||||
- NCCL_P2P_DISABLE=1
|
||||
- VLLM_NO_USAGE_STATS=1
|
||||
- OMP_NUM_THREADS=1
|
||||
- PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True,max_split_size_mb:512
|
||||
- VLLM_ALLOW_LONG_MAX_MODEL_LEN=1
|
||||
- TRITON_CACHE_DIR=/root/.triton/cache
|
||||
shm_size: "16gb"
|
||||
ipc: host
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
entrypoint:
|
||||
- /bin/bash
|
||||
- -c
|
||||
- |
|
||||
# PR #40886 overlay must run BEFORE `vllm serve` imports the model
|
||||
# module — installs the AWQ compressed-tensors key remapping into
|
||||
# gemma4.py via an anchor-based Python patcher (idempotent).
|
||||
bash /etc/club3090/install-pr40886.sh
|
||||
exec vllm serve ${VLLM_ENFORCE_EAGER:+--enforce-eager} "$@"
|
||||
- --
|
||||
command:
|
||||
- --host
|
||||
- 0.0.0.0
|
||||
- --port
|
||||
- "8000"
|
||||
- --model
|
||||
- /root/.cache/huggingface/gemma-4-26b-a4b-awq-4bit
|
||||
- --served-model-name
|
||||
- gemma-4-26b-a4b-awq
|
||||
- --tensor-parallel-size
|
||||
- "${TP:-2}"
|
||||
- --pipeline-parallel-size
|
||||
- "${PP:-1}"
|
||||
- --max-model-len
|
||||
- "${MAX_MODEL_LEN:-32768}"
|
||||
- --gpu-memory-utilization
|
||||
- "${GPU_MEMORY_UTILIZATION:-0.92}"
|
||||
- --max-num-seqs
|
||||
- "256"
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --limit-mm-per-prompt
|
||||
- '{"image":0,"audio":0}'
|
||||
- --kv-cache-dtype
|
||||
- auto
|
||||
- --trust-remote-code
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- gemma4
|
||||
- --chat-template
|
||||
- /vllm-workspace/examples/tool_chat_template_gemma4.jinja
|
||||
- --enable-prefix-caching
|
||||
- --enable-chunked-prefill
|
||||
@@ -0,0 +1,110 @@
|
||||
# ===========================================================================
|
||||
# Profile (at-a-glance):
|
||||
# Model: Gemma 4 26B-A4B MoE (Intel AutoRound INT4 mixed)
|
||||
# Topology: Dual 3090 (TP=2)
|
||||
# Drafter: none (base smoke compose — MTP via gemma-26b-it-assistant
|
||||
# can be added once base boot is validated)
|
||||
# KV: bfloat16 (sidesteps Ampere fp8 dispatch issues — same logic
|
||||
# as gemma-4-31b/dual composes)
|
||||
# Vision: off (limit-mm-per-prompt image=0 audio=0) for first boot
|
||||
# Max ctx: 32K (first-boot smoke; can extend after validation)
|
||||
# Genesis: none (Gemma 4 family doesn't need Genesis — engine
|
||||
# vllm-nightly-clean rides latest nightly)
|
||||
# Status: 🔵 v0.7.3 ONBOARDING — primary bench target
|
||||
# Active params: ~4B (128 experts × 8 active = ~4B routed)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Gemma 4 26B-A4B-it (Intel AutoRound INT4 mixed) — dual-card production
|
||||
# bench target for v0.7.3 MoE onboarding alongside Qwen 3.6 35B-A3B.
|
||||
#
|
||||
# Why TP=2: weights ~16 GB → ~8 GB/card, leaving ~14 GB/card for KV pool
|
||||
# and activations. Mirrors the production posture used across the matrix.
|
||||
# Single-card variant exists at single/docker-compose.yml (24 GB-only).
|
||||
#
|
||||
# Why vllm-nightly-clean (not vllm-nightly-mtp):
|
||||
# Gemma 4 family doesn't need Genesis patches (no DeltaNet quirks).
|
||||
# By routing through the unconstrained nightly we get latest upstream
|
||||
# features (incl. MoE-loader improvements) without waiting on Genesis
|
||||
# re-anchor cycles. Genesis-anchored MTP path is still vllm-nightly-mtp,
|
||||
# pinned to nightly-01d4d1ad (Sander's v7.72.2 PROD pin).
|
||||
#
|
||||
# Why no drafter on this compose:
|
||||
# The 26B-A4B has an external MTP assistant (google/gemma-4-26B-A4B-it-
|
||||
# assistant, ~0.97 GB) — separate weights from the 31B assistant.
|
||||
# Validating base boot first, then layering MTP on top in a follow-up
|
||||
# compose.
|
||||
#
|
||||
# Models:
|
||||
# target: Intel/gemma-4-26B-A4B-it-int4-mixed-AutoRound
|
||||
# (~16 GB on disk — quant mix of MoE expert layers)
|
||||
# draft : none on this compose (gemma-26b-it-assistant compose TBD)
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 2
|
||||
# Tensor-parallel: 2
|
||||
# Requires-sm: 7.5+
|
||||
services:
|
||||
vllm-gemma-4-26b-a4b-tp2:
|
||||
image: ${VLLM_IMAGE:-vllm/vllm-openai:nightly-${VLLM_NIGHTLY_SHA}}
|
||||
container_name: "${ESTATE_CONTAINER:-vllm-gemma-4-26b-a4b-tp2}"
|
||||
restart: "no"
|
||||
ports:
|
||||
- "${ESTATE_PORT:-${PORT:-8041}}:8000"
|
||||
volumes:
|
||||
- ${MODEL_DIR:-../../../../../models-cache}:/root/.cache/huggingface
|
||||
- ../../cache/torch_compile:/root/.cache/vllm/torch_compile_cache
|
||||
- ../../cache/triton:/root/.triton/cache
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=${ESTATE_GPUS:-${NVIDIA_VISIBLE_DEVICES:-all}}
|
||||
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-}
|
||||
- VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||
- NCCL_CUMEM_ENABLE=0
|
||||
- NCCL_P2P_DISABLE=1
|
||||
- VLLM_NO_USAGE_STATS=1
|
||||
- OMP_NUM_THREADS=1
|
||||
- PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True,max_split_size_mb:512
|
||||
- VLLM_ALLOW_LONG_MAX_MODEL_LEN=1
|
||||
- TRITON_CACHE_DIR=/root/.triton/cache
|
||||
shm_size: "16gb"
|
||||
ipc: host
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
command:
|
||||
- --host
|
||||
- 0.0.0.0
|
||||
- --port
|
||||
- "8000"
|
||||
- --model
|
||||
- /root/.cache/huggingface/gemma-4-26b-a4b-autoround-int4-mixed
|
||||
- --served-model-name
|
||||
- gemma-4-26b-a4b-autoround
|
||||
- --tensor-parallel-size
|
||||
- "${TP:-2}"
|
||||
- --pipeline-parallel-size
|
||||
- "${PP:-1}"
|
||||
- --max-model-len
|
||||
- "${MAX_MODEL_LEN:-32768}"
|
||||
- --gpu-memory-utilization
|
||||
- "${GPU_MEMORY_UTILIZATION:-0.92}"
|
||||
- --max-num-seqs
|
||||
- "256"
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --limit-mm-per-prompt
|
||||
- '{"image":0,"audio":0}'
|
||||
- --kv-cache-dtype
|
||||
- auto
|
||||
- --trust-remote-code
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- gemma4
|
||||
- --chat-template
|
||||
- /vllm-workspace/examples/tool_chat_template_gemma4.jinja
|
||||
- --enable-prefix-caching
|
||||
- --enable-chunked-prefill
|
||||
@@ -0,0 +1,108 @@
|
||||
# ===========================================================================
|
||||
# Profile (at-a-glance):
|
||||
# Model: Gemma 4 26B-A4B MoE (Intel AutoRound INT4 mixed)
|
||||
# Topology: Single 3090 (TP=1)
|
||||
# Drafter: none (base smoke compose — MTP via gemma-26b-it-assistant
|
||||
# can be added once base boot is validated)
|
||||
# KV: bfloat16 (sidesteps Ampere fp8 dispatch issues — same logic
|
||||
# as gemma-4-31b/single/docker-compose.yml)
|
||||
# Vision: off (limit-mm-per-prompt image=0 audio=0) for first boot
|
||||
# Max ctx: 8K (conservative — first-boot smoke; bump after validation)
|
||||
# Genesis: none (Gemma 4 family doesn't need Genesis — engine
|
||||
# vllm-nightly-clean rides latest nightly)
|
||||
# Status: 🔵 v0.7.3 ONBOARDING — first boot pending
|
||||
# Active params: ~4B (128 experts × 8 active = ~4B routed)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Gemma 4 26B-A4B-it (Intel AutoRound INT4 mixed) — first MoE model added to
|
||||
# club-3090 alongside Qwen 3.6 35B-A3B.
|
||||
#
|
||||
# Why vllm-nightly-clean (not vllm-nightly-mtp):
|
||||
# Gemma 4 family doesn't need Genesis patches (no DeltaNet quirks).
|
||||
# By routing through the unconstrained nightly we get latest upstream
|
||||
# features (incl. continued MoE-loader improvements) without waiting on
|
||||
# Genesis re-anchor cycles.
|
||||
#
|
||||
# Why no drafter on this compose:
|
||||
# The 26B-A4B has an external MTP assistant (google/gemma-4-26B-A4B-it-
|
||||
# assistant, ~0.97 GB) — separate weights from the 31B assistant.
|
||||
# Validating base boot first, then layering MTP on top in a follow-up.
|
||||
#
|
||||
# Models:
|
||||
# target: Intel/gemma-4-26B-A4B-it-int4-mixed-AutoRound
|
||||
# (~14 GB — quant mix of MoE expert layers)
|
||||
# draft : none on this compose (gemma-26b-it-assistant compose TBD)
|
||||
#
|
||||
# KV format pinned to bf16:
|
||||
# - fp8_e5m2 → blocked by gemma4_mm.py assert (vLLM allowlist)
|
||||
# - fp8_e4m3 → Triton "fp8e4nv not supported" on sm_86 Ampere
|
||||
# - default bf16 → sidesteps both — same pattern as 31B single
|
||||
# Smaller KV pool than fp8 but at 8K initial ctx not the bottleneck.
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 1
|
||||
# Tensor-parallel: 1
|
||||
# Requires-sm: 7.5+
|
||||
services:
|
||||
vllm-gemma-4-26b-a4b-tp1:
|
||||
image: ${VLLM_IMAGE:-vllm/vllm-openai:nightly-${VLLM_NIGHTLY_SHA}}
|
||||
container_name: "${ESTATE_CONTAINER:-vllm-gemma-4-26b-a4b-tp1}"
|
||||
restart: "no"
|
||||
ports:
|
||||
- "${ESTATE_PORT:-${PORT:-8040}}:8000"
|
||||
volumes:
|
||||
- ${MODEL_DIR:-../../../../../models-cache}:/root/.cache/huggingface
|
||||
- ../../cache/torch_compile:/root/.cache/vllm/torch_compile_cache
|
||||
- ../../cache/triton:/root/.triton/cache
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=${ESTATE_GPUS:-${NVIDIA_VISIBLE_DEVICES:-all}}
|
||||
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-}
|
||||
- VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||
- NCCL_CUMEM_ENABLE=0
|
||||
- NCCL_P2P_DISABLE=1
|
||||
- VLLM_NO_USAGE_STATS=1
|
||||
- OMP_NUM_THREADS=1
|
||||
- PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True,max_split_size_mb:512
|
||||
- VLLM_ALLOW_LONG_MAX_MODEL_LEN=1
|
||||
- TRITON_CACHE_DIR=/root/.triton/cache
|
||||
shm_size: "16gb"
|
||||
ipc: host
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
command:
|
||||
- --host
|
||||
- 0.0.0.0
|
||||
- --port
|
||||
- "8000"
|
||||
- --model
|
||||
- /root/.cache/huggingface/gemma-4-26b-a4b-autoround-int4-mixed
|
||||
- --served-model-name
|
||||
- gemma-4-26b-a4b-autoround
|
||||
- --tensor-parallel-size
|
||||
- "${TP:-1}"
|
||||
- --pipeline-parallel-size
|
||||
- "${PP:-1}"
|
||||
- --max-model-len
|
||||
- "${MAX_MODEL_LEN:-8192}"
|
||||
- --gpu-memory-utilization
|
||||
- "${GPU_MEMORY_UTILIZATION:-0.92}"
|
||||
- --max-num-seqs
|
||||
- "256"
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --limit-mm-per-prompt
|
||||
- '{"image":0,"audio":0}'
|
||||
- --kv-cache-dtype
|
||||
- auto
|
||||
- --trust-remote-code
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- gemma4
|
||||
- --chat-template
|
||||
- /vllm-workspace/examples/tool_chat_template_gemma4.jinja
|
||||
@@ -0,0 +1,79 @@
|
||||
# vLLM PR #40886 overlay — AWQ compressed-tensors MoE key remapping
|
||||
|
||||
## What this fixes
|
||||
|
||||
`cyankiwi/gemma-4-26B-A4B-it-AWQ-4bit` ships in `compressed-tensors` pack-quantized format, which stores per-expert weights with `_packed` (int32) and `_scale` (bfloat16) suffixes. The vLLM `gemma4.py::_weight_iterator` (as of bf610c2f, 2026-05-15) only handles the float-checkpoint key pattern and does not remap the `_packed` / `_scale` suffixed variants on MoE expert weights. Without this overlay, model load fails with a `KeyError` on the first packed expert key.
|
||||
|
||||
The fix is from [vLLM PR #40886](https://github.com/vllm-project/vllm/pull/40886) by @tajwali (open as of 2026-05-15, last updated 2026-04-25). PR author tested on **RTX 3090 24 GB** (same SKU as ours) with vLLM 0.19.1. The patch is +23 / -0 — pure insertion of 4 conditional branches in `_weight_iterator` that intercept the four `_packed` / `_scale` MoE key patterns and yield per-expert float-shaped keys that the existing FusedMoE loader path expects.
|
||||
|
||||
## Why an anchor-based Python patcher (not full-file replacement)
|
||||
|
||||
The PR is a small insertion. Full-file replacement risks shadowing unrelated upstream changes in `gemma4.py` that we DO want (PR #41745 Gemma 4 MTP support, etc.). The anchor-based patcher in `install.sh` locates the existing `if "moe.gate_up_proj" in name and weight.dim() == 3:` line inside `_weight_iterator` and inserts the patch immediately above it. Idempotent: a sentinel comment is included in the inserted block, and re-running on an already-patched file is a no-op.
|
||||
|
||||
## How to use
|
||||
|
||||
### From a compose
|
||||
|
||||
Bind-mount `install.sh` into the container at a known path, then invoke it from the entrypoint **before** `vllm serve` runs:
|
||||
|
||||
```yaml
|
||||
services:
|
||||
vllm-gemma-4-26b-a4b-awq-tp2:
|
||||
image: ${VLLM_IMAGE:-vllm/vllm-openai:nightly-${VLLM_NIGHTLY_SHA}}
|
||||
volumes:
|
||||
- ../../patches/vllm-pr40886-awq-moe-keys/install.sh:/etc/club3090/install-pr40886.sh:ro
|
||||
entrypoint:
|
||||
- /bin/bash
|
||||
- -c
|
||||
- |
|
||||
bash /etc/club3090/install-pr40886.sh
|
||||
exec vllm serve "$@"
|
||||
- --
|
||||
command:
|
||||
- --model
|
||||
- /root/.cache/huggingface/gemma-4-26b-a4b-awq-4bit
|
||||
# ... other flags ...
|
||||
```
|
||||
|
||||
Same sidecar pattern as `vllm-pr35936-required-fallback/install.sh` — runs once per container start, leaves the container's RW layer in the patched state.
|
||||
|
||||
### Override env vars
|
||||
|
||||
If vLLM moves the install path of `gemma4.py`:
|
||||
|
||||
```bash
|
||||
CLUB3090_PR40886_TARGET=/some/other/path/gemma4.py bash install.sh
|
||||
```
|
||||
|
||||
## When to drop this overlay
|
||||
|
||||
When **both** of these are true:
|
||||
|
||||
1. PR #40886 has merged upstream
|
||||
2. The engine's pinned nightly SHA is past the merge commit
|
||||
|
||||
Track in `docs/UPSTREAM.md`.
|
||||
|
||||
## Smoke test
|
||||
|
||||
Manual:
|
||||
|
||||
```bash
|
||||
# Spin up a transient container, install the patch, verify the sentinel.
|
||||
docker run --rm \
|
||||
-v $(pwd)/install.sh:/install.sh:ro \
|
||||
vllm/vllm-openai:nightly-bf610c2f56764e1b30bc6065f4ceace3d6e59036 \
|
||||
bash -c 'bash /install.sh && grep -c "club3090/pr40886" /usr/local/lib/python3.12/dist-packages/vllm/model_executor/models/gemma4.py'
|
||||
```
|
||||
|
||||
Expected: prints `1` (sentinel present after install).
|
||||
|
||||
## Source PR
|
||||
|
||||
- PR head: `tajwali/vllm` @ `652819dad0bf9bbb0436d6660822e7aff30c3ff0` (branch `fix/gemma4-compressed-tensors-moe-key-remapping`)
|
||||
- Vendored as of 2026-05-15
|
||||
- Patch summary: 4 `if` branches added to `_weight_iterator` in `vllm/model_executor/models/gemma4.py`
|
||||
- `moe.gate_up_proj_packed [E, 2I, H/8]` → split into per-expert `gate_proj.weight_packed` + `up_proj.weight_packed`
|
||||
- `moe.gate_up_proj_scale [E, 2I, G]` → same split for scales
|
||||
- `moe.down_proj_packed [E, H, I/8]` → yield per-expert `down_proj.weight_packed`
|
||||
- `moe.down_proj_scale [E, H, G]` → yield per-expert `down_proj.weight_scale`
|
||||
@@ -0,0 +1,104 @@
|
||||
#!/usr/bin/env bash
|
||||
# Install vLLM PR #40886 — compressed-tensors AWQ MoE key remapping for
|
||||
# Gemma 4 26B-A4B. Without this patch, `cyankiwi/gemma-4-26B-A4B-it-AWQ-4bit`
|
||||
# fails to load with KeyError on `moe.gate_up_proj_packed` (or similar)
|
||||
# because vLLM's `gemma4.py::_weight_iterator` doesn't handle the
|
||||
# `_packed`/`_scale` suffix on MoE expert weights.
|
||||
#
|
||||
# PR: https://github.com/vllm-project/vllm/pull/40886
|
||||
# Head commit at time of vendor: 652819dad0bf9bbb0436d6660822e7aff30c3ff0
|
||||
# Author tested on: RTX 3090 24 GB (same SKU as ours), vLLM 0.19.1
|
||||
#
|
||||
# WHY a Python anchor-based patcher instead of full-file replacement:
|
||||
# The PR's diff is +23 / -0 — pure insertion before an existing branch
|
||||
# in `_weight_iterator`. Anchor-based insertion is robust to upstream
|
||||
# drift around the function (vLLM nightly may add unrelated logic
|
||||
# elsewhere in gemma4.py without breaking our patch).
|
||||
#
|
||||
# Idempotent: the patcher checks for a sentinel comment before inserting,
|
||||
# so re-running it on an already-patched file is a no-op.
|
||||
#
|
||||
# Drop when: vLLM PR #40886 merges upstream AND our engine pin bumps
|
||||
# past the merge commit.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# Container's vLLM install path. Override via env if vLLM moves.
|
||||
GEMMA4_PY="${CLUB3090_PR40886_TARGET:-/usr/local/lib/python3.12/dist-packages/vllm/model_executor/models/gemma4.py}"
|
||||
|
||||
if [ ! -f "$GEMMA4_PY" ]; then
|
||||
echo "[club3090/pr40886] ERROR: $GEMMA4_PY not found; aborting overlay install" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
python3 - <<'PY'
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
target = os.environ.get(
|
||||
"CLUB3090_PR40886_TARGET",
|
||||
"/usr/local/lib/python3.12/dist-packages/vllm/model_executor/models/gemma4.py",
|
||||
)
|
||||
|
||||
SENTINEL = "# PATCH: AWQ compressed-tensors key remapping (club3090/pr40886)"
|
||||
|
||||
PATCH_BLOCK = ''' # PATCH: AWQ compressed-tensors key remapping (club3090/pr40886)
|
||||
if "moe.gate_up_proj_packed" in name and weight.dim() == 3:
|
||||
mid = weight.size(1) // 2
|
||||
for e in range(weight.size(0)):
|
||||
base = name.replace("moe.", f"moe.experts.{e}.")
|
||||
yield base.replace("gate_up_proj_packed", "gate_proj.weight_packed"), weight[e, :mid]
|
||||
yield base.replace("gate_up_proj_packed", "up_proj.weight_packed"), weight[e, mid:]
|
||||
continue
|
||||
if "moe.gate_up_proj_scale" in name and weight.dim() == 3:
|
||||
mid = weight.size(1) // 2
|
||||
for e in range(weight.size(0)):
|
||||
base = name.replace("moe.", f"moe.experts.{e}.")
|
||||
yield base.replace("gate_up_proj_scale", "gate_proj.weight_scale"), weight[e, :mid]
|
||||
yield base.replace("gate_up_proj_scale", "up_proj.weight_scale"), weight[e, mid:]
|
||||
continue
|
||||
if "moe.down_proj_packed" in name and weight.dim() == 3:
|
||||
for e in range(weight.size(0)):
|
||||
yield name.replace("moe.", f"moe.experts.{e}.").replace("down_proj_packed", "down_proj.weight_packed"), weight[e]
|
||||
continue
|
||||
if "moe.down_proj_scale" in name and weight.dim() == 3:
|
||||
for e in range(weight.size(0)):
|
||||
yield name.replace("moe.", f"moe.experts.{e}.").replace("down_proj_scale", "down_proj.weight_scale"), weight[e]
|
||||
continue
|
||||
'''
|
||||
|
||||
# Insert immediately above the existing `if "moe.gate_up_proj" in name and weight.dim() == 3:`
|
||||
# branch inside `_weight_iterator`. This is the line the upstream PR puts the patch above.
|
||||
ANCHOR_RE = re.compile(
|
||||
r'^(?P<indent>[ \t]+)(?P<line>if "moe\.gate_up_proj" in name and weight\.dim\(\) == 3:)',
|
||||
re.MULTILINE,
|
||||
)
|
||||
|
||||
with open(target, "r", encoding="utf-8") as f:
|
||||
src = f.read()
|
||||
|
||||
if SENTINEL in src:
|
||||
print(f"[club3090/pr40886] {target}: sentinel present, patch already applied; no-op", file=sys.stderr)
|
||||
sys.exit(0)
|
||||
|
||||
m = ANCHOR_RE.search(src)
|
||||
if not m:
|
||||
print(
|
||||
f"[club3090/pr40886] ERROR: anchor 'if \"moe.gate_up_proj\" in name and weight.dim() == 3:' "
|
||||
f"not found in {target}. vLLM nightly may have changed gemma4.py — overlay needs re-anchoring.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
# Insertion point: start of the matched line
|
||||
insert_at = m.start()
|
||||
patched = src[:insert_at] + PATCH_BLOCK + src[insert_at:]
|
||||
|
||||
with open(target, "w", encoding="utf-8") as f:
|
||||
f.write(patched)
|
||||
|
||||
print(f"[club3090/pr40886] {target}: PR #40886 (AWQ MoE key remapping) applied", file=sys.stderr)
|
||||
PY
|
||||
|
||||
echo "[club3090/pr40886] install complete" >&2
|
||||
@@ -78,6 +78,10 @@ services:
|
||||
- ../../cache/triton_awq:/root/.triton/cache
|
||||
# NVLink auto-detection — runs inside container at boot.
|
||||
- ../../../../../scripts/detect_nvlink.sh:/etc/club3090/detect_nvlink.sh:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=${ESTATE_GPUS:-${NVIDIA_VISIBLE_DEVICES:-all}}
|
||||
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-}
|
||||
@@ -112,6 +116,7 @@ services:
|
||||
pip install --quiet --upgrade transformers==5.8.0
|
||||
# NVLink auto-detection (sets NCCL env vars, _NVLINK_ENABLED).
|
||||
source /etc/club3090/detect_nvlink.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
if [ "${_NVLINK_ENABLED:-0}" = "1" ]; then
|
||||
exec vllm serve "$@"
|
||||
else
|
||||
|
||||
@@ -91,7 +91,7 @@
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-mtp
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 2
|
||||
# Tensor-parallel: 2
|
||||
services:
|
||||
|
||||
@@ -145,6 +145,10 @@ services:
|
||||
# NVLink auto-detection — runs inside container at boot.
|
||||
- ../../../../../scripts/detect_nvlink.sh:/etc/club3090/detect_nvlink.sh:ro
|
||||
# --------------------------------------------------------------------
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=${ESTATE_GPUS:-${NVIDIA_VISIBLE_DEVICES:-all}}
|
||||
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-}
|
||||
@@ -177,6 +181,7 @@ services:
|
||||
pip install --quiet --upgrade transformers==5.8.0
|
||||
# NVLink auto-detection (sets NCCL env vars, _NVLINK_ENABLED).
|
||||
source /etc/club3090/detect_nvlink.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
if [ "${_NVLINK_ENABLED:-0}" = "1" ]; then
|
||||
exec vllm serve "$@"
|
||||
else
|
||||
|
||||
@@ -110,6 +110,10 @@ services:
|
||||
# NVLink auto-detection — runs inside container at boot.
|
||||
- ../../../../../scripts/detect_nvlink.sh:/etc/club3090/detect_nvlink.sh:ro
|
||||
# --------------------------------------------------------------------
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=${ESTATE_GPUS:-${NVIDIA_VISIBLE_DEVICES:-all}}
|
||||
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-}
|
||||
@@ -142,6 +146,7 @@ services:
|
||||
pip install --quiet --upgrade transformers==5.8.0
|
||||
# NVLink auto-detection (sets NCCL env vars, _NVLINK_ENABLED).
|
||||
source /etc/club3090/detect_nvlink.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
if [ "${_NVLINK_ENABLED:-0}" = "1" ]; then
|
||||
exec vllm serve "$@"
|
||||
else
|
||||
|
||||
@@ -50,7 +50,7 @@
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-mtp
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 2
|
||||
# Tensor-parallel: 2
|
||||
services:
|
||||
|
||||
@@ -220,6 +220,10 @@ services:
|
||||
# NVLink auto-detection — runs inside container at boot.
|
||||
- ../../../../../scripts/detect_nvlink.sh:/etc/club3090/detect_nvlink.sh:ro
|
||||
# --------------------------------------------------------------------
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=${ESTATE_GPUS:-${NVIDIA_VISIBLE_DEVICES:-all}}
|
||||
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-}
|
||||
@@ -253,6 +257,7 @@ services:
|
||||
echo "[club3090-tq3] Launching vllm serve..." >&2
|
||||
# NVLink auto-detection (sets NCCL env vars, _NVLINK_ENABLED).
|
||||
source /etc/club3090/detect_nvlink.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
if [ "${_NVLINK_ENABLED:-0}" = "1" ]; then
|
||||
exec vllm serve "$@"
|
||||
else
|
||||
|
||||
@@ -124,6 +124,10 @@ services:
|
||||
# NVLink auto-detection — runs inside container at boot.
|
||||
- ../../../../../scripts/detect_nvlink.sh:/etc/club3090/detect_nvlink.sh:ro
|
||||
# --------------------------------------------------------------------
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=${ESTATE_GPUS:-${NVIDIA_VISIBLE_DEVICES:-all}}
|
||||
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-}
|
||||
@@ -151,6 +155,7 @@ services:
|
||||
- |
|
||||
# NVLink auto-detection (sets NCCL env vars, _NVLINK_ENABLED).
|
||||
source /etc/club3090/detect_nvlink.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
if [ "${_NVLINK_ENABLED:-0}" = "1" ]; then
|
||||
exec vllm serve "$@"
|
||||
else
|
||||
|
||||
@@ -50,7 +50,7 @@
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 32
|
||||
# Engine-profile: vllm-nightly-mtp
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 1
|
||||
# Tensor-parallel: 1
|
||||
# Requires-sm: 9.0+
|
||||
|
||||
@@ -39,6 +39,19 @@ Memory budget: 14.5 GB (Q3_K_XL) + 4.5 GB KV @ 262K + 0.8 GB mmproj ≈ 20 GB /
|
||||
|
||||
Trade max context for parallelism. Same image, `--parallel 4` + smaller ctx pool.
|
||||
|
||||
### Tuning knobs
|
||||
|
||||
Both Docker composes expose llama.cpp's batch-size controls without editing YAML:
|
||||
|
||||
| Env var | llama.cpp flag | Default | Sensible range on 24 GB | Notes |
|
||||
|---|---|---:|---:|---|
|
||||
| `BATCH_SIZE` | `-b` | `4096` | `2048`-`8192` | Logical prompt-processing batch. Higher can improve prefill throughput if VRAM headroom allows. |
|
||||
| `UBATCH_SIZE` | `-ub` | `2048` | `1024`-`4096` | Physical microbatch. Lower this first if long prompts OOM during prefill. |
|
||||
|
||||
These are throughput-tuning knobs inside llama.cpp. They are orthogonal to
|
||||
`ESTATE_GPUS` and `ESTATE_PORT`, which only isolate GPU assignment and host port
|
||||
when `scripts/launch.sh --estate` boots multiple instances.
|
||||
|
||||
---
|
||||
|
||||
## Recipes (host-binary alternative)
|
||||
|
||||
@@ -59,6 +59,8 @@ services:
|
||||
-m /models/${GGUF_FILE:-qwen3.6-27b-gguf/unsloth-q3kxl/Qwen3.6-27B-UD-Q3_K_XL.gguf}
|
||||
--mmproj /models/${MMPROJ_FILE:-qwen3.6-27b-gguf/mmproj-F16.gguf}
|
||||
-c ${CTX_SIZE:-192000}
|
||||
-b ${BATCH_SIZE:-4096}
|
||||
-ub ${UBATCH_SIZE:-2048}
|
||||
-ngl 99
|
||||
-fa on
|
||||
--cache-type-k ${KV_TYPE:-q4_0}
|
||||
|
||||
@@ -56,6 +56,8 @@
|
||||
# GGUF_FILE path under /models (default: qwen3.6-27b-gguf/unsloth-q3kxl/Qwen3.6-27B-UD-Q3_K_XL.gguf)
|
||||
# MMPROJ_FILE path under /models (default: qwen3.6-27b-gguf/mmproj-F16.gguf)
|
||||
# CTX_SIZE total KV pool (default: 262144)
|
||||
# BATCH_SIZE llama.cpp -b (default: 4096)
|
||||
# UBATCH_SIZE llama.cpp -ub (default: 2048)
|
||||
# KV_TYPE K and V quant type (default: q4_0)
|
||||
# REASONING_FORMAT reasoning channel routing (default: none — for opencode/IDE-agent compat)
|
||||
# Override to `auto` to get separate `reasoning_content` field
|
||||
@@ -118,6 +120,10 @@ services:
|
||||
- /models/${MMPROJ_FILE:-qwen3.6-27b-gguf/mmproj-F16.gguf}
|
||||
- -c
|
||||
- ${CTX_SIZE:-262144}
|
||||
- -b
|
||||
- ${BATCH_SIZE:-4096}
|
||||
- -ub
|
||||
- ${UBATCH_SIZE:-2048}
|
||||
- -ngl
|
||||
- "99"
|
||||
- -fa
|
||||
|
||||
@@ -22,7 +22,7 @@
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-mtp
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 2
|
||||
# Tensor-parallel: 2
|
||||
services:
|
||||
|
||||
@@ -50,7 +50,7 @@
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-mtp
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 2
|
||||
# Tensor-parallel: 2
|
||||
services:
|
||||
|
||||
@@ -75,6 +75,9 @@ services:
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/chat_completion/serving.py:/etc/club3090/pr35936-chat-completion-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/engine/serving.py:/etc/club3090/pr35936-engine-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/install.sh:/etc/club3090/install-pr35936.sh:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode
|
||||
# HTTP 400 on get_max_tokens(). See patches/vllm-pr41800.../README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
# froggeric/Qwen-Fixed-Chat-Templates qwen3.6 — fixes 7 default-template
|
||||
# bugs (empty <think></think> spam, </thinking> hallucination, unclosed
|
||||
# think before tool call, no-user-query crash, developer role, etc.).
|
||||
@@ -114,6 +117,7 @@ services:
|
||||
# hardware where graph capture causes OOM or instability (e.g. WSL2).
|
||||
# Install PR #35936 overlay before vllm imports (drop when upstream lands).
|
||||
bash /etc/club3090/install-pr35936.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
# NVLink auto-detection (sets NCCL env vars, _NVLINK_ENABLED).
|
||||
source /etc/club3090/detect_nvlink.sh
|
||||
if [ "${_NVLINK_ENABLED:-0}" = "1" ]; then
|
||||
|
||||
@@ -99,6 +99,9 @@ services:
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/chat_completion/serving.py:/etc/club3090/pr35936-chat-completion-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/engine/serving.py:/etc/club3090/pr35936-engine-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/install.sh:/etc/club3090/install-pr35936.sh:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode
|
||||
# HTTP 400 on get_max_tokens(). See patches/vllm-pr41800.../README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
# froggeric/Qwen-Fixed-Chat-Templates qwen3.6 — fixes 7 default-template
|
||||
# bugs (empty <think></think> spam, </thinking> hallucination, unclosed
|
||||
# think before tool call, no-user-query crash, developer role, etc.).
|
||||
@@ -138,6 +141,7 @@ services:
|
||||
# hardware where graph capture causes OOM or instability (e.g. WSL2).
|
||||
# Install PR #35936 overlay before vllm imports (drop when upstream lands).
|
||||
bash /etc/club3090/install-pr35936.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
# NVLink auto-detection (sets NCCL env vars, _NVLINK_ENABLED).
|
||||
source /etc/club3090/detect_nvlink.sh
|
||||
if [ "${_NVLINK_ENABLED:-0}" = "1" ]; then
|
||||
|
||||
@@ -51,7 +51,7 @@
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-mtp
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 2
|
||||
# Tensor-parallel: 2
|
||||
services:
|
||||
@@ -87,6 +87,9 @@ services:
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/chat_completion/serving.py:/etc/club3090/pr35936-chat-completion-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/engine/serving.py:/etc/club3090/pr35936-engine-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/install.sh:/etc/club3090/install-pr35936.sh:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode
|
||||
# HTTP 400 on get_max_tokens(). See patches/vllm-pr41800.../README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
# froggeric/Qwen-Fixed-Chat-Templates qwen3.6 — fixes 7 default-template
|
||||
# bugs (empty <think></think> spam, </thinking> hallucination, unclosed
|
||||
# think before tool call, no-user-query crash, developer role, etc.).
|
||||
@@ -128,6 +131,7 @@ services:
|
||||
# stability. See docs/HARDWARE.md "Note for WSL2 / Windows users".
|
||||
# Install PR #35936 overlay before vllm imports (drop when upstream lands).
|
||||
bash /etc/club3090/install-pr35936.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
# NVLink auto-detection (sets NCCL env vars, _NVLINK_ENABLED).
|
||||
source /etc/club3090/detect_nvlink.sh
|
||||
if [ "${_NVLINK_ENABLED:-0}" = "1" ]; then
|
||||
|
||||
@@ -57,6 +57,10 @@ services:
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/chat_completion/serving.py:/etc/club3090/pr35936-chat-completion-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/engine/serving.py:/etc/club3090/pr35936-engine-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/install.sh:/etc/club3090/install-pr35936.sh:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
# froggeric/Qwen-Fixed-Chat-Templates qwen3.6 — fixes 7 default-template
|
||||
# bugs (empty <think></think> spam, </thinking> hallucination, unclosed
|
||||
# think before tool call, no-user-query crash, developer role, etc.).
|
||||
@@ -88,6 +92,7 @@ services:
|
||||
- |
|
||||
# Install PR #35936 overlay before vllm imports (drop when upstream lands).
|
||||
bash /etc/club3090/install-pr35936.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
exec vllm serve ${VLLM_ENFORCE_EAGER:+--enforce-eager} "$@"
|
||||
- --
|
||||
command:
|
||||
|
||||
@@ -35,7 +35,7 @@
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-mtp
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 2
|
||||
# Tensor-parallel: 2
|
||||
services:
|
||||
|
||||
@@ -105,6 +105,10 @@ services:
|
||||
# See docs/UPSTREAM.md "Community templates / model assets" + the row at
|
||||
# https://huggingface.co/froggeric/Qwen-Fixed-Chat-Templates
|
||||
- ../../patches/froggeric-chat-template/chat_template.jinja:/etc/qwen-froggeric-chat-template.jinja:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=${ESTATE_GPUS:-${NVIDIA_VISIBLE_DEVICES:-all}}
|
||||
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-}
|
||||
@@ -246,6 +250,7 @@ services:
|
||||
# v7.72.2 (P78 + PN34). Mounts and invocations dropped 2026-05-05.
|
||||
# VLLM_ENFORCE_EAGER=1 in compose/.env disables CUDA graphs — use on
|
||||
# hardware where Cliff 2 GDN activation spikes occur at runtime.
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
exec vllm serve ${VLLM_ENFORCE_EAGER:+--enforce-eager} "$@"
|
||||
- --
|
||||
command:
|
||||
|
||||
@@ -104,6 +104,10 @@ services:
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/chat_completion/serving.py:/etc/club3090/pr35936-chat-completion-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/engine/serving.py:/etc/club3090/pr35936-engine-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/install.sh:/etc/club3090/install-pr35936.sh:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
# froggeric/Qwen-Fixed-Chat-Templates qwen3.6 — fixes 7 default-template
|
||||
# bugs (empty <think></think> spam, </thinking> hallucination, unclosed
|
||||
# think before tool call, no-user-query crash, developer role, etc.).
|
||||
@@ -136,6 +140,7 @@ services:
|
||||
- |
|
||||
# Install PR #35936 overlay before vllm imports (drop when upstream lands).
|
||||
bash /etc/club3090/install-pr35936.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
exec vllm serve ${VLLM_ENFORCE_EAGER:+--enforce-eager} "$@"
|
||||
- --
|
||||
command:
|
||||
|
||||
@@ -88,6 +88,10 @@ services:
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/chat_completion/serving.py:/etc/club3090/pr35936-chat-completion-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/engine/serving.py:/etc/club3090/pr35936-engine-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/install.sh:/etc/club3090/install-pr35936.sh:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
# froggeric/Qwen-Fixed-Chat-Templates qwen3.6 — fixes 7 default-template
|
||||
# bugs (empty <think></think> spam, </thinking> hallucination, unclosed
|
||||
# think before tool call, no-user-query crash, developer role, etc.).
|
||||
@@ -119,6 +123,7 @@ services:
|
||||
- |
|
||||
# Install PR #35936 overlay before vllm imports (drop when upstream lands).
|
||||
bash /etc/club3090/install-pr35936.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
exec vllm serve ${VLLM_ENFORCE_EAGER:+--enforce-eager} "$@"
|
||||
- --
|
||||
command:
|
||||
|
||||
@@ -84,6 +84,9 @@ services:
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/chat_completion/serving.py:/etc/club3090/pr35936-chat-completion-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/engine/serving.py:/etc/club3090/pr35936-engine-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/install.sh:/etc/club3090/install-pr35936.sh:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode
|
||||
# HTTP 400 on get_max_tokens(). See patches/vllm-pr41800.../README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
# froggeric/Qwen-Fixed-Chat-Templates qwen3.6 — fixes 7 default-template
|
||||
# bugs (empty <think></think> spam, </thinking> hallucination, unclosed
|
||||
# think before tool call, no-user-query crash, developer role, etc.).
|
||||
@@ -222,6 +225,7 @@ services:
|
||||
# Install PR #35936 overlay BEFORE Genesis runs so Genesis can write hooks
|
||||
# to chat_completion/serving.py without hitting RO-mount errors.
|
||||
bash /etc/club3090/install-pr35936.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
python3 -m vllm._genesis.patches.apply_all
|
||||
# Tool-parser deferred-commit fix for qwen3coder SSE-silence bug (issue #72).
|
||||
# Drops out when vllm-project/vllm lands the upstream fix.
|
||||
|
||||
@@ -92,6 +92,10 @@ services:
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/chat_completion/serving.py:/etc/club3090/pr35936-chat-completion-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/engine/serving.py:/etc/club3090/pr35936-engine-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/install.sh:/etc/club3090/install-pr35936.sh:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
# froggeric/Qwen-Fixed-Chat-Templates qwen3.6 — fixes 7 default-template
|
||||
# bugs (empty <think></think> spam, </thinking> hallucination, unclosed
|
||||
# think before tool call, no-user-query crash, developer role, etc.).
|
||||
@@ -128,6 +132,7 @@ services:
|
||||
# hardware where graph capture causes OOM or instability (e.g. WSL2).
|
||||
# Install PR #35936 overlay before vllm imports (drop when upstream lands).
|
||||
bash /etc/club3090/install-pr35936.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
exec vllm serve ${VLLM_ENFORCE_EAGER:+--enforce-eager} "$@"
|
||||
- --
|
||||
command:
|
||||
|
||||
@@ -59,7 +59,7 @@
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-mtp
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 4
|
||||
# Tensor-parallel: 4
|
||||
services:
|
||||
|
||||
@@ -161,6 +161,10 @@ services:
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/chat_completion/serving.py:/etc/club3090/pr35936-chat-completion-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/engine/serving.py:/etc/club3090/pr35936-engine-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/install.sh:/etc/club3090/install-pr35936.sh:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
# froggeric/Qwen-Fixed-Chat-Templates qwen3.6 — fixes 7 default-template
|
||||
# bugs (empty <think></think> spam, </thinking> hallucination, unclosed
|
||||
# think before tool call, no-user-query crash, developer role, etc.).
|
||||
@@ -290,6 +294,7 @@ services:
|
||||
# Install PR #35936 overlay BEFORE Genesis runs so Genesis can write hooks
|
||||
# to chat_completion/serving.py without hitting RO-mount errors.
|
||||
bash /etc/club3090/install-pr35936.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
python3 -m vllm._genesis.patches.apply_all
|
||||
# Tool-parser deferred-commit fix for qwen3coder SSE-silence bug (issue #72).
|
||||
# Drops out when vllm-project/vllm lands the upstream fix.
|
||||
|
||||
@@ -132,6 +132,10 @@ services:
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/chat_completion/serving.py:/etc/club3090/pr35936-chat-completion-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/engine/serving.py:/etc/club3090/pr35936-engine-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/install.sh:/etc/club3090/install-pr35936.sh:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
# froggeric/Qwen-Fixed-Chat-Templates qwen3.6 — fixes 7 default-template
|
||||
# bugs (empty <think></think> spam, </thinking> hallucination, unclosed
|
||||
# think before tool call, no-user-query crash, developer role, etc.).
|
||||
@@ -221,6 +225,7 @@ services:
|
||||
# Install PR #35936 overlay BEFORE Genesis runs so Genesis can write hooks
|
||||
# to chat_completion/serving.py without hitting RO-mount errors.
|
||||
bash /etc/club3090/install-pr35936.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
python3 -m vllm._genesis.patches.apply_all
|
||||
# Tool-parser deferred-commit fix for qwen3coder SSE-silence bug (issue #72).
|
||||
# Drops out when vllm-project/vllm lands the upstream fix.
|
||||
|
||||
@@ -149,6 +149,10 @@ services:
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/chat_completion/serving.py:/etc/club3090/pr35936-chat-completion-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/engine/serving.py:/etc/club3090/pr35936-engine-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/install.sh:/etc/club3090/install-pr35936.sh:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
# froggeric/Qwen-Fixed-Chat-Templates qwen3.6 — fixes 7 default-template
|
||||
# bugs (empty <think></think> spam, </thinking> hallucination, unclosed
|
||||
# think before tool call, no-user-query crash, developer role, etc.).
|
||||
@@ -315,6 +319,7 @@ services:
|
||||
# Install PR #35936 overlay BEFORE Genesis runs so Genesis can write hooks
|
||||
# to chat_completion/serving.py without hitting RO-mount errors.
|
||||
bash /etc/club3090/install-pr35936.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
python3 -m vllm._genesis.patches.apply_all
|
||||
# Tool-parser deferred-commit fix for qwen3coder SSE-silence bug (issue #72).
|
||||
# Drops out when vllm-project/vllm lands the upstream fix.
|
||||
|
||||
@@ -159,6 +159,10 @@ services:
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/chat_completion/serving.py:/etc/club3090/pr35936-chat-completion-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/engine/serving.py:/etc/club3090/pr35936-engine-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/install.sh:/etc/club3090/install-pr35936.sh:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
# froggeric/Qwen-Fixed-Chat-Templates qwen3.6 — fixes 7 default-template
|
||||
# bugs (empty <think></think> spam, </thinking> hallucination, unclosed
|
||||
# think before tool call, no-user-query crash, developer role, etc.).
|
||||
@@ -332,6 +336,7 @@ services:
|
||||
# Install PR #35936 overlay BEFORE Genesis runs so Genesis can write hooks
|
||||
# to chat_completion/serving.py without hitting RO-mount errors.
|
||||
bash /etc/club3090/install-pr35936.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
python3 -m vllm._genesis.patches.apply_all
|
||||
# Tool-parser deferred-commit fix for qwen3coder SSE-silence bug (issue #72).
|
||||
# Drops out when vllm-project/vllm lands the upstream fix.
|
||||
|
||||
@@ -127,6 +127,10 @@ services:
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/chat_completion/serving.py:/etc/club3090/pr35936-chat-completion-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/vllm/entrypoints/openai/engine/serving.py:/etc/club3090/pr35936-engine-serving.py:ro
|
||||
- ../../patches/vllm-pr35936-required-fallback/install.sh:/etc/club3090/install-pr35936.sh:ro
|
||||
# vLLM PR #41800 (truncate_prompt_tokens kwarg) overlay — fixes opencode-style
|
||||
# HTTP 400 on get_max_tokens(). No-op on post-fix nightlies.
|
||||
# See ../../patches/vllm-pr41800-truncate-prompt-tokens/README.md.
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
# froggeric/Qwen-Fixed-Chat-Templates qwen3.6 — fixes 7 default-template
|
||||
# bugs (empty <think></think> spam, </thinking> hallucination, unclosed
|
||||
# think before tool call, no-user-query crash, developer role, etc.).
|
||||
@@ -256,6 +260,7 @@ services:
|
||||
# Install PR #35936 overlay BEFORE Genesis runs so Genesis can write hooks
|
||||
# to chat_completion/serving.py without hitting RO-mount errors.
|
||||
bash /etc/club3090/install-pr35936.sh
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
python3 -m vllm._genesis.patches.apply_all
|
||||
# Tool-parser deferred-commit fix for qwen3coder SSE-silence bug (issue #72).
|
||||
# Drops out when vllm-project/vllm lands the upstream fix.
|
||||
|
||||
@@ -33,7 +33,7 @@
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 20
|
||||
# Engine-profile: vllm-nightly-mtp
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 1
|
||||
# Tensor-parallel: 1
|
||||
services:
|
||||
|
||||
@@ -35,7 +35,7 @@
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-mtp
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 1
|
||||
# Tensor-parallel: 1
|
||||
services:
|
||||
|
||||
@@ -0,0 +1,94 @@
|
||||
# vLLM PR #41800 overlay — `truncate_prompt_tokens` kwarg on `get_max_tokens`
|
||||
|
||||
## What this fixes
|
||||
|
||||
Agentic clients (opencode, codex-cli, and similar IDE/agent runtimes) send `truncate_prompt_tokens` on chat-completion requests. Pre-[vLLM PR #41800](https://github.com/vllm-project/vllm/pull/41800), `vllm.entrypoints.utils.get_max_tokens()` doesn't accept that kwarg — and the kwarg propagates from the request handler down into the function call — so requests fail with:
|
||||
|
||||
```
|
||||
HTTP 400: {"error":{"message":"get_max_tokens() got an unexpected keyword argument 'truncate_prompt_tokens'",...}}
|
||||
```
|
||||
|
||||
The fix is upstream PR #41800 (merged 2026-05-06 at commit `d5b31c95`). It adds the kwarg to the function signature and a small body block that clamps `input_length` to `min(input_length, truncate_prompt_tokens or max_model_len)` before the existing length check.
|
||||
|
||||
## When this overlay is needed
|
||||
|
||||
This overlay is needed on engines pinned to vLLM SHAs that **predate `d5b31c95`**:
|
||||
|
||||
| Engine | Pinned SHA | Pre-fix? |
|
||||
|---|---|---|
|
||||
| `vllm-nightly-mtp` | `01d4d1ad` (2026-05-04) | ✅ needs overlay |
|
||||
| `vllm-nightly-dflash` | `e47c98ef` (~2026-05-05) | ✅ needs overlay (20 commits behind d5b31c95) |
|
||||
| `vllm-nightly-full` | `e47c98ef` | ✅ needs overlay |
|
||||
| `vllm-nightly-clean` | `bf610c2f` (2026-05-15) | ❌ already includes fix |
|
||||
|
||||
If a compose routes through `vllm-nightly-clean`, the overlay is unnecessary — the function signature already accepts the kwarg upstream.
|
||||
|
||||
## How the overlay works
|
||||
|
||||
`install.sh` is a Python anchor-based in-place patcher. It does two surgical edits to the in-container `/usr/local/lib/python3.12/dist-packages/vllm/entrypoints/utils.py`:
|
||||
|
||||
1. **Signature**: adds `truncate_prompt_tokens: int | None = None,` to `get_max_tokens`'s signature, anchored to the existing `override_max_tokens: int | None = None,` line.
|
||||
2. **Body**: inserts a 6-line truncation-aware `input_length` adjustment block before the existing `if max_model_len < input_length:` check, anchored to that line.
|
||||
|
||||
Each insertion carries a sentinel comment (`# PATCH: truncate_prompt_tokens kwarg (club3090/pr41800)`) so re-running the install on an already-patched file is a no-op. Post-patch the file is AST-validated before write.
|
||||
|
||||
Why anchor-based and not full-file replacement: the PR diff is +14 / -0 across a 200-line file — replacing the full file would shadow other upstream changes in `utils.py`. Anchor-based insertion is drift-resistant to unrelated upstream movement.
|
||||
|
||||
## Composes that wire this overlay in (as of v0.7.3 ship)
|
||||
|
||||
* `models/qwen3.6-27b/vllm/compose/dual/docker-compose.yml` (gpu-mode `27b`)
|
||||
* `models/qwen3.6-27b/vllm/compose/dual/turbo.yml` (gpu-mode `27b-turbo`)
|
||||
* `models/qwen3.6-27b/vllm/compose/dual/dflash.yml` (gpu-mode `27b-dflash`)
|
||||
* `models/qwen3.6-27b/vllm/compose/dual/dflash-noviz.yml` (gpu-mode `27b-dflash-noviz`, the compose from issue #138)
|
||||
|
||||
## How to add this overlay to another affected compose
|
||||
|
||||
In any compose that routes through `vllm-nightly-mtp` / `vllm-nightly-dflash` / `vllm-nightly-full`, add:
|
||||
|
||||
1. **Volume mount** in the `volumes:` block:
|
||||
|
||||
```yaml
|
||||
- ../../patches/vllm-pr41800-truncate-prompt-tokens/install.sh:/etc/club3090/install-pr41800.sh:ro
|
||||
```
|
||||
|
||||
2. **Install line** in the `entrypoint:` bash script, before `exec vllm serve`:
|
||||
|
||||
```bash
|
||||
bash /etc/club3090/install-pr41800.sh
|
||||
```
|
||||
|
||||
Run `bash install.sh` (the file in this directory) standalone to test against a transient vLLM container before wiring into a compose. See the smoke test in the next section.
|
||||
|
||||
## Smoke test
|
||||
|
||||
```bash
|
||||
docker run --rm --entrypoint /bin/bash \
|
||||
-v $(pwd)/install.sh:/install.sh:ro \
|
||||
vllm/vllm-openai:nightly-01d4d1ad375dc5854779c593eee093bcebb0cada \
|
||||
-c '
|
||||
python3 -c "from vllm.entrypoints.utils import get_max_tokens; import inspect; print(inspect.signature(get_max_tokens))"
|
||||
bash /install.sh
|
||||
python3 -c "from vllm.entrypoints.utils import get_max_tokens; import inspect; print(inspect.signature(get_max_tokens))"
|
||||
'
|
||||
```
|
||||
|
||||
Expected: signature lacks `truncate_prompt_tokens` BEFORE install, has it AFTER. Verified on `01d4d1ad` (2026-05-15).
|
||||
|
||||
## When to drop this overlay
|
||||
|
||||
When **both** are true:
|
||||
|
||||
1. PR #41800 has merged upstream (it has — 2026-05-06 at `d5b31c95`)
|
||||
2. The engine's pinned nightly SHA bumps past `d5b31c95`
|
||||
|
||||
For the Genesis-anchored engines, the bump happens with Sander's next Genesis release cycle (v7.73.x). For `vllm-nightly-dflash` and `vllm-nightly-full`, the bump happens when their respective overlays (PR #41703 DFlash, PR #42102 INT8 PTH KV) are re-validated against a newer nightly.
|
||||
|
||||
Track in `docs/UPSTREAM.md`.
|
||||
|
||||
## Source
|
||||
|
||||
- vLLM PR #41800: https://github.com/vllm-project/vllm/pull/41800
|
||||
- Merged commit: `d5b31c95`
|
||||
- Tracking issue: noonghunna/club-3090#139
|
||||
- Triggered by: noonghunna/club-3090#138 (SEVENID's opencode boot failure)
|
||||
- Patch summary: +7 lines in `vllm/entrypoints/utils.py` (the actual fix) + 5 call-site forward-compat additions in other files (we skip those — the signature fix alone unblocks all known TypeError reports)
|
||||
+141
@@ -0,0 +1,141 @@
|
||||
#!/usr/bin/env bash
|
||||
# Install vLLM PR #41800 — `truncate_prompt_tokens` kwarg on get_max_tokens.
|
||||
#
|
||||
# WHY THIS OVERLAY EXISTS:
|
||||
# opencode (and other agentic clients like codex-cli) send `truncate_prompt_tokens`
|
||||
# on chat-completion requests. Pre-#41800, vLLM's `get_max_tokens()` doesn't
|
||||
# accept that kwarg — and somewhere upstream of the function the kwarg gets
|
||||
# unpacked into the call — so requests fail with:
|
||||
# HTTP 400: get_max_tokens() got an unexpected keyword argument 'truncate_prompt_tokens'
|
||||
#
|
||||
# PR: https://github.com/vllm-project/vllm/pull/41800
|
||||
# Merged: 2026-05-06 at commit d5b31c95
|
||||
# Affected pins on master:
|
||||
# - vllm-nightly-mtp (01d4d1ad, 2026-05-04) — pre-fix
|
||||
# - vllm-nightly-dflash (e47c98ef) — pre-fix
|
||||
# - vllm-nightly-full (e47c98ef) — pre-fix
|
||||
# (vllm-nightly-clean at bf610c2f is POST-fix; doesn't need the overlay)
|
||||
#
|
||||
# Tracking issue: #139 (noonghunna/club-3090)
|
||||
# Triggered by: #138 — SEVENID's opencode boot failure on dual-dflash-noviz.
|
||||
#
|
||||
# WHY A PYTHON ANCHOR-BASED PATCHER:
|
||||
# The PR is +7 lines in `vllm/entrypoints/utils.py` (the actual fix) plus a
|
||||
# handful of forward-compat call-site additions in 5 other files. The
|
||||
# function-signature change in utils.py is the ONLY thing required to fix
|
||||
# the TypeError — once `get_max_tokens` accepts the kwarg, requests stop
|
||||
# crashing. The call-site changes are nice-to-have semantic completeness
|
||||
# (actually applying the truncation), so we patch those too via anchors.
|
||||
#
|
||||
# Idempotent: each anchor checks for a sentinel marker before inserting.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# Container's vLLM install path. Override via env if vLLM moves.
|
||||
SITE_PACKAGES="${CLUB3090_PR41800_SITE_PACKAGES:-/usr/local/lib/python3.12/dist-packages}"
|
||||
|
||||
UTILS_PY="$SITE_PACKAGES/vllm/entrypoints/utils.py"
|
||||
|
||||
if [ ! -f "$UTILS_PY" ]; then
|
||||
echo "[club3090/pr41800] ERROR: $UTILS_PY not found; aborting overlay install" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
python3 - <<'PY'
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
site_packages = os.environ.get(
|
||||
"CLUB3090_PR41800_SITE_PACKAGES",
|
||||
"/usr/local/lib/python3.12/dist-packages",
|
||||
)
|
||||
utils_py = f"{site_packages}/vllm/entrypoints/utils.py"
|
||||
|
||||
SENTINEL = "# PATCH: truncate_prompt_tokens kwarg (club3090/pr41800)"
|
||||
|
||||
# Function signature update: add `truncate_prompt_tokens: int | None = None,`
|
||||
# as a kwarg on `get_max_tokens`. Anchor on the existing line that closes
|
||||
# the signature (`override_max_tokens: int | None = None,` line right before `) -> int:`).
|
||||
SIGNATURE_ANCHOR_RE = re.compile(
|
||||
r'^(?P<indent>[ \t]+)override_max_tokens: int \| None = None,\n(?P<close>[ \t]*\) -> int:)',
|
||||
re.MULTILINE,
|
||||
)
|
||||
SIGNATURE_INSERT = ''' override_max_tokens: int | None = None,
|
||||
truncate_prompt_tokens: int | None = None, # PATCH: truncate_prompt_tokens kwarg (club3090/pr41800)
|
||||
) -> int:'''
|
||||
|
||||
# Body update: insert truncation-aware input_length adjustment BEFORE the
|
||||
# `if max_model_len < input_length:` check. Anchor on that line.
|
||||
BODY_ANCHOR_RE = re.compile(
|
||||
r'^(?P<indent>[ \t]+)if max_model_len < input_length:',
|
||||
re.MULTILINE,
|
||||
)
|
||||
BODY_INSERT = ''' # PATCH: truncate_prompt_tokens kwarg (club3090/pr41800)
|
||||
if truncate_prompt_tokens is not None:
|
||||
limit = truncate_prompt_tokens
|
||||
input_length = min(
|
||||
input_length,
|
||||
max_model_len if limit == -1 else limit,
|
||||
)
|
||||
if max_model_len < input_length:'''
|
||||
|
||||
with open(utils_py, "r", encoding="utf-8") as f:
|
||||
src = f.read()
|
||||
|
||||
if SENTINEL in src:
|
||||
print(f"[club3090/pr41800] {utils_py}: sentinel present, patch already applied; no-op", file=sys.stderr)
|
||||
sys.exit(0)
|
||||
|
||||
# Upstream-fix detection: if the function signature already accepts the kwarg
|
||||
# (i.e. the engine pinned a post-#41800 nightly), the overlay is unnecessary
|
||||
# and should no-op gracefully so composes that mount it on a post-fix image
|
||||
# (e.g. via vllm-nightly-clean) still boot cleanly.
|
||||
UPSTREAM_RE = re.compile(
|
||||
r'def get_max_tokens\([^)]*truncate_prompt_tokens\b',
|
||||
re.DOTALL,
|
||||
)
|
||||
if UPSTREAM_RE.search(src):
|
||||
print(f"[club3090/pr41800] {utils_py}: upstream get_max_tokens() already accepts truncate_prompt_tokens; no-op", file=sys.stderr)
|
||||
sys.exit(0)
|
||||
|
||||
# Apply signature patch first (so the function accepts the kwarg)
|
||||
m = SIGNATURE_ANCHOR_RE.search(src)
|
||||
if not m:
|
||||
print(
|
||||
f"[club3090/pr41800] ERROR: signature anchor "
|
||||
f"'override_max_tokens: int | None = None, ... ) -> int:' not found in {utils_py}. "
|
||||
f"vLLM nightly may have changed entrypoints/utils.py — overlay needs re-anchoring.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
src = SIGNATURE_ANCHOR_RE.sub(SIGNATURE_INSERT, src, count=1)
|
||||
|
||||
# Apply body patch
|
||||
m = BODY_ANCHOR_RE.search(src)
|
||||
if not m:
|
||||
print(
|
||||
f"[club3090/pr41800] ERROR: body anchor 'if max_model_len < input_length:' not found in {utils_py} "
|
||||
f"after signature patch. vLLM nightly diverged unexpectedly — overlay needs re-anchoring.",
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
src = BODY_ANCHOR_RE.sub(BODY_INSERT, src, count=1)
|
||||
|
||||
with open(utils_py, "w", encoding="utf-8") as f:
|
||||
f.write(src)
|
||||
|
||||
# Quick validity check
|
||||
import ast
|
||||
try:
|
||||
ast.parse(src)
|
||||
except SyntaxError as e:
|
||||
print(f"[club3090/pr41800] ERROR: post-patch utils.py is not valid Python: {e}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
print(f"[club3090/pr41800] {utils_py}: signature + body patches applied (truncate_prompt_tokens kwarg)", file=sys.stderr)
|
||||
PY
|
||||
|
||||
echo "[club3090/pr41800] install complete" >&2
|
||||
@@ -0,0 +1,103 @@
|
||||
# ===========================================================================
|
||||
# Profile (at-a-glance):
|
||||
# Model: Qwen 3.6 35B-A3B MoE (AutoRound INT4, ~20 GB weights)
|
||||
# Topology: Dual 3090 (TP=2)
|
||||
# Drafter: MTP n=3 (built-in head — qwen-mtp-builtin DrafterProfile)
|
||||
# KV: fp8_e5m2 (no TQ3 — Genesis-only, not available here)
|
||||
# Vision: off (limit-mm-per-prompt image=0 audio=0)
|
||||
# Max ctx: 16K (conservative — matches preview.yml for direct A/B)
|
||||
# Genesis: none — preview-MTP path on vllm-nightly-clean
|
||||
# Status: 🔵 PREVIEW — measures MTP n=3 uplift on the Qwen MoE
|
||||
# ---------------------------------------------------------------------------
|
||||
# Same shape as preview.yml plus the built-in MTP head wired via
|
||||
# `--speculative-config`. Qwen 3.6 35B-A3B's `mtp_num_hidden_layers: 1`
|
||||
# (per the model config) provides the head natively — no separate drafter
|
||||
# download needed. n=3 mirrors the production Qwen 27B convention.
|
||||
#
|
||||
# Direct A/B target: preview.yml gives 182.68 / 177.45 wall TPS (2026-05-15,
|
||||
# no spec-decode). This compose measures MTP's contribution on the same
|
||||
# rig + config so calibration data has a tight baseline pair.
|
||||
#
|
||||
# Cliff 2 mitigations still UNAVAILABLE without Genesis — keep max_ctx
|
||||
# below ~21K accumulated until Genesis v7.73.x re-anchors on a post-#42521
|
||||
# nightly. Same caveat as preview.yml.
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 2
|
||||
# Tensor-parallel: 2
|
||||
# Requires-sm: 7.5+
|
||||
services:
|
||||
vllm-qwen36-35b-a3b-preview-mtp-tp2:
|
||||
image: ${VLLM_IMAGE:-vllm/vllm-openai:nightly-${VLLM_NIGHTLY_SHA}}
|
||||
container_name: "${ESTATE_CONTAINER:-vllm-qwen36-35b-a3b-preview-mtp-tp2}"
|
||||
restart: "no"
|
||||
ports:
|
||||
- "${BIND_HOST:-0.0.0.0}:${ESTATE_PORT:-${PORT:-8052}}:8000"
|
||||
volumes:
|
||||
- ${MODEL_DIR:-../../../../../models-cache}:/root/.cache/huggingface
|
||||
- ../../cache/torch_compile:/root/.cache/vllm/torch_compile_cache
|
||||
- ../../cache/triton:/root/.triton/cache
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=${ESTATE_GPUS:-${NVIDIA_VISIBLE_DEVICES:-all}}
|
||||
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-}
|
||||
- VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||
- NCCL_CUMEM_ENABLE=0
|
||||
- NCCL_P2P_DISABLE=1
|
||||
- VLLM_NO_USAGE_STATS=1
|
||||
- PYTORCH_CUDA_ALLOC_CONF=${PYTORCH_CUDA_ALLOC_CONF:-expandable_segments:True}
|
||||
- OMP_NUM_THREADS=1
|
||||
shm_size: "16gb"
|
||||
ipc: host
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
command:
|
||||
- --host
|
||||
- 0.0.0.0
|
||||
- --port
|
||||
- "8000"
|
||||
- --model
|
||||
- /root/.cache/huggingface/qwen3.6-35b-a3b-autoround-int4
|
||||
- --served-model-name
|
||||
- qwen3.6-35b-a3b-autoround
|
||||
- --quantization
|
||||
- auto_round
|
||||
- --dtype
|
||||
- float16
|
||||
- --tensor-parallel-size
|
||||
- "${TP:-2}"
|
||||
- --pipeline-parallel-size
|
||||
- "${PP:-1}"
|
||||
- --max-model-len
|
||||
- "${MAX_MODEL_LEN:-16384}"
|
||||
- --gpu-memory-utilization
|
||||
- "${GPU_MEMORY_UTILIZATION:-0.92}"
|
||||
- --max-num-seqs
|
||||
- "1"
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --limit-mm-per-prompt
|
||||
- '{"image":0,"audio":0}'
|
||||
- --kv-cache-dtype
|
||||
- "${KV_CACHE_DTYPE:-fp8_e5m2}"
|
||||
- --trust-remote-code
|
||||
- --reasoning-parser
|
||||
- qwen3
|
||||
- --default-chat-template-kwargs
|
||||
- '{"enable_thinking": false}'
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- qwen3_coder
|
||||
- --enable-prefix-caching
|
||||
- --enable-chunked-prefill
|
||||
# Built-in MTP head — n=3 matches the production Qwen 27B convention.
|
||||
# `model` points at the same model dir; vLLM extracts the MTP head
|
||||
# from `mtp_num_hidden_layers: 1` in the model config.
|
||||
- --speculative-config
|
||||
- '{"method":"mtp","num_speculative_tokens":3}'
|
||||
@@ -0,0 +1,124 @@
|
||||
# ===========================================================================
|
||||
# Profile (at-a-glance):
|
||||
# Model: Qwen 3.6 35B-A3B MoE (AutoRound INT4, ~20 GB weights)
|
||||
# Topology: Dual 3090 (TP=2)
|
||||
# Drafter: none (smoke compose — built-in MTP head added in a follow-up
|
||||
# once base boot is validated)
|
||||
# KV: fp8_e5m2 (no TQ3 — Genesis-only, not available here)
|
||||
# Vision: off (limit-mm-per-prompt image=0 audio=0) for first boot
|
||||
# Max ctx: 16K (conservative for first boot; can extend after validation)
|
||||
# Genesis: none — preview path on vllm-nightly-clean
|
||||
# Status: 🔵 PREVIEW — v0.7.3 MoE onboarding primary bench target
|
||||
# Caveats: Cliff 2 mitigations UNAVAILABLE without Genesis. Do NOT use
|
||||
# past ~21-26K accumulated context — risk of OOM / instability.
|
||||
# Production path (Genesis-anchored, TQ3 KV, MTP, higher ctx)
|
||||
# is parked until Genesis v7.73.x re-anchors on a post-#42521
|
||||
# nightly. This compose exists to exercise PR #42521 (qwen3_5_moe
|
||||
# weight loading) and produce a first benchmark row.
|
||||
# ---------------------------------------------------------------------------
|
||||
# Qwen 3.6 35B-A3B (AutoRound INT4) — dual-card preview alongside Gemma 4
|
||||
# 26B-A4B for v0.7.3 MoE onboarding.
|
||||
#
|
||||
# Why TP=2 (not TP=1): weights ~20 GB. Single-card 24 GB leaves only ~2 GB
|
||||
# for KV pool + activations + cudagraph capture — boot OOMs are highly
|
||||
# likely. TP=2 splits weights to ~10 GB/card, leaving ~12 GB/card for KV.
|
||||
# num_kv_heads=2 caps valid_tp at [1, 2] — TP=2 is the maximum.
|
||||
#
|
||||
# Why vllm-nightly-clean (not vllm-nightly-mtp):
|
||||
# vllm-nightly-clean rides nightly-bf610c2f (2026-05-15), which INCLUDES
|
||||
# PR #42521 (qwen3_5_moe weight loading, merged 2026-05-14). The Genesis-
|
||||
# anchored vllm-nightly-mtp is pinned to nightly-01d4d1ad (Sander's
|
||||
# v7.72.2 PROD pin), which PRE-DATES that fix. Trade-off: no Genesis
|
||||
# patches → Cliff 2 stabilization missing, no TQ3 KV. Acceptable for
|
||||
# low-ctx smoke + first bench row; not acceptable for production
|
||||
# long-ctx workloads.
|
||||
#
|
||||
# Re-test trigger to graduate this compose to production-anchored variant:
|
||||
# Genesis v7.73.x is released → bump vllm-nightly-mtp.spec to a
|
||||
# post-#42521 nightly SHA → port this compose to engine vllm-nightly-mtp
|
||||
# → drop the "preview" naming → add TQ3 KV + MTP variant.
|
||||
#
|
||||
# Models:
|
||||
# target: Qwen/Qwen3-MoE-A3B-Instruct-AutoRound-Int4-mixed (path:
|
||||
# qwen3.6-35b-a3b-autoround-int4, arch:
|
||||
# Qwen3_5MoeForConditionalGeneration, ~20 GB on disk)
|
||||
# draft : none on this compose
|
||||
#
|
||||
# KV format:
|
||||
# - fp8_e5m2 chosen over fp8_e4m3 (Triton fp8e4nv unsupported on sm_86).
|
||||
# - TQ3 not available without Genesis.
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 2
|
||||
# Tensor-parallel: 2
|
||||
# Requires-sm: 7.5+
|
||||
services:
|
||||
vllm-qwen36-35b-a3b-preview-tp2:
|
||||
image: ${VLLM_IMAGE:-vllm/vllm-openai:nightly-${VLLM_NIGHTLY_SHA}}
|
||||
container_name: "${ESTATE_CONTAINER:-vllm-qwen36-35b-a3b-preview-tp2}"
|
||||
restart: "no"
|
||||
ports:
|
||||
- "${BIND_HOST:-0.0.0.0}:${ESTATE_PORT:-${PORT:-8051}}:8000"
|
||||
volumes:
|
||||
- ${MODEL_DIR:-../../../../../models-cache}:/root/.cache/huggingface
|
||||
- ../../cache/torch_compile:/root/.cache/vllm/torch_compile_cache
|
||||
- ../../cache/triton:/root/.triton/cache
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=${ESTATE_GPUS:-${NVIDIA_VISIBLE_DEVICES:-all}}
|
||||
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-}
|
||||
- VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||
- NCCL_CUMEM_ENABLE=0
|
||||
- NCCL_P2P_DISABLE=1
|
||||
- VLLM_NO_USAGE_STATS=1
|
||||
- PYTORCH_CUDA_ALLOC_CONF=${PYTORCH_CUDA_ALLOC_CONF:-expandable_segments:True}
|
||||
- OMP_NUM_THREADS=1
|
||||
shm_size: "16gb"
|
||||
ipc: host
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
command:
|
||||
- --host
|
||||
- 0.0.0.0
|
||||
- --port
|
||||
- "8000"
|
||||
- --model
|
||||
- /root/.cache/huggingface/qwen3.6-35b-a3b-autoround-int4
|
||||
- --served-model-name
|
||||
- qwen3.6-35b-a3b-autoround
|
||||
- --quantization
|
||||
- auto_round
|
||||
- --dtype
|
||||
- float16
|
||||
- --tensor-parallel-size
|
||||
- "${TP:-2}"
|
||||
- --pipeline-parallel-size
|
||||
- "${PP:-1}"
|
||||
- --max-model-len
|
||||
- "${MAX_MODEL_LEN:-16384}"
|
||||
- --gpu-memory-utilization
|
||||
- "${GPU_MEMORY_UTILIZATION:-0.92}"
|
||||
- --max-num-seqs
|
||||
- "1"
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --limit-mm-per-prompt
|
||||
- '{"image":0,"audio":0}'
|
||||
- --kv-cache-dtype
|
||||
- "${KV_CACHE_DTYPE:-fp8_e5m2}"
|
||||
- --trust-remote-code
|
||||
- --reasoning-parser
|
||||
- qwen3
|
||||
- --default-chat-template-kwargs
|
||||
- '{"enable_thinking": false}'
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- qwen3_coder
|
||||
- --enable-prefix-caching
|
||||
- --enable-chunked-prefill
|
||||
@@ -0,0 +1,118 @@
|
||||
# ===========================================================================
|
||||
# Profile (at-a-glance):
|
||||
# Model: Qwen 3.6 35B-A3B MoE (AutoRound INT4, ~20 GB weights)
|
||||
# Topology: Single 3090 (TP=1)
|
||||
# Drafter: none (smoke compose — built-in MTP head added in a follow-up
|
||||
# once base boot is validated)
|
||||
# KV: fp8_e5m2 (no TQ3 — Genesis-only, not available here)
|
||||
# Vision: off (limit-mm-per-prompt image=0 audio=0) for first boot
|
||||
# Max ctx: 8K (conservative — KV pool tight after 20 GB weights on 24 GB card)
|
||||
# Genesis: none — preview path on vllm-nightly-clean
|
||||
# Status: 🔵 PREVIEW — v0.7.3 MoE onboarding smoke
|
||||
# Caveats: Cliff 2 mitigations UNAVAILABLE without Genesis. Do NOT use
|
||||
# past ~21-26K accumulated context — risk of OOM / instability.
|
||||
# Production path (Genesis-anchored, TQ3 KV, MTP, higher ctx)
|
||||
# is parked until Genesis v7.73.x re-anchors on a post-#42521
|
||||
# nightly. This compose exists to exercise PR #42521 (qwen3_5_moe
|
||||
# weight loading) and produce a first benchmark row.
|
||||
# ---------------------------------------------------------------------------
|
||||
# Qwen 3.6 35B-A3B (AutoRound INT4) — first MoE model from the Qwen3-Next
|
||||
# family on club-3090, alongside Gemma 4 26B-A4B.
|
||||
#
|
||||
# Why vllm-nightly-clean (not vllm-nightly-mtp):
|
||||
# vllm-nightly-clean rides nightly-bf610c2f (2026-05-15), which INCLUDES
|
||||
# PR #42521 (qwen3_5_moe weight loading, merged 2026-05-14). The Genesis-
|
||||
# anchored vllm-nightly-mtp is pinned to nightly-1acd67a7 (2026-05-08),
|
||||
# which PRE-DATES that fix. Trade-off: no Genesis patches → Cliff 2
|
||||
# stabilization missing, no TQ3 KV. Acceptable for low-ctx smoke + first
|
||||
# bench row; not acceptable for production long-ctx workloads.
|
||||
#
|
||||
# Re-test trigger to graduate this compose to production-anchored variant:
|
||||
# Genesis v7.73.x is released → bump vllm-nightly-mtp.spec to a post-#42521
|
||||
# nightly SHA → port this compose to engine vllm-nightly-mtp → drop the
|
||||
# "preview" naming → add a Cliff-2-mitigated dual.yml.
|
||||
#
|
||||
# Models:
|
||||
# target: Qwen/Qwen3-MoE-A3B-Instruct-AutoRound-Int4-mixed (path:
|
||||
# qwen3.6-35b-a3b-autoround-int4, arch:
|
||||
# Qwen3_5MoeForConditionalGeneration, ~20 GB on disk)
|
||||
# draft : none on this compose
|
||||
#
|
||||
# KV format:
|
||||
# - fp8_e5m2 chosen over fp8_e4m3 (Triton fp8e4nv unsupported on sm_86).
|
||||
# - TQ3 not available without Genesis.
|
||||
# ===========================================================================
|
||||
# Hardware metadata (parsed by scripts/preflight.sh):
|
||||
# Requires-min-vram-gb: 24
|
||||
# Engine-profile: vllm-nightly-clean
|
||||
# Requires-min-gpu-count: 1
|
||||
# Tensor-parallel: 1
|
||||
# Requires-sm: 7.5+
|
||||
services:
|
||||
vllm-qwen36-35b-a3b-preview:
|
||||
image: ${VLLM_IMAGE:-vllm/vllm-openai:nightly-${VLLM_NIGHTLY_SHA}}
|
||||
container_name: "${ESTATE_CONTAINER:-vllm-qwen36-35b-a3b-preview}"
|
||||
restart: "no"
|
||||
ports:
|
||||
- "${BIND_HOST:-0.0.0.0}:${ESTATE_PORT:-${PORT:-8050}}:8000"
|
||||
volumes:
|
||||
- ${MODEL_DIR:-../../../../../models-cache}:/root/.cache/huggingface
|
||||
- ../../cache/torch_compile:/root/.cache/vllm/torch_compile_cache
|
||||
- ../../cache/triton:/root/.triton/cache
|
||||
environment:
|
||||
- NVIDIA_VISIBLE_DEVICES=${ESTATE_GPUS:-${NVIDIA_VISIBLE_DEVICES:-all}}
|
||||
- HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-}
|
||||
- VLLM_WORKER_MULTIPROC_METHOD=spawn
|
||||
- NCCL_CUMEM_ENABLE=0
|
||||
- NCCL_P2P_DISABLE=1
|
||||
- VLLM_NO_USAGE_STATS=1
|
||||
- PYTORCH_CUDA_ALLOC_CONF=${PYTORCH_CUDA_ALLOC_CONF:-expandable_segments:True}
|
||||
- OMP_NUM_THREADS=1
|
||||
shm_size: "16gb"
|
||||
ipc: host
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: all
|
||||
capabilities: [gpu]
|
||||
command:
|
||||
- --host
|
||||
- 0.0.0.0
|
||||
- --port
|
||||
- "8000"
|
||||
- --model
|
||||
- /root/.cache/huggingface/qwen3.6-35b-a3b-autoround-int4
|
||||
- --served-model-name
|
||||
- qwen3.6-35b-a3b-autoround
|
||||
- --quantization
|
||||
- auto_round
|
||||
- --dtype
|
||||
- float16
|
||||
- --tensor-parallel-size
|
||||
- "${TP:-1}"
|
||||
- --pipeline-parallel-size
|
||||
- "${PP:-1}"
|
||||
- --max-model-len
|
||||
- "${MAX_MODEL_LEN:-8192}"
|
||||
- --gpu-memory-utilization
|
||||
- "${GPU_MEMORY_UTILIZATION:-0.92}"
|
||||
- --max-num-seqs
|
||||
- "1"
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --limit-mm-per-prompt
|
||||
- '{"image":0,"audio":0}'
|
||||
- --kv-cache-dtype
|
||||
- "${KV_CACHE_DTYPE:-fp8_e5m2}"
|
||||
- --trust-remote-code
|
||||
- --reasoning-parser
|
||||
- qwen3
|
||||
- --default-chat-template-kwargs
|
||||
- '{"enable_thinking": false}'
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- qwen3_coder
|
||||
- --enable-prefix-caching
|
||||
- --enable-chunked-prefill
|
||||
+134
-5
@@ -7,6 +7,7 @@
|
||||
# - per-run: wall time, TTFT (via streaming), completion tokens,
|
||||
# wall_TPS (= comp / wall), decode_TPS (= comp / (wall - TTFT))
|
||||
# - per-prompt summary: mean / std / CV for both TPS metrics + mean TTFT
|
||||
# + prompt-processing throughput (`PP tok/s`)
|
||||
# - shows MTP SpecDecoding metrics from docker logs at the end
|
||||
#
|
||||
# Why two TPS metrics:
|
||||
@@ -35,10 +36,15 @@
|
||||
# MAX_TOKENS_CODE Default: 800
|
||||
# ONLY Set to "narr" or "code" to skip the other. Default: both
|
||||
# QUIET Set to 1 to skip per-run lines (just print summary)
|
||||
# PP Set to 1 to add the long-prompt PP fallback probe.
|
||||
# llama.cpp containers enable this automatically.
|
||||
# PP_FALLBACK_TOKENS Approximate filler-token target for PP=1. Default: 10000
|
||||
# PP_MAX_TOKENS Completion cap for the PP fallback request. Default: 16
|
||||
#
|
||||
# Usage:
|
||||
# bash scripts/bench.sh
|
||||
# ONLY=code bash scripts/bench.sh
|
||||
# PP=1 bash scripts/bench.sh
|
||||
# RUNS=10 bash scripts/bench.sh
|
||||
|
||||
set -euo pipefail
|
||||
@@ -49,7 +55,7 @@ ROOT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
if [[ -f "${ROOT_DIR}/scripts/preflight.sh" ]]; then
|
||||
# shellcheck source=preflight.sh
|
||||
source "${ROOT_DIR}/scripts/preflight.sh"
|
||||
preflight_autodetect_endpoint
|
||||
preflight_autodetect_endpoint || true
|
||||
fi
|
||||
URL="${URL:-http://localhost:8020}"
|
||||
MODEL="${MODEL:-qwen3.6-27b-autoround}"
|
||||
@@ -62,6 +68,9 @@ PROMPT_NARR="${PROMPT_NARR:-Write a detailed 800-word essay explaining transform
|
||||
PROMPT_CODE="${PROMPT_CODE:-Write a Python implementation of quicksort with comments explaining each step.}"
|
||||
ONLY="${ONLY:-both}"
|
||||
QUIET="${QUIET:-0}"
|
||||
PP="${PP:-0}"
|
||||
PP_FALLBACK_TOKENS="${PP_FALLBACK_TOKENS:-10000}"
|
||||
PP_MAX_TOKENS="${PP_MAX_TOKENS:-16}"
|
||||
|
||||
need() {
|
||||
command -v "$1" >/dev/null 2>&1 || { echo "ERROR: '$1' not in PATH." >&2; exit 1; }
|
||||
@@ -69,6 +78,51 @@ need() {
|
||||
need curl
|
||||
need python3
|
||||
|
||||
ENGINE_KIND="${ENGINE_KIND:-unknown}"
|
||||
if [[ "$ENGINE_KIND" == "unknown" && "${CONTAINER:-}" != "none" ]] && command -v docker >/dev/null 2>&1 && docker inspect "${CONTAINER}" >/dev/null 2>&1; then
|
||||
container_image="$(docker inspect --format '{{.Config.Image}}' "${CONTAINER}" 2>/dev/null || true)"
|
||||
container_name="$(docker inspect --format '{{.Name}}' "${CONTAINER}" 2>/dev/null || true)"
|
||||
if [[ "${container_image} ${container_name}" == *"llama.cpp"* || "${container_image} ${container_name}" == *"llama-cpp"* ]]; then
|
||||
ENGINE_KIND="llamacpp"
|
||||
elif [[ "${container_image} ${container_name}" == *"vllm"* ]]; then
|
||||
ENGINE_KIND="vllm"
|
||||
fi
|
||||
fi
|
||||
|
||||
PP_MODE="log"
|
||||
if [[ "$PP" == "1" || "$ENGINE_KIND" == "llamacpp" ]]; then
|
||||
PP_MODE="fallback"
|
||||
fi
|
||||
|
||||
if [[ "${BENCH_MOCK:-0}" == "1" ]]; then
|
||||
if [[ "$PP_MODE" == "fallback" ]]; then
|
||||
cat <<'EOF'
|
||||
|
||||
========== PROMPT-PROCESSING (fallback target=10000 prompt tokens, max_tokens=16) ==========
|
||||
=== measured (1) ===
|
||||
run-1 wall= 3.20s ttft= 2500ms prompt_toks= 9876 PP_tok/s=3950.40
|
||||
|
||||
=== summary [prompt-processing] (n=1) ===
|
||||
PP tok/s mean=3950.40 std= 0.00 CV= 0.0% min=3950.40 max=3950.40
|
||||
TTFT mean= 2500ms std= 0ms min=2500ms max=2500ms
|
||||
EOF
|
||||
else
|
||||
cat <<'EOF'
|
||||
|
||||
========== NARRATIVE (prompt=61 chars, max_tokens=1000) ==========
|
||||
=== measured (1) ===
|
||||
run-1 wall= 4.20s ttft= 120ms toks=1000 wall_TPS=238.10 decode_TPS=245.10
|
||||
|
||||
=== summary [narrative] (n=1) ===
|
||||
wall_TPS mean= 238.10 std= 0.00 CV= 0.0% min=238.10 max=238.10
|
||||
decode_TPS mean= 245.10 std= 0.00 CV= 0.0% min=245.10 max=245.10
|
||||
TTFT mean= 120ms std= 0ms min=120ms max=120ms
|
||||
PP tok/s mean=2843.21 std= 0.00 CV= 0.0% min=2843.21 max=2843.21
|
||||
EOF
|
||||
fi
|
||||
exit 0
|
||||
fi
|
||||
|
||||
if ! curl -sf "${URL}/v1/models" >/dev/null; then
|
||||
echo "ERROR: service not reachable at ${URL}/v1/models" >&2
|
||||
echo " Start with: cd compose && docker compose up -d" >&2
|
||||
@@ -76,14 +130,17 @@ if ! curl -sf "${URL}/v1/models" >/dev/null; then
|
||||
fi
|
||||
|
||||
python3 - "$URL" "$MODEL" "$WARMUPS" "$RUNS" "$QUIET" "$ONLY" \
|
||||
"$CONTAINER" "$PP_MODE" "$PP_FALLBACK_TOKENS" "$PP_MAX_TOKENS" \
|
||||
"$PROMPT_NARR" "$MAX_TOKENS_NARR" \
|
||||
"$PROMPT_CODE" "$MAX_TOKENS_CODE" << 'PYEOF'
|
||||
import json, sys, time, urllib.request, statistics as s
|
||||
import json, re, shutil, subprocess, sys, time, urllib.request, statistics as s
|
||||
|
||||
(URL, MODEL, WARMUPS, RUNS, QUIET, ONLY,
|
||||
CONTAINER, PP_MODE, PP_FALLBACK_TOKENS, PP_MAX_TOKENS,
|
||||
PROMPT_NARR, MAX_NARR, PROMPT_CODE, MAX_CODE) = sys.argv[1:]
|
||||
WARMUPS = int(WARMUPS); RUNS = int(RUNS); QUIET = int(QUIET) == 1
|
||||
MAX_NARR = int(MAX_NARR); MAX_CODE = int(MAX_CODE)
|
||||
PP_FALLBACK_TOKENS = int(PP_FALLBACK_TOKENS); PP_MAX_TOKENS = int(PP_MAX_TOKENS)
|
||||
|
||||
def run_once(prompt, max_tokens):
|
||||
body = json.dumps({
|
||||
@@ -101,6 +158,7 @@ def run_once(prompt, max_tokens):
|
||||
t_send = time.time()
|
||||
ttft = None
|
||||
completion_tokens = 0
|
||||
prompt_tokens = 0
|
||||
with urllib.request.urlopen(req, timeout=600) as r:
|
||||
for line in r:
|
||||
line = line.decode("utf-8", errors="ignore").rstrip()
|
||||
@@ -122,11 +180,14 @@ def run_once(prompt, max_tokens):
|
||||
usage = chunk.get("usage")
|
||||
if usage:
|
||||
completion_tokens = usage.get("completion_tokens", completion_tokens)
|
||||
prompt_tokens = usage.get("prompt_tokens", prompt_tokens)
|
||||
t_end = time.time()
|
||||
wall = t_end - t_send
|
||||
if ttft is None:
|
||||
ttft = wall
|
||||
return wall, ttft, completion_tokens
|
||||
if not prompt_tokens:
|
||||
prompt_tokens = max(1, len(prompt.split()))
|
||||
return wall, ttft, completion_tokens, prompt_tokens
|
||||
|
||||
def fmt(label, wall, ttft, toks):
|
||||
decode_t = max(wall - ttft, 1e-6)
|
||||
@@ -135,18 +196,44 @@ def fmt(label, wall, ttft, toks):
|
||||
line = f" {label:<10s} wall={wall:6.2f}s ttft={ttft*1000:6.0f}ms toks={toks:>4d} wall_TPS={wtps:6.2f} decode_TPS={dtps:6.2f}"
|
||||
return wtps, dtps, ttft, line
|
||||
|
||||
def fmt_pp(label, wall, ttft, prompt_tokens):
|
||||
pp = prompt_tokens / max(ttft, 1e-6)
|
||||
line = f" {label:<10s} wall={wall:6.2f}s ttft={ttft*1000:6.0f}ms prompt_toks={prompt_tokens:>6d} PP_tok/s={pp:7.2f}"
|
||||
return pp, ttft, line
|
||||
|
||||
def stats(name, xs, unit=""):
|
||||
m = s.mean(xs)
|
||||
sd = s.stdev(xs) if len(xs) > 1 else 0
|
||||
cv = (sd / m * 100) if m > 0 else 0
|
||||
return f" {name:<14s} mean={m:7.2f}{unit} std={sd:6.2f} CV={cv:4.1f}% min={min(xs):.2f} max={max(xs):.2f}"
|
||||
|
||||
def scrape_prompt_throughput(container, n):
|
||||
if not container or container == "none" or shutil.which("docker") is None:
|
||||
return []
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["docker", "logs", container],
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
text=True,
|
||||
errors="replace",
|
||||
timeout=10,
|
||||
check=False,
|
||||
)
|
||||
except Exception:
|
||||
return []
|
||||
vals = [
|
||||
float(m.group(1))
|
||||
for m in re.finditer(r"Avg prompt throughput:\s*([0-9]+(?:\.[0-9]+)?)\s*tokens/s", proc.stdout)
|
||||
]
|
||||
return vals[-max(n, 1):]
|
||||
|
||||
def run_set(label, prompt, max_tokens):
|
||||
print(f"\n========== {label.upper()} (prompt={len(prompt)} chars, max_tokens={max_tokens}) ==========")
|
||||
print(f"=== warmups ({WARMUPS}) ===")
|
||||
for i in range(WARMUPS):
|
||||
try:
|
||||
w, t, k = run_once(prompt, max_tokens)
|
||||
w, t, k, _ = run_once(prompt, max_tokens)
|
||||
_, _, _, line = fmt(f"warm-{i+1}", w, t, k)
|
||||
if not QUIET:
|
||||
print(line)
|
||||
@@ -156,7 +243,7 @@ def run_set(label, prompt, max_tokens):
|
||||
walls, decodes, ttfts = [], [], []
|
||||
for i in range(RUNS):
|
||||
try:
|
||||
w, t, k = run_once(prompt, max_tokens)
|
||||
w, t, k, _ = run_once(prompt, max_tokens)
|
||||
wtps, dtps, ttft, line = fmt(f"run-{i+1}", w, t, k)
|
||||
if not QUIET:
|
||||
print(line)
|
||||
@@ -168,11 +255,53 @@ def run_set(label, prompt, max_tokens):
|
||||
print(stats("wall_TPS", walls))
|
||||
print(stats("decode_TPS", decodes))
|
||||
print(f" TTFT mean={s.mean(ttfts)*1000:6.0f}ms std={s.stdev(ttfts)*1000 if len(ttfts) > 1 else 0:5.0f}ms min={min(ttfts)*1000:.0f}ms max={max(ttfts)*1000:.0f}ms")
|
||||
if PP_MODE == "log":
|
||||
pp_vals = scrape_prompt_throughput(CONTAINER, len(walls))
|
||||
if pp_vals:
|
||||
print(stats("PP tok/s", pp_vals))
|
||||
else:
|
||||
print(" PP tok/s n/a (vLLM log scrape unavailable; use PP=1 for long-prompt fallback)")
|
||||
else:
|
||||
print(" PP tok/s n/a (long-prompt fallback below)")
|
||||
|
||||
def long_prompt(target_tokens):
|
||||
filler = (
|
||||
"club3090 prompt processing calibration filler with stable token shape. "
|
||||
"This sentence is intentionally plain so tokenizer variance stays modest. "
|
||||
)
|
||||
words_per_chunk = max(len(filler.split()), 1)
|
||||
chunks = max(1, target_tokens // words_per_chunk)
|
||||
return (
|
||||
"Read the following calibration text. Reply with one concise sentence summarizing its purpose.\n\n"
|
||||
+ filler * chunks
|
||||
)
|
||||
|
||||
def run_pp_fallback():
|
||||
prompt = long_prompt(PP_FALLBACK_TOKENS)
|
||||
print(
|
||||
f"\n========== PROMPT-PROCESSING "
|
||||
f"(fallback target={PP_FALLBACK_TOKENS} prompt tokens, max_tokens={PP_MAX_TOKENS}) =========="
|
||||
)
|
||||
print("=== measured (1) ===")
|
||||
pp_vals, ttfts = [], []
|
||||
try:
|
||||
w, t, _k, prompt_tokens = run_once(prompt, PP_MAX_TOKENS)
|
||||
pp, ttft, line = fmt_pp("run-1", w, t, prompt_tokens)
|
||||
print(line)
|
||||
pp_vals.append(pp); ttfts.append(ttft)
|
||||
except Exception as e:
|
||||
print(f" run-1 FAIL: {e}")
|
||||
if pp_vals:
|
||||
print("\n=== summary [prompt-processing] (n=1) ===")
|
||||
print(stats("PP tok/s", pp_vals))
|
||||
print(f" TTFT mean={s.mean(ttfts)*1000:6.0f}ms std= 0ms min={min(ttfts)*1000:.0f}ms max={max(ttfts)*1000:.0f}ms")
|
||||
|
||||
if ONLY in ("both", "narr"):
|
||||
run_set("narrative", PROMPT_NARR, MAX_NARR)
|
||||
if ONLY in ("both", "code"):
|
||||
run_set("code", PROMPT_CODE, MAX_CODE)
|
||||
if PP_MODE == "fallback":
|
||||
run_pp_fallback()
|
||||
PYEOF
|
||||
|
||||
# GPU state
|
||||
|
||||
@@ -12,6 +12,13 @@ COMPOSE_BASE="$CLUB3090_DIR/services"
|
||||
DUAL_27B_DIR="$CLUB3090_DIR/models/qwen3.6-27b/vllm/compose/dual"
|
||||
GEMMA_DUAL_DIR="$CLUB3090_DIR/models/gemma-4-31b/vllm/compose/dual"
|
||||
|
||||
# Estate planner state file (v0.7.0+). Instances booted via launch.sh --estate
|
||||
# or --estate-file are tracked here and persist via Docker `restart:
|
||||
# unless-stopped`, so they DO survive a plain mode-switch unless explicitly
|
||||
# torn down via launch.sh --down-estate. mode_off uses this path to clean
|
||||
# them up alongside the older vLLM/Gemma/ComfyUI services.
|
||||
ESTATE_YAML="${HOME}/.club3090/estate.yml"
|
||||
|
||||
GREEN='\033[0;32m'
|
||||
YELLOW='\033[1;33m'
|
||||
RED='\033[0;31m'
|
||||
@@ -559,11 +566,33 @@ mode_bigmodel() {
|
||||
echo -e " --n-gpu-layers 99 --ctx-size 32768 --host 0.0.0.0 --port 8001"
|
||||
}
|
||||
|
||||
stop_estate() {
|
||||
# Tear down any estate-managed instances (launch.sh --estate-file or --estate
|
||||
# bookings persist via Docker `restart: unless-stopped`). No-op if no estate
|
||||
# plan exists or launch.sh is unavailable.
|
||||
if [[ ! -f "$ESTATE_YAML" ]]; then
|
||||
return 0
|
||||
fi
|
||||
if ! command -v bash >/dev/null 2>&1 || [[ ! -x "$CLUB3090_DIR/scripts/launch.sh" ]]; then
|
||||
return 0
|
||||
fi
|
||||
if ! python3 -c "import yaml; d=yaml.safe_load(open('$ESTATE_YAML')); raise SystemExit(0 if d and d.get('estate') else 1)" 2>/dev/null; then
|
||||
return 0 # empty/missing estate list
|
||||
fi
|
||||
printf " ${RED}▼${NC} Stopping estate-managed instances..."
|
||||
if bash "$CLUB3090_DIR/scripts/launch.sh" --down-estate "$ESTATE_YAML" >/dev/null 2>&1; then
|
||||
echo "done"
|
||||
else
|
||||
echo "skipped (no instances or already down)"
|
||||
fi
|
||||
}
|
||||
|
||||
mode_off() {
|
||||
echo -e "${CYAN}═══ Stopping ALL services ═══${NC}"
|
||||
stop_all_27b
|
||||
stop_all_gemma
|
||||
stop_comfyui
|
||||
stop_estate
|
||||
for svc in "${SERVICES[@]}"; do
|
||||
stop_service "$svc"
|
||||
done
|
||||
|
||||
@@ -11,8 +11,10 @@
|
||||
# bash scripts/launch.sh --variant <name> # skip wizard, boot directly
|
||||
# bash scripts/launch.sh --estate # multi-model estate wizard
|
||||
# bash scripts/launch.sh --estate-file <path> # boot an existing estate plan
|
||||
# bash scripts/launch.sh --estate-file <path> --parallel --parallel-jobs 3 --parallel-stagger 30
|
||||
# bash scripts/launch.sh --validate-estate <path> # validate estate.yml, no boot
|
||||
# bash scripts/launch.sh --down-estate <path> # stop estate instances
|
||||
# bash scripts/launch.sh --topology # print GPU topology advisory, no boot
|
||||
# bash scripts/launch.sh --model qwen3.6-27b --gpus 0,1
|
||||
# bash scripts/launch.sh --engine vllm --cards 1 # deprecated; prefer --gpus
|
||||
# bash scripts/launch.sh --workload long-ctx-single # profile-aware filter
|
||||
@@ -66,6 +68,10 @@ ESTATE_FILE=""
|
||||
VALIDATE_ESTATE=""
|
||||
DOWN_ESTATE=""
|
||||
ONLY_NAMES=""
|
||||
TOPOLOGY_ONLY=0
|
||||
PARALLEL_BOOT=0
|
||||
PARALLEL_JOBS=""
|
||||
PARALLEL_STAGGER=""
|
||||
CARDS=""
|
||||
VARIANT=""
|
||||
MODEL_NAME=""
|
||||
@@ -87,6 +93,10 @@ while [[ $# -gt 0 ]]; do
|
||||
--validate-estate) VALIDATE_ESTATE="$2"; shift 2 ;;
|
||||
--down-estate) DOWN_ESTATE="$2"; shift 2 ;;
|
||||
--only) ONLY_NAMES="$2"; shift 2 ;;
|
||||
--topology) TOPOLOGY_ONLY=1; SKIP_PREFLIGHT=1; shift ;;
|
||||
--parallel) PARALLEL_BOOT=1; shift ;;
|
||||
--parallel-jobs) PARALLEL_BOOT=1; PARALLEL_JOBS="$2"; shift 2 ;;
|
||||
--parallel-stagger) PARALLEL_BOOT=1; PARALLEL_STAGGER="$2"; shift 2 ;;
|
||||
--engine) ENGINE="$2"; shift 2 ;;
|
||||
--workload) WORKLOAD_ID="$2"; shift 2 ;;
|
||||
--drafter) DRAFTER_ID="$2"; shift 2 ;;
|
||||
@@ -602,6 +612,77 @@ selected_gpu_profile_spec() {
|
||||
printf '%s' "$joined"
|
||||
}
|
||||
|
||||
select_topology_gpus() {
|
||||
GPU_LINES="$(compose_hw_detect_gpus 2>/dev/null || true)"
|
||||
[[ -n "$GPU_LINES" ]] || return 1
|
||||
CARD_INDICES=()
|
||||
CARD_NAMES=()
|
||||
CARD_MEM_MIB=()
|
||||
CARD_SM=()
|
||||
|
||||
if [[ -n "$GPU_ARG" && "$GPU_ARG" != "all" ]]; then
|
||||
IFS=',' read -ra _launch_topology_tokens <<< "$GPU_ARG"
|
||||
local idx
|
||||
for idx in "${_launch_topology_tokens[@]}"; do
|
||||
idx="$(_compose_meta_trim "$idx")"
|
||||
[[ -z "$idx" ]] && continue
|
||||
gpu_exists "$idx" || { echo "[launch] ERROR: requested GPU ${idx}, but it was not detected." >&2; exit 1; }
|
||||
append_selected_gpu "$idx"
|
||||
done
|
||||
elif [[ -n "$CARDS" ]]; then
|
||||
[[ "$CARDS" =~ ^[0-9]+$ && "$CARDS" -ge 1 ]] || { echo "[launch] ERROR: --cards expects a positive integer." >&2; exit 1; }
|
||||
local idx name mem_mib sm selected=0
|
||||
while IFS=$'\t' read -r idx name mem_mib sm; do
|
||||
[[ -z "$idx" ]] && continue
|
||||
append_selected_gpu "$idx"
|
||||
selected=$((selected + 1))
|
||||
(( selected >= CARDS )) && break
|
||||
done <<< "$GPU_LINES"
|
||||
(( selected == CARDS )) || { echo "[launch] ERROR: --cards ${CARDS} requested, but only ${selected} GPU(s) were detected." >&2; exit 1; }
|
||||
else
|
||||
local idx name mem_mib sm
|
||||
while IFS=$'\t' read -r idx name mem_mib sm; do
|
||||
[[ -z "$idx" ]] && continue
|
||||
append_selected_gpu "$idx"
|
||||
done <<< "$GPU_LINES"
|
||||
fi
|
||||
|
||||
[[ "${#CARD_INDICES[@]}" -gt 0 ]] || return 1
|
||||
SELECTED_GPU_CSV="$(IFS=','; echo "${CARD_INDICES[*]}")"
|
||||
summarize_selected_vram >/dev/null
|
||||
return 0
|
||||
}
|
||||
|
||||
print_topology_advisory() {
|
||||
local output
|
||||
output="$(python3 "$LAUNCH_PROFILE" topology --gpu-spec "$(selected_gpu_profile_spec)" --format wizard 2>&1)" || {
|
||||
echo "$output" >&2
|
||||
exit 2
|
||||
}
|
||||
if [[ -n "$output" ]]; then
|
||||
echo "$output" >&2
|
||||
fi
|
||||
}
|
||||
|
||||
print_topology_and_exit() {
|
||||
local output
|
||||
if ! select_topology_gpus; then
|
||||
echo "Detected hardware:"
|
||||
echo " no NVIDIA GPUs detected"
|
||||
echo ""
|
||||
echo "Topology class: unavailable"
|
||||
echo ""
|
||||
echo "For details, see docs/MULTI_CARD.md."
|
||||
exit 0
|
||||
fi
|
||||
output="$(python3 "$LAUNCH_PROFILE" topology --gpu-spec "$(selected_gpu_profile_spec)" --format standalone 2>&1)" || {
|
||||
echo "$output" >&2
|
||||
exit 0
|
||||
}
|
||||
echo "$output"
|
||||
exit 0
|
||||
}
|
||||
|
||||
launch_nvlink_active() {
|
||||
if [[ "${#CARD_INDICES[@]}" -ne 2 ]]; then
|
||||
printf '0'
|
||||
@@ -927,11 +1008,18 @@ if [[ "$ESTATE_MODE" -eq 1 || -n "$ESTATE_FILE" ]]; then
|
||||
else
|
||||
_estate_cmd=(python3 "$ESTATE_HELPER" boot --file "$ESTATE_FILE")
|
||||
[[ -n "$ONLY_NAMES" ]] && _estate_cmd+=(--only "$ONLY_NAMES")
|
||||
[[ "$PARALLEL_BOOT" -eq 1 ]] && _estate_cmd+=(--parallel)
|
||||
[[ -n "$PARALLEL_JOBS" ]] && _estate_cmd+=(--parallel-jobs "$PARALLEL_JOBS")
|
||||
[[ -n "$PARALLEL_STAGGER" ]] && _estate_cmd+=(--parallel-stagger "$PARALLEL_STAGGER")
|
||||
fi
|
||||
"${_estate_cmd[@]}"
|
||||
exit $?
|
||||
fi
|
||||
|
||||
if [[ "$TOPOLOGY_ONLY" -eq 1 ]]; then
|
||||
print_topology_and_exit
|
||||
fi
|
||||
|
||||
# --- wizard ---
|
||||
if [[ -z "$VARIANT" ]]; then
|
||||
echo "" >&2
|
||||
@@ -939,6 +1027,7 @@ if [[ -z "$VARIANT" ]]; then
|
||||
echo "(Use --variant <name> next time to skip the wizard.)" >&2
|
||||
choose_model
|
||||
choose_gpus
|
||||
print_topology_advisory
|
||||
pick_parallelism
|
||||
if [[ "$MODEL_NAME" == "gemma-4-31b" && "${#CARD_INDICES[@]}" -eq 1 && "$MIN_VRAM_GB" -lt 32 ]]; then
|
||||
gemma_single_24gb_guidance
|
||||
|
||||
@@ -266,6 +266,26 @@ def peak_vram(internal: dict[str, Any], gpu_count: int) -> str:
|
||||
return f"{max(vals) / 1024:.1f} GB{suffix}"
|
||||
|
||||
|
||||
def pp_display(bench: dict[str, Any]) -> str:
|
||||
vals: list[float] = []
|
||||
for kind in ("narrative", "code"):
|
||||
try:
|
||||
val = bench.get(kind, {}).get("pp_tps_mean")
|
||||
if val is not None:
|
||||
vals.append(float(val))
|
||||
except Exception:
|
||||
pass
|
||||
if vals:
|
||||
return f"{sum(vals) / len(vals):.0f}"
|
||||
try:
|
||||
val = bench.get("prompt_processing", {}).get("pp_tps_mean")
|
||||
if val is not None:
|
||||
return f"{float(val):.0f}"
|
||||
except Exception:
|
||||
pass
|
||||
return "—"
|
||||
|
||||
|
||||
def parse_jsonish(value: str) -> dict[str, Any]:
|
||||
try:
|
||||
parsed = json.loads(value)
|
||||
@@ -370,6 +390,7 @@ def format_row(data: dict[str, Any]) -> str:
|
||||
kv = kv_display(c["kv"], c["served"])
|
||||
max_ctx = fmt_ctx(c["max_ctx"])
|
||||
tps = f"**{fmt_tps(narrative.get('wall_tps_mean'))} / {fmt_tps(code.get('wall_tps_mean'))}**"
|
||||
pp = pp_display(bench)
|
||||
peak = peak_vram(internal, gpu_count)
|
||||
spec_n = c["spec"].get("num_speculative_tokens")
|
||||
notes = [soak_note(soak), f"verify-stress {verify}"]
|
||||
@@ -393,12 +414,12 @@ def format_row(data: dict[str, Any]) -> str:
|
||||
per_pos = str(mtp.get("per_position") or "—")
|
||||
return (
|
||||
f"| `{compose}` | {rig_cell(rig)} | {kv} | {max_ctx} | {tps} | "
|
||||
f"{al} | {per_pos} | {peak} | {data['date']} | {note_cell} |"
|
||||
f"{pp} | {al} | {per_pos} | {peak} | {data['date']} | {note_cell} |"
|
||||
)
|
||||
|
||||
return (
|
||||
f"| `{compose}` | {rig_cell(rig)} | {kv} | {max_ctx} | {tps} | "
|
||||
f"{peak} | {data['date']} | {note_cell} |"
|
||||
f"{pp} | {peak} | {data['date']} | {note_cell} |"
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
schema_version: 1
|
||||
model: gemma-4-26b-a4b
|
||||
# Low-anchor calibration: both rows use the same bf16 / 32K / seqs=256 / TP=2
|
||||
# envelope and only vary the external MTP assistant. Add lower-concurrency and
|
||||
# longer-context rows before tightening the MoE activation coefficient.
|
||||
rows:
|
||||
- compose: vllm/gemma-a4b-awq
|
||||
vram_gb: 24
|
||||
measured_peak_gb: 23.45
|
||||
ctx_override: null
|
||||
status: low-anchor-calibration
|
||||
engine_pin: vllm-nightly-bf610c2f
|
||||
genesis_pin: null
|
||||
source: "BENCHMARKS.md#MoE models gemma-4-26b-a4b/dual/awq.yml @noonghunna 2026-05-15"
|
||||
- compose: vllm/gemma-a4b-awq-mtp
|
||||
vram_gb: 24
|
||||
measured_peak_gb: 23.50
|
||||
ctx_override: null
|
||||
status: low-anchor-calibration
|
||||
engine_pin: vllm-nightly-bf610c2f
|
||||
genesis_pin: null
|
||||
source: "BENCHMARKS.md#MoE models gemma-4-26b-a4b/dual/awq-mtp.yml @noonghunna 2026-05-15"
|
||||
@@ -0,0 +1,22 @@
|
||||
schema_version: 1
|
||||
model: qwen3.6-35b-a3b
|
||||
# Low-anchor calibration: both rows use the same fp8_e5m2 / 16K / seqs=1 /
|
||||
# TP=2 envelope and only vary built-in MTP. Add max_ctx / max_num_seqs A/B
|
||||
# rows before tightening the MoE activation coefficient.
|
||||
rows:
|
||||
- compose: vllm/qwen-a3b-preview
|
||||
vram_gb: 24
|
||||
measured_peak_gb: 21.94
|
||||
ctx_override: null
|
||||
status: low-anchor-calibration
|
||||
engine_pin: vllm-nightly-bf610c2f
|
||||
genesis_pin: null
|
||||
source: "BENCHMARKS.md#MoE models qwen3.6-35b-a3b/dual/preview.yml @noonghunna 2026-05-15"
|
||||
- compose: vllm/qwen-a3b-preview-mtp
|
||||
vram_gb: 24
|
||||
measured_peak_gb: 22.72
|
||||
ctx_override: null
|
||||
status: low-anchor-calibration
|
||||
engine_pin: vllm-nightly-bf610c2f
|
||||
genesis_pin: null
|
||||
source: "BENCHMARKS.md#MoE models qwen3.6-35b-a3b/dual/preview-mtp.yml @noonghunna 2026-05-15"
|
||||
@@ -12,6 +12,7 @@ import os
|
||||
import subprocess
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
|
||||
@@ -26,7 +27,7 @@ from .compose_registry import COMPOSE_REGISTRY
|
||||
SUPPORTED_SCHEMA_VERSIONS = {1}
|
||||
PROFILE_ROOT = Path(__file__).resolve().parent
|
||||
REPO_ROOT = Path(__file__).resolve().parents[3]
|
||||
CONSTRAINT_IDS = [f"C{i}" for i in range(1, 16)]
|
||||
CONSTRAINT_IDS = [f"C{i}" for i in range(1, 17)]
|
||||
ESTATE_CONSTRAINT_IDS = [f"E{i}" for i in range(1, 5)]
|
||||
|
||||
|
||||
@@ -42,6 +43,37 @@ class CrossReferenceError(ProfileError):
|
||||
"""Raised when a profile references a missing profile id."""
|
||||
|
||||
|
||||
class TopologyClass(str, Enum):
|
||||
SINGLE_CARD = "single_card"
|
||||
HOMOGENEOUS = "homogeneous"
|
||||
VRAM_MATCHED_COMPUTE_MISMATCHED = "vram_matched_compute_mismatched"
|
||||
VRAM_MISMATCHED = "vram_mismatched"
|
||||
HETEROGENEOUS_MIXED = "heterogeneous_mixed"
|
||||
|
||||
|
||||
TOPOLOGY_ADVISORY = {
|
||||
TopologyClass.SINGLE_CARD: None,
|
||||
TopologyClass.HOMOGENEOUS: None,
|
||||
TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED: (
|
||||
"Compute mismatch detected (VRAM matched). TP=N works fine but the faster card "
|
||||
"waits at every NCCL allreduce — effective throughput caps at slower card's speed "
|
||||
"(~30% of faster card idle at allreduce). Full per-card VRAM capacity preserved. "
|
||||
"Alternative: estate planner (--estate) to run different models per card at full speed."
|
||||
),
|
||||
TopologyClass.VRAM_MISMATCHED: (
|
||||
"VRAM mismatch detected. TP=N would cap to smaller card's usable model size. "
|
||||
"Recommended paths: (a) llama.cpp `--tensor-split` for weighted layer split, "
|
||||
"(b) PP=N (manual flag flip — `--pipeline-parallel-size N` on a vllm/dual compose; "
|
||||
"no shipping PP compose), (c) estate planner (--estate) to run different models per card."
|
||||
),
|
||||
TopologyClass.HETEROGENEOUS_MIXED: (
|
||||
"Heterogeneous hardware detected (multiple VRAM and compute tiers). Manual selection "
|
||||
"recommended. Consider the estate planner (--estate) to put different models on "
|
||||
"different card subsets, or run a single model on the largest matched subset."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def _logger() -> logging.Logger:
|
||||
logger = logging.getLogger("compat")
|
||||
if not logger.handlers:
|
||||
@@ -92,6 +124,30 @@ class HardwareProfile:
|
||||
notes: Optional[str] = None
|
||||
|
||||
|
||||
def classify_hardware_topology(hardware: list[HardwareProfile]) -> TopologyClass:
|
||||
"""Classify selected GPUs for TP-vs-PP/estate advisory output."""
|
||||
if not hardware:
|
||||
raise ProfileError("classify_hardware_topology requires at least one HardwareProfile")
|
||||
if len(hardware) == 1:
|
||||
return TopologyClass.SINGLE_CARD
|
||||
|
||||
vrams = sorted(hw.vram_gb for hw in hardware)
|
||||
sms = {hw.sm for hw in hardware}
|
||||
|
||||
vram_clusters = 1
|
||||
for i in range(1, len(vrams)):
|
||||
if vrams[i] - vrams[i - 1] > 1.0:
|
||||
vram_clusters += 1
|
||||
|
||||
if vram_clusters == 1 and len(sms) == 1:
|
||||
return TopologyClass.HOMOGENEOUS
|
||||
if vram_clusters == 1 and len(sms) > 1:
|
||||
return TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED
|
||||
if vram_clusters > 1:
|
||||
return TopologyClass.VRAM_MISMATCHED
|
||||
return TopologyClass.HETEROGENEOUS_MIXED
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ModelProfile:
|
||||
schema_version: int
|
||||
@@ -123,6 +179,25 @@ class ModelProfile:
|
||||
head_dim_sliding: Optional[int] = None
|
||||
global_head_dim: Optional[int] = None
|
||||
sliding_window: Optional[int] = None
|
||||
# Asymmetric KV head counts for SWA-hybrid models where global layers
|
||||
# have a different KV head count than sliding layers (e.g. Gemma 4
|
||||
# 26B-A4B: 8 sliding, 2 global). Leave None for symmetric models.
|
||||
num_global_kv_heads: Optional[int] = None
|
||||
# MoE fields (None for dense models; set for MoE variants)
|
||||
num_experts: Optional[int] = None
|
||||
num_experts_per_tok: Optional[int] = None
|
||||
moe_intermediate_size: Optional[int] = None
|
||||
shared_expert_intermediate_size: Optional[int] = None
|
||||
active_params_b: Optional[float] = None
|
||||
# Optional architectural metadata
|
||||
mtp_num_hidden_layers: Optional[int] = None
|
||||
attn_output_gate: Optional[bool] = None
|
||||
vision_capable: Optional[bool] = None
|
||||
# C12 (KV projection via tools/kv-calc.py) only supports models whose
|
||||
# architecture has been added to MODEL_SPECS in kv-calc.py. New MoE /
|
||||
# hybrid models can set kv_calc_supported=false to skip C12 until
|
||||
# kv-calc gains MoE-aware activation/KV formulas.
|
||||
kv_calc_supported: bool = True
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -203,6 +278,7 @@ class FitsResult:
|
||||
world_size: Optional[int] = None
|
||||
bottleneck_vram_gb: Optional[float] = None
|
||||
homogeneous: Optional[bool] = None
|
||||
topology_class: Optional[TopologyClass] = None
|
||||
kv_projection: Optional[dict[str, Any]] = None
|
||||
compose_name: Optional[str] = None
|
||||
weights_variant: Optional[str] = None
|
||||
@@ -306,6 +382,15 @@ def _model(data: dict[str, Any]) -> ModelProfile:
|
||||
head_dim_sliding=data.get("head_dim_sliding"),
|
||||
global_head_dim=data.get("global_head_dim"),
|
||||
sliding_window=data.get("sliding_window"),
|
||||
num_global_kv_heads=data.get("num_global_kv_heads"),
|
||||
num_experts=data.get("num_experts"),
|
||||
num_experts_per_tok=data.get("num_experts_per_tok"),
|
||||
moe_intermediate_size=data.get("moe_intermediate_size"),
|
||||
shared_expert_intermediate_size=data.get("shared_expert_intermediate_size"),
|
||||
active_params_b=data.get("active_params_b"),
|
||||
mtp_num_hidden_layers=data.get("mtp_num_hidden_layers"),
|
||||
attn_output_gate=data.get("attn_output_gate"),
|
||||
vision_capable=data.get("vision_capable"),
|
||||
max_ctx_supported=int(data["max_ctx_supported"]),
|
||||
attention_k_eq_v=bool(data["attention_k_eq_v"]),
|
||||
weights=_dict(data.get("weights")),
|
||||
@@ -313,6 +398,7 @@ def _model(data: dict[str, Any]) -> ModelProfile:
|
||||
compatible_drafters=_tuple(data.get("compatible_drafters")),
|
||||
valid_tp=tuple(int(x) for x in _tuple(data.get("valid_tp"))),
|
||||
requires_genesis=bool(data.get("requires_genesis", False)),
|
||||
kv_calc_supported=bool(data.get("kv_calc_supported", True)),
|
||||
)
|
||||
|
||||
|
||||
@@ -576,6 +662,14 @@ def _kv_calc_weights_variant(model: ModelProfile, variant: str) -> str:
|
||||
if variant == "bf16":
|
||||
return "bf16"
|
||||
return "int4"
|
||||
if model.family == "gemma4-swa-moe":
|
||||
if variant == "awq_compressed_tensors":
|
||||
return "awq"
|
||||
return "int4"
|
||||
if model.family == "qwen3-next-moe":
|
||||
if variant == "gptq_int4":
|
||||
return "gptq"
|
||||
return "default"
|
||||
return "default"
|
||||
|
||||
|
||||
@@ -686,6 +780,7 @@ def fits(
|
||||
effective_max_num_seqs = max_num_seqs if max_num_seqs is not None else int(workload.defaults.get("max_num_seqs", 1))
|
||||
effective_weights = resolve_weights_variant(model, engine, weights_variant)
|
||||
homogeneous = len({hw.id for hw in hardware}) <= 1
|
||||
topology_class = classify_hardware_topology(hardware) if hardware else None
|
||||
bottleneck = min((hw.vram_gb for hw in hardware), default=None)
|
||||
effective_cudagraph = _cudagraph_mode(hardware)
|
||||
|
||||
@@ -793,6 +888,9 @@ def fits(
|
||||
elif engine.type != "vllm":
|
||||
skipped.append("C12")
|
||||
notes.append("KV projection not available for non-vLLM engines")
|
||||
elif not model.kv_calc_supported:
|
||||
skipped.append("C12")
|
||||
notes.append(f"KV projection skipped: {model.id} not yet wired into tools/kv-calc.py")
|
||||
else:
|
||||
kv_calc_invoked = True
|
||||
if effective_mem_util is None or bottleneck is None:
|
||||
@@ -830,6 +928,14 @@ def fits(
|
||||
else:
|
||||
ok("C12")
|
||||
|
||||
if topology_class is None:
|
||||
skip("C16", "Topology advisory not run; no hardware profiles provided.")
|
||||
else:
|
||||
ok("C16")
|
||||
advisory = TOPOLOGY_ADVISORY.get(topology_class)
|
||||
if advisory:
|
||||
notes.append(f"C16 topology={topology_class.value}: {advisory}")
|
||||
|
||||
diagnostics = {
|
||||
"constraints_evaluated": list(CONSTRAINT_IDS),
|
||||
"constraints_passed": passed,
|
||||
@@ -850,6 +956,7 @@ def fits(
|
||||
world_size=world_size,
|
||||
bottleneck_vram_gb=bottleneck,
|
||||
homogeneous=homogeneous,
|
||||
topology_class=topology_class,
|
||||
kv_projection=kv_projection,
|
||||
weights_variant=effective_weights,
|
||||
diagnostics=diagnostics,
|
||||
|
||||
@@ -88,14 +88,14 @@ COMPOSE_REGISTRY = {
|
||||
),
|
||||
"vllm/tools-text": _entry(
|
||||
model="qwen3.6-27b", weights_variant="autoround_int4", workload="tool-heavy",
|
||||
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||||
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||||
tp=1, max_ctx=75000, max_num_seqs=1, mem_util=0.97,
|
||||
compose_path="models/qwen3.6-27b/vllm/compose/single/tools-text.yml",
|
||||
default_port=8020,
|
||||
),
|
||||
"vllm/minimal": _entry(
|
||||
model="qwen3.6-27b", weights_variant="autoround_int4", workload="fast-chat",
|
||||
engine="vllm-nightly-mtp", drafter=None, kv_format="fp8_e5m2",
|
||||
engine="vllm-nightly-clean", drafter=None, kv_format="fp8_e5m2",
|
||||
tp=1, max_ctx=32768, max_num_seqs=1, mem_util=0.92,
|
||||
compose_path="models/qwen3.6-27b/vllm/compose/single/minimal.yml",
|
||||
default_port=8020,
|
||||
@@ -104,7 +104,7 @@ COMPOSE_REGISTRY = {
|
||||
# Qwen 3.6 27B, vLLM dual/multi-card.
|
||||
"vllm/dual": _entry(
|
||||
model="qwen3.6-27b", weights_variant="autoround_int4", workload="long-ctx-single",
|
||||
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||||
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||||
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
|
||||
compose_path="models/qwen3.6-27b/vllm/compose/dual/docker-compose.yml",
|
||||
default_port=8010, recommended_engine_features=["marlin_pad_sub_tile_n"],
|
||||
@@ -133,7 +133,7 @@ COMPOSE_REGISTRY = {
|
||||
),
|
||||
"vllm/dual-bf16": _entry(
|
||||
model="qwen3.6-27b", weights_variant="autoround_int4", workload="long-ctx-single",
|
||||
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="bf16",
|
||||
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="bf16",
|
||||
tp=2, max_ctx=200000, max_num_seqs=1, mem_util=0.92,
|
||||
compose_path="models/qwen3.6-27b/vllm/compose/dual/bf16.yml",
|
||||
default_port=8012,
|
||||
@@ -168,21 +168,21 @@ COMPOSE_REGISTRY = {
|
||||
),
|
||||
"vllm/dual-carnice-bf16mtp": _entry(
|
||||
model="qwen3.6-27b", weights_variant="carnice_bf16mtp", workload="long-ctx-single",
|
||||
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||||
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||||
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
|
||||
compose_path="models/qwen3.6-27b/vllm/compose/dual/carnice-bf16mtp.yml",
|
||||
default_port=8070,
|
||||
),
|
||||
"vllm/dual-qwopus-bf16mtp": _entry(
|
||||
model="qwen3.6-27b", weights_variant="qwopus_bf16mtp", workload="long-ctx-single",
|
||||
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||||
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||||
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
|
||||
compose_path="models/qwen3.6-27b/vllm/compose/dual/qwopus-bf16mtp.yml",
|
||||
default_port=8071,
|
||||
),
|
||||
"vllm/dual-nvlink": _entry(
|
||||
model="qwen3.6-27b", weights_variant="autoround_int4", workload="long-ctx-single",
|
||||
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||||
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||||
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
|
||||
compose_path="models/qwen3.6-27b/vllm/compose/dual/nvlink.yml",
|
||||
default_port=8014, requires_nvlink=True, recommended_engine_features=["marlin_pad_sub_tile_n"],
|
||||
@@ -211,7 +211,7 @@ COMPOSE_REGISTRY = {
|
||||
),
|
||||
"vllm/dual4": _entry(
|
||||
model="qwen3.6-27b", weights_variant="autoround_int4", workload="multi-stream-tenant",
|
||||
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||||
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||||
tp=4, max_ctx=262144, max_num_seqs=4, mem_util=0.92,
|
||||
compose_path="models/qwen3.6-27b/vllm/compose/multi4/docker-compose.yml",
|
||||
default_port=8015,
|
||||
@@ -243,14 +243,14 @@ COMPOSE_REGISTRY = {
|
||||
# Gemma 4 31B, vLLM.
|
||||
"vllm/gemma-mtp-tp1": _entry(
|
||||
model="gemma-4-31b", weights_variant="autoround_int4", workload="fast-chat",
|
||||
engine="vllm-nightly-mtp", drafter="gemma-it-assistant", kv_format="fp8_e4m3",
|
||||
engine="vllm-nightly-clean", drafter="gemma-it-assistant", kv_format="fp8_e4m3",
|
||||
tp=1, max_ctx=8192, max_num_seqs=256, mem_util=0.95,
|
||||
compose_path="models/gemma-4-31b/vllm/compose/single/docker-compose.yml",
|
||||
default_port=8031, required_sm=9.0,
|
||||
),
|
||||
"vllm/gemma-mtp": _entry(
|
||||
model="gemma-4-31b", weights_variant="autoround_int4", workload="fast-chat",
|
||||
engine="vllm-nightly-mtp", drafter="gemma-it-assistant", kv_format="bf16",
|
||||
engine="vllm-nightly-clean", drafter="gemma-it-assistant", kv_format="bf16",
|
||||
tp=2, max_ctx=32768, max_num_seqs=4, mem_util=0.92,
|
||||
compose_path="models/gemma-4-31b/vllm/compose/dual/docker-compose.yml",
|
||||
default_port=8030,
|
||||
@@ -292,7 +292,7 @@ COMPOSE_REGISTRY = {
|
||||
),
|
||||
"vllm/gemma-bf16": _entry(
|
||||
model="gemma-4-31b", weights_variant="autoround_int4", workload="long-ctx-single",
|
||||
engine="vllm-nightly-mtp", drafter="gemma-it-assistant", kv_format="bf16",
|
||||
engine="vllm-nightly-clean", drafter="gemma-it-assistant", kv_format="bf16",
|
||||
tp=2, max_ctx=200000, max_num_seqs=1, mem_util=0.95,
|
||||
compose_path="models/gemma-4-31b/vllm/compose/dual/bf16.yml",
|
||||
default_port=8033,
|
||||
@@ -304,5 +304,60 @@ COMPOSE_REGISTRY = {
|
||||
compose_path="models/gemma-4-31b/vllm/compose/dual/awq.yml",
|
||||
default_port=8033,
|
||||
),
|
||||
|
||||
# v0.7.3 MoE onboarding — Gemma 4 26B-A4B + Qwen 3.6 35B-A3B.
|
||||
# Both target the unconstrained-nightly engine (vllm-nightly-clean) which
|
||||
# rides nightly-bf610c2f (2026-05-15, post-PR-#42521). Gemma is the
|
||||
# shippable path; Qwen 35B-A3B is preview-only until Genesis v7.73.x
|
||||
# re-anchors on a post-#42521 nightly.
|
||||
"vllm/gemma-a4b-single": _entry(
|
||||
model="gemma-4-26b-a4b", weights_variant="autoround_int4_mixed", workload="fast-chat",
|
||||
engine="vllm-nightly-clean", drafter=None, kv_format="bf16",
|
||||
tp=1, max_ctx=8192, max_num_seqs=256, mem_util=0.92,
|
||||
compose_path="models/gemma-4-26b-a4b/vllm/compose/single/docker-compose.yml",
|
||||
default_port=8040,
|
||||
),
|
||||
"vllm/gemma-a4b": _entry(
|
||||
model="gemma-4-26b-a4b", weights_variant="autoround_int4_mixed", workload="fast-chat",
|
||||
engine="vllm-nightly-clean", drafter=None, kv_format="bf16",
|
||||
tp=2, max_ctx=32768, max_num_seqs=256, mem_util=0.92,
|
||||
compose_path="models/gemma-4-26b-a4b/vllm/compose/dual/docker-compose.yml",
|
||||
default_port=8041,
|
||||
),
|
||||
"vllm/gemma-a4b-awq": _entry(
|
||||
model="gemma-4-26b-a4b", weights_variant="awq_compressed_tensors", workload="fast-chat",
|
||||
engine="vllm-nightly-clean", drafter=None, kv_format="bf16",
|
||||
tp=2, max_ctx=32768, max_num_seqs=256, mem_util=0.92,
|
||||
compose_path="models/gemma-4-26b-a4b/vllm/compose/dual/awq.yml",
|
||||
default_port=8042,
|
||||
),
|
||||
"vllm/gemma-a4b-awq-mtp": _entry(
|
||||
model="gemma-4-26b-a4b", weights_variant="awq_compressed_tensors", workload="fast-chat",
|
||||
engine="vllm-nightly-clean", drafter="gemma-26b-it-assistant", kv_format="bf16",
|
||||
tp=2, max_ctx=32768, max_num_seqs=256, mem_util=0.92,
|
||||
compose_path="models/gemma-4-26b-a4b/vllm/compose/dual/awq-mtp.yml",
|
||||
default_port=8043,
|
||||
),
|
||||
"vllm/qwen-a3b-preview-single": _entry(
|
||||
model="qwen3.6-35b-a3b", weights_variant="autoround_int4", workload="fast-chat",
|
||||
engine="vllm-nightly-clean", drafter=None, kv_format="fp8_e5m2",
|
||||
tp=1, max_ctx=8192, max_num_seqs=1, mem_util=0.92,
|
||||
compose_path="models/qwen3.6-35b-a3b/vllm/compose/single/preview.yml",
|
||||
default_port=8050,
|
||||
),
|
||||
"vllm/qwen-a3b-preview": _entry(
|
||||
model="qwen3.6-35b-a3b", weights_variant="autoround_int4", workload="fast-chat",
|
||||
engine="vllm-nightly-clean", drafter=None, kv_format="fp8_e5m2",
|
||||
tp=2, max_ctx=16384, max_num_seqs=1, mem_util=0.92,
|
||||
compose_path="models/qwen3.6-35b-a3b/vllm/compose/dual/preview.yml",
|
||||
default_port=8051,
|
||||
),
|
||||
"vllm/qwen-a3b-preview-mtp": _entry(
|
||||
model="qwen3.6-35b-a3b", weights_variant="autoround_int4", workload="fast-chat",
|
||||
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||||
tp=2, max_ctx=16384, max_num_seqs=1, mem_util=0.92,
|
||||
compose_path="models/qwen3.6-35b-a3b/vllm/compose/dual/preview-mtp.yml",
|
||||
default_port=8052,
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,14 @@
|
||||
schema_version: 1
|
||||
id: gemma-26b-it-assistant
|
||||
display_name: Google Gemma 4 26B-A4B MTP assistant
|
||||
spec_method: mtp_assistant
|
||||
model_compat:
|
||||
- gemma-4-26b-a4b
|
||||
n_default: 4
|
||||
n_max: 4
|
||||
download:
|
||||
hf_repo: google/gemma-4-26B-A4B-it-assistant
|
||||
size_gb: 0.97
|
||||
format: fp16
|
||||
vram_footprint_gb: 0.97
|
||||
status: production
|
||||
@@ -1,9 +1,10 @@
|
||||
schema_version: 1
|
||||
id: qwen-mtp-builtin
|
||||
display_name: Qwen 3.6 27B built-in MTP head
|
||||
display_name: Qwen 3.6 built-in MTP head
|
||||
spec_method: mtp
|
||||
model_compat:
|
||||
- qwen3.6-27b
|
||||
- qwen3.6-35b-a3b
|
||||
n_default: 3
|
||||
n_max: 3
|
||||
download: false
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
schema_version: 1
|
||||
id: vllm-nightly-clean
|
||||
display_name: vLLM nightly (no Genesis)
|
||||
type: vllm
|
||||
stability: nightly
|
||||
install:
|
||||
method: docker_image
|
||||
spec: vllm/vllm-openai:nightly-bf610c2f56764e1b30bc6065f4ceace3d6e59036
|
||||
min_sm: 7.5
|
||||
supported_model_families:
|
||||
- dense
|
||||
- gemma4-swa-dense
|
||||
- gemma4-swa-moe
|
||||
- qwen3-next-hybrid
|
||||
- qwen3-next-moe
|
||||
features:
|
||||
int8_per_token_head: false
|
||||
turboquant_3bit_nc: false
|
||||
marlin_pad_sub_tile_n: false
|
||||
qwen3_coder_tool_parser: false
|
||||
supported_kv_formats:
|
||||
- bf16
|
||||
- fp16
|
||||
- fp8_e5m2
|
||||
- fp8_e4m3
|
||||
- q4_0
|
||||
- k8v4
|
||||
supported_drafters:
|
||||
- eagle3
|
||||
- mtp
|
||||
- mtp_assistant
|
||||
supported_weight_formats:
|
||||
- bf16
|
||||
- fp16
|
||||
- autoround
|
||||
- awq
|
||||
- gptq
|
||||
- compressed-tensors
|
||||
required_overlays: []
|
||||
vendored_overlays: []
|
||||
required_genesis: false
|
||||
notes: "Unconstrained-nightly path: free to bump to whatever's current on Docker Hub, since no Genesis patches are anchored here. Per the TQ3-only Genesis policy (Genesis is strictly required only for turboquant_3bit_nc KV), this engine accepts every supported model family — non-TQ3 composes for Qwen 3.6-27B (qwen3-next-hybrid) and Gemma 4-31B (gemma4-swa-dense) route here, alongside the new MoE additions (qwen3-next-moe, gemma4-swa-moe). For Qwen3-Next workloads at long context (>21-26K), Genesis-anchored vllm-nightly-mtp is *recommended* for Cliff 2 mitigations but not strictly required to boot. Excludes turboquant_3bit_nc KV (Genesis-only feature) — TQ3 composes must route to vllm-nightly-mtp. Launch exports VLLM_NIGHTLY_SHA from this spec; VLLM_IMAGE override still works."
|
||||
@@ -5,12 +5,14 @@ type: vllm
|
||||
stability: nightly
|
||||
install:
|
||||
method: docker_image
|
||||
spec: vllm/vllm-openai:nightly-1acd67a795ebccdf9b9db7697ae9082058301657
|
||||
spec: vllm/vllm-openai:nightly-01d4d1ad375dc5854779c593eee093bcebb0cada
|
||||
min_sm: 7.5
|
||||
supported_model_families:
|
||||
- dense
|
||||
- qwen3-next-hybrid
|
||||
- qwen3-next-moe
|
||||
- gemma4-swa-dense
|
||||
- gemma4-swa-moe
|
||||
features:
|
||||
int8_per_token_head: false
|
||||
turboquant_3bit_nc: true
|
||||
@@ -40,4 +42,4 @@ required_overlays: []
|
||||
vendored_overlays: []
|
||||
required_genesis: true
|
||||
genesis_pin: v7.72.2
|
||||
notes: "Production nightly path for Qwen MTP and Gemma MTP without DFlash or INT8 PTH overlays. Launch exports this SHA as VLLM_NIGHTLY_SHA; VLLM_IMAGE can override the full image ref."
|
||||
notes: "Production nightly path for Qwen MTP. Pin held at nightly-01d4d1ad (2026-05-04, Sander's v7.72.2 PROD pin) — this is the SHA Genesis v7.72.2 patches were anchored against and all v7.72.2 BENCHMARKS rows reference. Bumping requires either a Genesis pin bump cycle (Sander v7.73.x) or local patch re-anchor + Qwen 27B/Gemma 31B rebench gate. supported_model_families lists qwen3-next-moe + gemma4-swa-moe so the schema is ready; the actual MoE production boot needs a Genesis re-anchor on a post-#42521 nightly. Until then qwen3-next-moe routes to the preview path on vllm-nightly-clean. Gemma 4 composes that don't need Genesis should route to vllm-nightly-clean directly. Launch exports this SHA as VLLM_NIGHTLY_SHA; VLLM_IMAGE can override the full image ref. PRIOR-BUG: commit 40f1ef78 (2026-05-14) accidentally set this to nightly-1acd67a7 (the post-Gemma4-merge Gemma SHA); fixed back to 01d4d1ad."
|
||||
|
||||
@@ -8,6 +8,7 @@ existing compose registry and validate_estate() profile checks.
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import concurrent.futures
|
||||
import os
|
||||
import re
|
||||
import shlex
|
||||
@@ -43,6 +44,8 @@ from scripts.lib.profiles.launch_compat import _hardware_id_from_gpu, resolve_en
|
||||
|
||||
SUPPORTED_ESTATE_SCHEMA_VERSIONS = {1}
|
||||
DEFAULT_ESTATE_PATH = Path("~/.club3090/estate.yml").expanduser()
|
||||
DEFAULT_BOOT_LOG_DIR = Path("/tmp/club3090-estate-boot")
|
||||
BOOT_LOG_KEEP = 5
|
||||
|
||||
|
||||
class EstateCliError(Exception):
|
||||
@@ -58,6 +61,17 @@ class GpuInfo:
|
||||
hardware_id: str
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BootOutcome:
|
||||
index: int
|
||||
total: int
|
||||
inst: InstanceSpec
|
||||
ok: bool
|
||||
elapsed_s: int
|
||||
log_path: Path
|
||||
error: str = ""
|
||||
|
||||
|
||||
def utc_now() -> str:
|
||||
return datetime.now(timezone.utc).replace(microsecond=0).isoformat().replace("+00:00", "Z")
|
||||
|
||||
@@ -353,11 +367,54 @@ def compose_cmd() -> list[str]:
|
||||
return shlex.split(os.environ.get("COMPOSE_BIN", "docker compose"))
|
||||
|
||||
|
||||
def run_compose(inst: InstanceSpec, action: str) -> None:
|
||||
def boot_log_dir() -> Path:
|
||||
return Path(os.environ.get("CLUB3090_ESTATE_BOOT_LOG_DIR", str(DEFAULT_BOOT_LOG_DIR))).expanduser()
|
||||
|
||||
|
||||
def instance_log_path(inst: InstanceSpec) -> Path:
|
||||
return boot_log_dir() / f"{safe_name(inst.name)}.log"
|
||||
|
||||
|
||||
def rotate_log(path: Path, keep: int = BOOT_LOG_KEEP) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
for i in range(keep - 1, 0, -1):
|
||||
src = path.with_name(f"{path.name}.{i}")
|
||||
dst = path.with_name(f"{path.name}.{i + 1}")
|
||||
if src.exists():
|
||||
src.replace(dst)
|
||||
if path.exists():
|
||||
path.replace(path.with_name(f"{path.name}.1"))
|
||||
|
||||
|
||||
def prepare_instance_log(inst: InstanceSpec) -> Path:
|
||||
path = instance_log_path(inst)
|
||||
rotate_log(path)
|
||||
path.touch(mode=0o600, exist_ok=True)
|
||||
return path
|
||||
|
||||
|
||||
def summarize_error(error: str, limit: int = 220) -> str:
|
||||
line = " ".join(part.strip() for part in str(error).splitlines() if part.strip())
|
||||
if not line:
|
||||
line = "unknown error"
|
||||
return line if len(line) <= limit else line[: limit - 1] + "…"
|
||||
|
||||
|
||||
def append_log(path: Path, message: str) -> None:
|
||||
with path.open("a", encoding="utf-8") as fh:
|
||||
fh.write(message.rstrip() + "\n")
|
||||
|
||||
|
||||
def run_compose(inst: InstanceSpec, action: str, log_path: Path | None = None) -> None:
|
||||
cmd = compose_cmd() + ["-p", project_name(inst.name), "-f", str(compose_abs_path(inst.compose_name)), action]
|
||||
if action == "up":
|
||||
cmd.append("-d")
|
||||
proc = subprocess.run(cmd, cwd=REPO_ROOT, env=compose_env(inst), text=True)
|
||||
if log_path is not None:
|
||||
append_log(log_path, f"$ {' '.join(cmd)}")
|
||||
with log_path.open("a", encoding="utf-8") as fh:
|
||||
proc = subprocess.run(cmd, cwd=REPO_ROOT, env=compose_env(inst), text=True, stdout=fh, stderr=subprocess.STDOUT)
|
||||
else:
|
||||
proc = subprocess.run(cmd, cwd=REPO_ROOT, env=compose_env(inst), text=True)
|
||||
if proc.returncode != 0:
|
||||
raise EstateCliError(f"`{' '.join(cmd)}` failed with exit {proc.returncode}")
|
||||
|
||||
@@ -406,6 +463,21 @@ def wait_ready(inst: InstanceSpec, timeout: int) -> None:
|
||||
time.sleep(4)
|
||||
|
||||
|
||||
def wait_ready_quiet(inst: InstanceSpec, timeout: int) -> int:
|
||||
start = time.monotonic()
|
||||
poll_interval = max(float(os.environ.get("CLUB3090_ESTATE_POLL_INTERVAL", "4")), 0.1)
|
||||
while True:
|
||||
if endpoint_ready(inst.port):
|
||||
return int(time.monotonic() - start)
|
||||
if not container_running(inst.name):
|
||||
logs = docker_logs_tail(inst.name)
|
||||
raise EstateCliError(f"container {container_name(inst.name)} stopped during boot\n{logs}")
|
||||
elapsed = time.monotonic() - start
|
||||
if elapsed >= timeout:
|
||||
raise EstateCliError(f"timeout waiting for {inst.name} after {timeout}s; logs: docker logs {container_name(inst.name)}")
|
||||
time.sleep(min(poll_interval, max(timeout - elapsed, 0.1)))
|
||||
|
||||
|
||||
def select_instances(instances: list[InstanceSpec], only: set[str] | None) -> list[InstanceSpec]:
|
||||
if not only:
|
||||
return instances
|
||||
@@ -430,6 +502,67 @@ def command_validate(args: argparse.Namespace) -> int:
|
||||
return 0 if result.valid else 1
|
||||
|
||||
|
||||
def effective_parallel_jobs(requested: int | None, total: int) -> int:
|
||||
if requested is None:
|
||||
return min(total, 4)
|
||||
if requested < 1:
|
||||
raise EstateCliError("--parallel-jobs must be >= 1")
|
||||
return min(requested, total, 4)
|
||||
|
||||
|
||||
def boot_instance_parallel(index: int, total: int, inst: InstanceSpec, timeout: int, log_path: Path) -> BootOutcome:
|
||||
start = time.monotonic()
|
||||
try:
|
||||
append_log(log_path, f"[estate] booting {inst.name}: {inst.compose_name} GPUs={list(inst.gpu_indices)} port={inst.port}")
|
||||
run_compose(inst, "up", log_path=log_path)
|
||||
elapsed_s = wait_ready_quiet(inst, timeout)
|
||||
append_log(log_path, f"[estate] healthy after {elapsed_s}s")
|
||||
return BootOutcome(index=index, total=total, inst=inst, ok=True, elapsed_s=elapsed_s, log_path=log_path)
|
||||
except Exception as exc:
|
||||
elapsed_s = int(time.monotonic() - start)
|
||||
summary = summarize_error(str(exc))
|
||||
append_log(log_path, f"[estate] ERROR after {elapsed_s}s: {summary}")
|
||||
return BootOutcome(index=index, total=total, inst=inst, ok=False, elapsed_s=elapsed_s, log_path=log_path, error=summary)
|
||||
|
||||
|
||||
def boot_instances_parallel(selected: list[InstanceSpec], timeout: int, requested_jobs: int | None, stagger_s: float) -> int:
|
||||
if stagger_s < 0:
|
||||
raise EstateCliError("--parallel-stagger must be >= 0")
|
||||
total = len(selected)
|
||||
jobs = effective_parallel_jobs(requested_jobs, total)
|
||||
print(f"[estate] parallel boot: {total} instance(s), jobs={jobs}, stagger={stagger_s:g}s")
|
||||
|
||||
futures: dict[concurrent.futures.Future[BootOutcome], tuple[int, InstanceSpec, Path]] = {}
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=jobs) as pool:
|
||||
for i, inst in enumerate(selected, start=1):
|
||||
log_path = prepare_instance_log(inst)
|
||||
print(f"[estate] [{i}/{total}] booting {inst.name} on GPUs {','.join(str(g) for g in inst.gpu_indices)} port {inst.port}... (started)")
|
||||
future = pool.submit(boot_instance_parallel, i, total, inst, timeout, log_path)
|
||||
futures[future] = (i, inst, log_path)
|
||||
if i < total and stagger_s > 0:
|
||||
time.sleep(stagger_s)
|
||||
|
||||
outcomes = [future.result() for future in concurrent.futures.as_completed(futures)]
|
||||
|
||||
outcomes.sort(key=lambda outcome: outcome.index)
|
||||
healthy = 0
|
||||
failed: list[BootOutcome] = []
|
||||
for outcome in outcomes:
|
||||
if outcome.ok:
|
||||
healthy += 1
|
||||
print(f"[estate] [{outcome.index}/{outcome.total}] {outcome.inst.name} ✓ healthy after {outcome.elapsed_s}s")
|
||||
else:
|
||||
failed.append(outcome)
|
||||
print(
|
||||
f"[estate] [{outcome.index}/{outcome.total}] {outcome.inst.name} ✗ failed after {outcome.elapsed_s}s: {outcome.error}"
|
||||
)
|
||||
|
||||
print(f"[estate] Summary: {healthy}/{total} healthy, {len(failed)} failed.")
|
||||
for outcome in failed:
|
||||
print(f"[estate] Failed instance: {outcome.inst.name}. See {outcome.log_path}")
|
||||
return 0 if not failed else 1
|
||||
|
||||
|
||||
def command_boot(args: argparse.Namespace) -> int:
|
||||
path = estate_path(args.file)
|
||||
try:
|
||||
@@ -439,6 +572,13 @@ def command_boot(args: argparse.Namespace) -> int:
|
||||
return 1
|
||||
selected = select_instances(instances, parse_only(args.only))
|
||||
persist_default_estate_source(path, data, instances, gpus, nvlink_active)
|
||||
if getattr(args, "parallel", False) and len(selected) > 1:
|
||||
return boot_instances_parallel(
|
||||
selected,
|
||||
args.timeout,
|
||||
getattr(args, "parallel_jobs", None),
|
||||
getattr(args, "parallel_stagger", 15.0),
|
||||
)
|
||||
total = len(selected)
|
||||
for i, inst in enumerate(selected, start=1):
|
||||
print(f"[estate] [{i}/{total}] booting {inst.name}: {inst.compose_name} GPUs={list(inst.gpu_indices)} port={inst.port}")
|
||||
@@ -699,6 +839,9 @@ def build_parser() -> argparse.ArgumentParser:
|
||||
boot.add_argument("--file", default=str(DEFAULT_ESTATE_PATH))
|
||||
boot.add_argument("--only", default="")
|
||||
boot.add_argument("--timeout", type=int, default=int(os.environ.get("READY_TIMEOUT", "600")))
|
||||
boot.add_argument("--parallel", action="store_true")
|
||||
boot.add_argument("--parallel-jobs", type=int, default=None)
|
||||
boot.add_argument("--parallel-stagger", type=float, default=15.0)
|
||||
boot.set_defaults(func=command_boot)
|
||||
|
||||
down = sub.add_parser("down")
|
||||
|
||||
@@ -19,7 +19,16 @@ if str(REPO_ROOT) not in sys.path:
|
||||
|
||||
os.environ.setdefault("CLUB3090_LOG_LEVEL", "ERROR")
|
||||
|
||||
from scripts.lib.profiles.compat import FitsResult, ProfileError, fits, load_profiles, to_compose_name # noqa: E402
|
||||
from scripts.lib.profiles.compat import ( # noqa: E402
|
||||
TOPOLOGY_ADVISORY,
|
||||
FitsResult,
|
||||
ProfileError,
|
||||
TopologyClass,
|
||||
classify_hardware_topology,
|
||||
fits,
|
||||
load_profiles,
|
||||
to_compose_name,
|
||||
)
|
||||
from scripts.lib.profiles.compose_registry import COMPOSE_REGISTRY # noqa: E402
|
||||
|
||||
|
||||
@@ -103,6 +112,26 @@ def _parse_gpu_specs(value: str, profiles) -> list:
|
||||
return hardware
|
||||
|
||||
|
||||
def _parse_gpu_specs_with_indices(value: str, profiles) -> list[tuple[str, object]]:
|
||||
hardware = []
|
||||
for raw in value.split(";"):
|
||||
raw = raw.strip()
|
||||
if not raw:
|
||||
continue
|
||||
try:
|
||||
idx, name, mem_mib, sm = raw.split("|", 3)
|
||||
except ValueError as exc:
|
||||
raise LaunchCompatError(f"invalid --gpu-spec entry `{raw}`") from exc
|
||||
hardware_id = _hardware_id_from_gpu(name, int(mem_mib), float(sm))
|
||||
try:
|
||||
hardware.append((idx, profiles.hardware[hardware_id]))
|
||||
except KeyError as exc:
|
||||
raise LaunchCompatError(f"hardware profile `{hardware_id}` is not installed") from exc
|
||||
if not hardware:
|
||||
raise LaunchCompatError("no GPU specs were provided for topology classification")
|
||||
return hardware
|
||||
|
||||
|
||||
def _engine_family(engine_type: str) -> str:
|
||||
return "llamacpp" if engine_type == "llama.cpp" else engine_type
|
||||
|
||||
@@ -364,6 +393,90 @@ def command_resolve_variant_pin(args: argparse.Namespace) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
def _hardware_line(index: str, hardware) -> str:
|
||||
return f" GPU {index}: {hardware.display_name} ({hardware.vram_gb:g} GB, sm {hardware.sm:g})"
|
||||
|
||||
|
||||
def _standalone_recommendation(topology: TopologyClass, count: int) -> list[str]:
|
||||
if topology == TopologyClass.SINGLE_CARD:
|
||||
return [
|
||||
"Recommended:",
|
||||
" 1. Use the largest single-card compose your model fits.",
|
||||
" 2. Add another matched card for TP=2 when long-context concurrency matters.",
|
||||
]
|
||||
if topology == TopologyClass.HOMOGENEOUS:
|
||||
return [
|
||||
"Recommended:",
|
||||
f" 1. TP={count} is the default path for matched cards; use the shipped vllm/dual* or multi-card composes.",
|
||||
" 2. Estate planner remains useful when you want separate models/endpoints instead of one larger TP instance.",
|
||||
]
|
||||
if topology == TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED:
|
||||
return [
|
||||
"Recommended:",
|
||||
f" 1. TP={count} works as-is. Compute mismatch means the faster card waits at every NCCL allreduce; effective throughput caps at the slower card's speed (~30% of faster card idle). Full per-card VRAM capacity preserved.",
|
||||
" 2. Estate planner — `bash scripts/launch.sh --estate` runs different models per card, each at full speed.",
|
||||
"",
|
||||
"Not recommended:",
|
||||
" - PP=N: possible as a manual flag flip (`--pipeline-parallel-size N`) on a vllm/dual compose, but no PP compose ships today.",
|
||||
]
|
||||
if topology == TopologyClass.VRAM_MISMATCHED:
|
||||
return [
|
||||
"Recommended:",
|
||||
" 1. llama.cpp `--tensor-split` for weighted layer split on mismatched VRAM.",
|
||||
" 2. PP=N as a manual vLLM flag flip (`--pipeline-parallel-size N`) if you are deliberately experimenting.",
|
||||
" 3. Estate planner — run different models per card or use the largest matched subset.",
|
||||
"",
|
||||
"Not recommended:",
|
||||
" - TP=N on the full mismatched set: the smaller card caps usable model size and KV headroom.",
|
||||
]
|
||||
return [
|
||||
"Recommended:",
|
||||
" 1. Manual selection. Use the largest matched subset for one model.",
|
||||
" 2. Estate planner — put different models on different card subsets.",
|
||||
]
|
||||
|
||||
|
||||
def command_topology(args: argparse.Namespace) -> int:
|
||||
_quiet_compat_logger()
|
||||
profiles = load_profiles()
|
||||
indexed_hardware = _parse_gpu_specs_with_indices(args.gpu_spec, profiles)
|
||||
hardware = [item[1] for item in indexed_hardware]
|
||||
topology = classify_hardware_topology(hardware)
|
||||
advisory = TOPOLOGY_ADVISORY.get(topology)
|
||||
|
||||
if args.format == "wizard":
|
||||
if topology in (TopologyClass.SINGLE_CARD, TopologyClass.HOMOGENEOUS):
|
||||
return 0
|
||||
detected = " + ".join(
|
||||
f"1x {hw.display_name} ({hw.vram_gb:g} GB, sm {hw.sm:g})"
|
||||
for _idx, hw in indexed_hardware
|
||||
)
|
||||
print(f"Detected: {detected}")
|
||||
print("")
|
||||
print(f"Topology: {topology.value}")
|
||||
if advisory:
|
||||
print(f" {advisory}")
|
||||
print("")
|
||||
print("Continue with the selected parallelism if that trade-off is acceptable.")
|
||||
return 0
|
||||
|
||||
print("Detected hardware:")
|
||||
for idx, hw in indexed_hardware:
|
||||
print(_hardware_line(idx, hw))
|
||||
print("")
|
||||
print(f"Topology class: {topology.value}")
|
||||
print("")
|
||||
for line in _standalone_recommendation(topology, len(hardware)):
|
||||
print(line)
|
||||
print("")
|
||||
if advisory:
|
||||
print("Advisory:")
|
||||
print(f" {advisory}")
|
||||
print("")
|
||||
print("For details, see docs/MULTI_CARD.md.")
|
||||
return 0
|
||||
|
||||
|
||||
def build_parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(description="Profile bridge for scripts/launch.sh")
|
||||
sub = parser.add_subparsers(dest="command", required=True)
|
||||
@@ -404,6 +517,11 @@ def build_parser() -> argparse.ArgumentParser:
|
||||
variant_pin.add_argument("--format", choices=("shell", "json", "value"), default="shell")
|
||||
variant_pin.set_defaults(func=command_resolve_variant_pin)
|
||||
|
||||
topology = sub.add_parser("topology")
|
||||
topology.add_argument("--gpu-spec", required=True)
|
||||
topology.add_argument("--format", choices=("standalone", "wizard"), default="standalone")
|
||||
topology.set_defaults(func=command_topology)
|
||||
|
||||
return parser
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
schema_version: 1
|
||||
id: gemma-4-26b-a4b
|
||||
display_name: Gemma 4 26B-A4B (MoE)
|
||||
family: gemma4-swa-moe
|
||||
hidden_size: 2816
|
||||
intermediate_size: 2112
|
||||
num_hidden_layers: 30
|
||||
# Hybrid SWA pattern: 5 full_attention layers at indices [5, 11, 17, 23, 29]
|
||||
# (every 6th layer, with last layer always global per Gemma 4 family convention)
|
||||
num_full_attn_layers: 5
|
||||
num_sliding_attn_layers: 25
|
||||
num_attn_heads: 16
|
||||
num_kv_heads: 8 # sliding-layer KV head count
|
||||
num_global_kv_heads: 2 # NEW: global-layer KV head count (asymmetric — distinct from sliding)
|
||||
head_dim_sliding: 256
|
||||
global_head_dim: 512
|
||||
sliding_window: 1024
|
||||
max_ctx_supported: 262144
|
||||
attention_k_eq_v: true # Gemma 4 family default — K and V share storage
|
||||
# MoE (from config.json text_config)
|
||||
num_experts: 128
|
||||
num_experts_per_tok: 8 # config field: top_k_experts
|
||||
moe_intermediate_size: 704
|
||||
active_params_b: 4.0
|
||||
mtp_num_hidden_layers: null # MTP drafter is external (gemma-4-26B-A4B-it-assistant), not built-in
|
||||
vision_capable: true # multimodal — vision_config + audio_config tokens present
|
||||
weights:
|
||||
autoround_int4_mixed:
|
||||
path: gemma-4-26b-a4b-autoround-int4-mixed
|
||||
size_gb: 16.0 # measured post-download 2026-05-15
|
||||
format: autoround
|
||||
status: ampere-blocked # moe_intermediate_size=704 % group_size=128 = 5.5
|
||||
# → Marlin K-dim alignment fails on SM86.
|
||||
# Boots fine on SM90+ (Cutlass W4A8 / Machete).
|
||||
awq_compressed_tensors:
|
||||
path: gemma-4-26b-a4b-awq-4bit
|
||||
size_gb: 17.0 # measured post-download 2026-05-15
|
||||
format: compressed-tensors # AWQ pack-quantized via vLLM compressed-tensors loader
|
||||
status: production # via vLLM PR #40886 overlay (Ampere-bootable)
|
||||
default_weight_variant: awq_compressed_tensors
|
||||
compatible_drafters:
|
||||
- gemma-26b-it-assistant # dedicated 26B-A4B MTP assistant (separate weights from 31B)
|
||||
# Sliding layers cap TP at num_kv_heads=8 → [1,2,4,8]
|
||||
# Global layers cap TP at num_global_kv_heads=2 → [1,2]
|
||||
# Effective valid_tp = intersection = [1, 2]
|
||||
valid_tp:
|
||||
- 1
|
||||
- 2
|
||||
requires_genesis: false # Gemma 4 family doesn't need Genesis (no DeltaNet quirks)
|
||||
# kv-calc.py models MoE + asymmetric KV heads (8 sliding / 2 global) as of
|
||||
# v0.7.3. Calibration is low-anchor (AWQ no-MTP + AWQ-MTP rows only); add
|
||||
# longer context / lower max_num_seqs rows before treating projections as
|
||||
# production-grade.
|
||||
kv_calc_supported: true
|
||||
@@ -51,5 +51,12 @@ valid_tp:
|
||||
- 1
|
||||
- 2
|
||||
- 4
|
||||
requires_genesis: true
|
||||
# Strictly bootable on any vLLM nightly that supports the qwen3-next-hybrid
|
||||
# family. Genesis is REQUIRED for TQ3 KV (turboquant_3bit_nc — Genesis-only
|
||||
# feature). For non-TQ3 KV formats (fp8, bf16, q4_0, k8v4) Genesis is
|
||||
# recommended but not strictly needed: it adds Cliff 2 mitigations
|
||||
# (PN12/PN25/PN34) for long-context stability past ~21-26K. Composes encode
|
||||
# the engine choice via their `Engine-profile:` header — non-TQ3 composes
|
||||
# route to vllm-nightly-clean, TQ3 composes route to vllm-nightly-mtp.
|
||||
requires_genesis: false
|
||||
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
schema_version: 1
|
||||
id: qwen3.6-35b-a3b
|
||||
display_name: Qwen 3.6 35B-A3B (MoE)
|
||||
family: qwen3-next-moe
|
||||
hidden_size: 2048
|
||||
num_hidden_layers: 40
|
||||
# Hybrid: 10 full_attention layers at indices 3,7,11,15,19,23,27,31,35,39
|
||||
# (one per 4-layer block, matching full_attention_interval=4 in config.json)
|
||||
num_gdn_layers: 30
|
||||
num_attn_layers: 10
|
||||
num_attn_heads: 16
|
||||
num_kv_heads: 2
|
||||
head_dim_attn: 256
|
||||
linear_num_v_heads: 32
|
||||
linear_num_k_heads: 16
|
||||
linear_v_head_dim: 128
|
||||
linear_k_head_dim: 128
|
||||
linear_conv_kernel_dim: 4
|
||||
max_ctx_supported: 262144
|
||||
attention_k_eq_v: false
|
||||
# MoE (from config.json text_config)
|
||||
num_experts: 256
|
||||
num_experts_per_tok: 8
|
||||
moe_intermediate_size: 512
|
||||
shared_expert_intermediate_size: 512
|
||||
active_params_b: 3.0
|
||||
# Built-in MTP drafter (1 dedicated MTP head)
|
||||
mtp_num_hidden_layers: 1
|
||||
attn_output_gate: true
|
||||
vision_capable: true
|
||||
weights:
|
||||
autoround_int4:
|
||||
path: qwen3.6-35b-a3b-autoround-int4
|
||||
size_gb: 20.0
|
||||
format: autoround
|
||||
status: production
|
||||
gptq_int4:
|
||||
path: qwen3.6-35b-a3b-gptq-int4
|
||||
size_gb: 22.0
|
||||
format: gptq
|
||||
status: experimental
|
||||
gguf:
|
||||
path: qwen3.6-35b-a3b-gguf
|
||||
size_gb: 90.0
|
||||
format: gguf
|
||||
status: production
|
||||
dflash:
|
||||
path: qwen3.6-35b-a3b-dflash
|
||||
size_gb: variable
|
||||
format: autoround
|
||||
status: experimental
|
||||
dflash_gguf:
|
||||
path: qwen3.6-35b-a3b-dflash-gguf
|
||||
size_gb: variable
|
||||
format: gguf
|
||||
status: experimental
|
||||
default_weight_variant: autoround_int4
|
||||
compatible_drafters:
|
||||
- qwen-mtp-builtin
|
||||
# num_kv_heads=2 caps TP at 2 (each rank needs at least 1 KV head)
|
||||
valid_tp:
|
||||
- 1
|
||||
- 2
|
||||
# Strictly bootable on upstream vLLM (PR #42521 onwards). Genesis is
|
||||
# RECOMMENDED for production: it provides TQ3 KV (Genesis-only) plus the
|
||||
# Cliff 2 / DeltaNet stabilization patches that matter past ~21-26K ctx.
|
||||
# Preview/smoke composes target vllm-nightly-clean (no Genesis); production
|
||||
# composes (TBD — gated on Genesis v7.73.x re-anchor) target vllm-nightly-mtp.
|
||||
requires_genesis: false
|
||||
# kv-calc.py models the MoE + DeltaNet+attention hybrid as of v0.7.3.
|
||||
# Calibration is low-anchor (preview + preview-MTP rows only); add longer
|
||||
# context / max_num_seqs rows before treating projections as production-grade.
|
||||
kv_calc_supported: true
|
||||
+52
-19
@@ -140,23 +140,34 @@ def parse_vllm_boot(boot_log: str) -> dict:
|
||||
def parse_bench(log: str) -> dict:
|
||||
"""Extract narrative + code TPS summaries from bench.sh output."""
|
||||
out: dict[str, Any] = {}
|
||||
# narrative summary block
|
||||
for kind in ("narrative", "code"):
|
||||
m = re.search(
|
||||
for kind in ("narrative", "code", "prompt-processing"):
|
||||
block = re.search(
|
||||
rf"=== summary \[{kind}\] \(n=\d+\) ===\s*\n"
|
||||
rf"\s+wall_TPS\s+mean=\s*([\d.]+)\s+std=\s*([\d.]+)\s+CV=\s*([\d.]+)%.*\n"
|
||||
rf"\s+decode_TPS\s+mean=\s*([\d.]+)\s+std=\s*([\d.]+)\s+CV=\s*([\d.]+)%.*\n"
|
||||
rf"\s+TTFT\s+mean=\s*([\d.]+)ms",
|
||||
rf"(?P<body>.*?)(?=\n==========|\n=== GPU state ===|\n=== Last|\Z)",
|
||||
log,
|
||||
re.DOTALL,
|
||||
)
|
||||
if not block:
|
||||
continue
|
||||
body = block.group("body")
|
||||
parsed: dict[str, Any] = {}
|
||||
m = re.search(r"\s+wall_TPS\s+mean=\s*([\d.]+)\s+std=\s*([\d.]+)\s+CV=\s*([\d.]+)%", body)
|
||||
if m:
|
||||
out[kind] = {
|
||||
"wall_tps_mean": float(m.group(1)),
|
||||
"wall_tps_cv": float(m.group(3)),
|
||||
"decode_tps_mean": float(m.group(4)),
|
||||
"decode_tps_cv": float(m.group(6)),
|
||||
"ttft_ms_mean": float(m.group(7)),
|
||||
}
|
||||
parsed["wall_tps_mean"] = float(m.group(1))
|
||||
parsed["wall_tps_cv"] = float(m.group(3))
|
||||
m = re.search(r"\s+decode_TPS\s+mean=\s*([\d.]+)\s+std=\s*([\d.]+)\s+CV=\s*([\d.]+)%", body)
|
||||
if m:
|
||||
parsed["decode_tps_mean"] = float(m.group(1))
|
||||
parsed["decode_tps_cv"] = float(m.group(3))
|
||||
m = re.search(r"\s+TTFT\s+mean=\s*([\d.]+)ms", body)
|
||||
if m:
|
||||
parsed["ttft_ms_mean"] = float(m.group(1))
|
||||
m = re.search(r"\s+PP tok/s\s+mean=\s*([\d.]+)\s+std=\s*([\d.]+)\s+CV=\s*([\d.]+)%", body)
|
||||
if m:
|
||||
parsed["pp_tps_mean"] = float(m.group(1))
|
||||
parsed["pp_tps_cv"] = float(m.group(3))
|
||||
if parsed:
|
||||
out[kind.replace("-", "_")] = parsed
|
||||
# GPU state at end
|
||||
gpu_block = re.search(r"=== GPU state ===\s*\n((?:\d.+\n){1,4})", log)
|
||||
if gpu_block:
|
||||
@@ -396,15 +407,22 @@ def render(report: dict) -> str:
|
||||
lines.append("## Performance — `bench.sh`")
|
||||
lines.append("")
|
||||
if bench.get("narrative") or bench.get("code"):
|
||||
lines.append("| Bench | wall TPS | decode TPS | TTFT | CV (wall/decode) |")
|
||||
lines.append("|---|---:|---:|---:|---:|")
|
||||
lines.append("| Bench | wall TPS | decode TPS | PP tok/s | TTFT | CV (wall/decode) |")
|
||||
lines.append("|---|---:|---:|---:|---:|---:|")
|
||||
for kind in ("narrative", "code"):
|
||||
b = bench.get(kind)
|
||||
if b:
|
||||
pp = f"{b['pp_tps_mean']:.0f}" if b.get("pp_tps_mean") is not None else "n/a"
|
||||
lines.append(
|
||||
f"| {kind} | {b['wall_tps_mean']:.2f} | **{b['decode_tps_mean']:.2f}** | "
|
||||
f"| {kind} | {b['wall_tps_mean']:.2f} | **{b['decode_tps_mean']:.2f}** | {pp} | "
|
||||
f"{b['ttft_ms_mean']:.0f} ms | {b['wall_tps_cv']:.1f}% / {b['decode_tps_cv']:.1f}% |"
|
||||
)
|
||||
pp_fallback = bench.get("prompt_processing")
|
||||
if pp_fallback and pp_fallback.get("pp_tps_mean") is not None:
|
||||
lines.append(
|
||||
f"| prompt-processing fallback | — | — | **{pp_fallback['pp_tps_mean']:.0f}** | "
|
||||
f"{pp_fallback.get('ttft_ms_mean', 0):.0f} ms | — |"
|
||||
)
|
||||
if bench.get("mtp"):
|
||||
m = bench["mtp"]
|
||||
lines.append("")
|
||||
@@ -600,9 +618,14 @@ def render_discuss(report: dict) -> str:
|
||||
if bench.get("narrative"):
|
||||
b_n, b_c = bench["narrative"], bench.get("code", {})
|
||||
lines.append("**TPS:**")
|
||||
lines.append(f"- Narrative: {b_n['decode_tps_mean']:.1f} TPS decode, {b_n['ttft_ms_mean']:.0f} ms TTFT (CV {b_n['decode_tps_cv']:.1f}%)")
|
||||
pp_n = f", PP {b_n['pp_tps_mean']:.0f} tok/s" if b_n.get("pp_tps_mean") is not None else ""
|
||||
lines.append(f"- Narrative: {b_n['decode_tps_mean']:.1f} TPS decode, {b_n['ttft_ms_mean']:.0f} ms TTFT{pp_n} (CV {b_n['decode_tps_cv']:.1f}%)")
|
||||
if b_c:
|
||||
lines.append(f"- Code: {b_c['decode_tps_mean']:.1f} TPS decode, {b_c['ttft_ms_mean']:.0f} ms TTFT (CV {b_c['decode_tps_cv']:.1f}%)")
|
||||
pp_c = f", PP {b_c['pp_tps_mean']:.0f} tok/s" if b_c.get("pp_tps_mean") is not None else ""
|
||||
lines.append(f"- Code: {b_c['decode_tps_mean']:.1f} TPS decode, {b_c['ttft_ms_mean']:.0f} ms TTFT{pp_c} (CV {b_c['decode_tps_cv']:.1f}%)")
|
||||
if bench.get("prompt_processing", {}).get("pp_tps_mean") is not None:
|
||||
pp = bench["prompt_processing"]["pp_tps_mean"]
|
||||
lines.append(f"- Prompt-processing fallback: {pp:.0f} tok/s")
|
||||
lines.append("")
|
||||
boot = report.get("vllm_boot", {})
|
||||
if boot.get("kv_cache_tokens"):
|
||||
@@ -641,11 +664,21 @@ def compute_tldr(report: dict) -> list[str]:
|
||||
bullets = []
|
||||
bench = report.get("bench", {})
|
||||
if bench.get("narrative") and bench.get("code"):
|
||||
pp_vals = [
|
||||
b.get("pp_tps_mean")
|
||||
for b in (bench.get("narrative", {}), bench.get("code", {}))
|
||||
if b.get("pp_tps_mean") is not None
|
||||
]
|
||||
pp_text = f", PP {sum(pp_vals) / len(pp_vals):.0f} tok/s" if pp_vals else ""
|
||||
bullets.append(
|
||||
f"TPS narrative **{bench['narrative']['decode_tps_mean']:.1f}** / "
|
||||
f"code **{bench['code']['decode_tps_mean']:.1f}** "
|
||||
f"(TTFT {bench['narrative']['ttft_ms_mean']:.0f}/{bench['code']['ttft_ms_mean']:.0f} ms)."
|
||||
f"(TTFT {bench['narrative']['ttft_ms_mean']:.0f}/{bench['code']['ttft_ms_mean']:.0f} ms{pp_text})."
|
||||
)
|
||||
elif bench.get("prompt_processing"):
|
||||
pp = bench["prompt_processing"].get("pp_tps_mean")
|
||||
if pp is not None:
|
||||
bullets.append(f"Prompt processing fallback: **{pp:.0f} tok/s**.")
|
||||
boot = report.get("vllm_boot", {})
|
||||
if boot.get("kv_cache_tokens") and boot.get("max_concurrency"):
|
||||
bullets.append(
|
||||
|
||||
@@ -384,6 +384,40 @@ if [[ -x scripts/lib/profiles/estate_cli.py || -f scripts/lib/profiles/estate_cl
|
||||
python3 scripts/lib/profiles/estate_cli.py report-state 2>&1 | redact || true
|
||||
fi
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# KV math calibration
|
||||
# ---------------------------------------------------------------------------
|
||||
# When a user files a VRAM-OOM or context-ceiling bug, the maintainer's first
|
||||
# question is "does kv-calc still agree with measured reality?" — a calibration
|
||||
# failure means the projection model has drifted from the actual VRAM cost of a
|
||||
# compose, so any "predicted PASS" verdict can't be trusted. Surface the
|
||||
# verdict line + any FAIL rows here so a triage reply can immediately see
|
||||
# whether to trust kv-calc projections for this user's config.
|
||||
|
||||
if have python3 && [[ -f tools/kv-calc.py ]]; then
|
||||
section "KV math calibration"
|
||||
calib_output=$(python3 tools/kv-calc.py --calibration 2>&1 || true)
|
||||
overall=$(echo "$calib_output" | grep -E '^Overall:' | head -1)
|
||||
fail_rows=$(echo "$calib_output" | grep -E '\bFAIL\b' || true)
|
||||
{
|
||||
if [[ -n "$overall" ]]; then
|
||||
echo "- ${overall}"
|
||||
else
|
||||
echo "- _kv-calc --calibration produced no Overall line; see output below._"
|
||||
fi
|
||||
if [[ -n "$fail_rows" ]]; then
|
||||
echo "- ⚠ Failing rows:"
|
||||
echo '```'
|
||||
echo "$fail_rows"
|
||||
echo '```'
|
||||
echo "- Math model is mis-calibrated against measured reality for the rows above. Any kv-calc projection on this checkout should be treated as suspect until the calibration anchors / formulas are reconciled."
|
||||
else
|
||||
echo "- No FAIL rows. kv-calc projections should agree with measured VRAM within the ±1.5 GB error band."
|
||||
fi
|
||||
} | redact
|
||||
echo "$calib_output" | redact | details "Full kv-calc --calibration output"
|
||||
fi
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Active container
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -828,6 +828,11 @@ def cmd_summary(turn_log, summary_path, boot_vram, growth_limit, timed_out, expe
|
||||
print(f"[soak] {label}:")
|
||||
for item in items:
|
||||
print(f"[soak] - {item}")
|
||||
if verdict == "PASS":
|
||||
print("[soak] note PASS = no failure signal on this sample;")
|
||||
print("[soak] not patch validation (topology alone can")
|
||||
print("[soak] sidestep what overlays target). See")
|
||||
print("[soak] scripts/soak-test.sh --help and docs/CLIFFS.md.")
|
||||
sys.exit(exit_code)
|
||||
|
||||
|
||||
|
||||
+35
-3
@@ -12,6 +12,27 @@
|
||||
# retention across sessions.
|
||||
# - Read-only against the running deployment.
|
||||
#
|
||||
# PASS verdict semantics:
|
||||
# PASS = no failure signal fired on the test sample. Specifically:
|
||||
# - silent_empty turns: 0 (no HTTP 200 + 0 completion tokens)
|
||||
# - max VRAM growth: under SOAK_MAX_GROWTH_MIB (default 200 MiB)
|
||||
# - TPS retention: first-5 vs last-5 median >= 98%
|
||||
# - request errors / stream interruptions: 0
|
||||
# PASS does NOT mean:
|
||||
# - "Patches in this compose's overlay set are doing useful work."
|
||||
# PASS-on-patched is consistent with patches working OR with patches
|
||||
# not being load-bearing for this workload + topology. Cliff 2 / 2b
|
||||
# mitigations target single-card 24 GB pressure; TP=2 (dual.yml)
|
||||
# structurally escapes Cliff 2 regardless of which patches load.
|
||||
# - "Deeper-context workloads will also pass." Continuous mode ramps
|
||||
# to ~22-25K accumulated tokens by turn 5; it does not push to
|
||||
# model max_ctx. Longer-context regimes can still fail.
|
||||
# - "The configuration is optimally tuned." Soak detects failures,
|
||||
# not whether perf is on the table.
|
||||
# For patch attribution, run the same soak on the same compose with
|
||||
# the overlay bind-mounts stripped (or on a baseline image) and compare
|
||||
# metrics. See https://github.com/noonghunna/club-3090/issues/140.
|
||||
#
|
||||
# Time budget:
|
||||
# Default SOAK_SESSIONS=20 x SOAK_TURNS=5, capped by SOAK_TIMEOUT_S=1800.
|
||||
# Expect 10-30 minutes depending on config.
|
||||
@@ -101,9 +122,20 @@ EXAMPLES
|
||||
CONTAINER=none ENDPOINT=http://localhost:8030 bash scripts/soak-test.sh
|
||||
|
||||
NOTES
|
||||
Soak-continuous is the only test that catches Cliff 2b. If you're filing a
|
||||
bench contribution, run with --continuous and paste the [soak] summary
|
||||
alongside your bench numbers. See docs/CLIFFS.md for context.
|
||||
Soak-continuous is the only test that surfaces Cliff 2b under
|
||||
multi-turn accumulating-context traffic on single-card configs.
|
||||
If you're filing a bench contribution, run with --continuous and
|
||||
paste the [soak] summary alongside your bench numbers.
|
||||
See docs/CLIFFS.md for context.
|
||||
|
||||
PASS VERDICT — WHAT IT DOES AND DOES NOT MEAN
|
||||
PASS = no failure signal on the test sample (silent_empty=0, VRAM
|
||||
growth under threshold, TPS retention >= 98%, zero errors).
|
||||
PASS does NOT validate that patches in the compose's overlay set
|
||||
are load-bearing for the workload — topology alone (e.g. TP=2)
|
||||
can sidestep the failure mode patches target. For patch attribution,
|
||||
re-run the same soak with overlays stripped and compare.
|
||||
Full discussion: docs/CLIFFS.md and issue #140.
|
||||
|
||||
EOF
|
||||
}
|
||||
|
||||
@@ -4,7 +4,8 @@ set -euo pipefail
|
||||
ROOT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../.." && pwd)"
|
||||
HELPER="${ROOT_DIR}/scripts/lib/profiles/launch_compat.py"
|
||||
GPU_3090='0|RTX_3090|24576|8.6'
|
||||
MTP_SHA="1acd67a795ebccdf9b9db7697ae9082058301657"
|
||||
MTP_SHA="01d4d1ad375dc5854779c593eee093bcebb0cada"
|
||||
CLEAN_SHA="bf610c2f56764e1b30bc6065f4ceace3d6e59036"
|
||||
DFLASH_SHA="e47c98ef7a38792996e452ef53914e21e41928e9"
|
||||
|
||||
assert_contains() {
|
||||
@@ -85,16 +86,19 @@ fi
|
||||
assert_contains "$out" "install.spec is not a docker nightly image"
|
||||
|
||||
out="$(python3 "$HELPER" resolve-variant-pin --variant vllm/dual --format shell)"
|
||||
assert_contains "$out" "VLLM_NIGHTLY_SHA=${CLEAN_SHA}"
|
||||
|
||||
out="$(python3 "$HELPER" resolve-variant-pin --variant vllm/dual-tq3-mtp --format shell)"
|
||||
assert_contains "$out" "VLLM_NIGHTLY_SHA=${MTP_SHA}"
|
||||
|
||||
out="$(python3 "$HELPER" resolve-variant-pin --variant vllm/gemma-dflash --format shell)"
|
||||
assert_contains "$out" "VLLM_NIGHTLY_SHA=${DFLASH_SHA}"
|
||||
|
||||
if command -v docker >/dev/null 2>&1 && docker compose version >/dev/null 2>&1; then
|
||||
out="$(VLLM_NIGHTLY_SHA="$MTP_SHA" docker compose -f "$ROOT_DIR/models/qwen3.6-27b/vllm/compose/dual/docker-compose.yml" config 2>/dev/null)"
|
||||
assert_contains "$out" "image: vllm/vllm-openai:nightly-${MTP_SHA}"
|
||||
out="$(VLLM_NIGHTLY_SHA="$CLEAN_SHA" docker compose -f "$ROOT_DIR/models/qwen3.6-27b/vllm/compose/dual/docker-compose.yml" config 2>/dev/null)"
|
||||
assert_contains "$out" "image: vllm/vllm-openai:nightly-${CLEAN_SHA}"
|
||||
|
||||
out="$(VLLM_NIGHTLY_SHA="$MTP_SHA" VLLM_IMAGE=ghcr.io/noonghunna/vllm-club3090:latest docker compose -f "$ROOT_DIR/models/qwen3.6-27b/vllm/compose/dual/docker-compose.yml" config 2>/dev/null)"
|
||||
out="$(VLLM_NIGHTLY_SHA="$CLEAN_SHA" VLLM_IMAGE=ghcr.io/noonghunna/vllm-club3090:latest docker compose -f "$ROOT_DIR/models/qwen3.6-27b/vllm/compose/dual/docker-compose.yml" config 2>/dev/null)"
|
||||
assert_contains "$out" "image: ghcr.io/noonghunna/vllm-club3090:latest"
|
||||
fi
|
||||
|
||||
@@ -102,12 +106,15 @@ out="$(python3 - <<'PY'
|
||||
from scripts.lib.profiles.compat import InstanceSpec
|
||||
from scripts.lib.profiles.estate_cli import compose_env
|
||||
|
||||
mtp = compose_env(InstanceSpec(name="qwen", compose_name="vllm/dual", gpu_indices=(0, 1), port=8010))
|
||||
clean = compose_env(InstanceSpec(name="qwen", compose_name="vllm/dual", gpu_indices=(0, 1), port=8010))
|
||||
tq3 = compose_env(InstanceSpec(name="qwen-tq3", compose_name="vllm/dual-tq3-mtp", gpu_indices=(0, 1), port=8010))
|
||||
dflash = compose_env(InstanceSpec(name="gemma", compose_name="vllm/gemma-dflash", gpu_indices=(0, 1), port=8032))
|
||||
print(mtp["VLLM_NIGHTLY_SHA"])
|
||||
print(clean["VLLM_NIGHTLY_SHA"])
|
||||
print(tq3["VLLM_NIGHTLY_SHA"])
|
||||
print(dflash["VLLM_NIGHTLY_SHA"])
|
||||
PY
|
||||
)"
|
||||
assert_contains "$out" "$CLEAN_SHA"
|
||||
assert_contains "$out" "$MTP_SHA"
|
||||
assert_contains "$out" "$DFLASH_SHA"
|
||||
|
||||
|
||||
Executable
+240
@@ -0,0 +1,240 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
ROOT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../.." && pwd)"
|
||||
TMP_DIR="$(mktemp -d)"
|
||||
trap 'rm -rf "$TMP_DIR"' EXIT
|
||||
|
||||
assert_contains() {
|
||||
local haystack="$1"
|
||||
local needle="$2"
|
||||
if [[ "$haystack" != *"$needle"* ]]; then
|
||||
echo "ASSERTION FAILED: expected output to contain: $needle" >&2
|
||||
echo "--- output ---" >&2
|
||||
echo "$haystack" >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
FAKE_GPUS_4X='0:RTX_3090:24576:8.6,1:RTX_3090:24576:8.6,2:RTX_3090:24576:8.6,3:RTX_3090:24576:8.6'
|
||||
|
||||
ESTATE4="${TMP_DIR}/estate4.yml"
|
||||
cat > "$ESTATE4" <<'YAML'
|
||||
schema_version: 1
|
||||
created: 2026-05-15T00:00:00Z
|
||||
rig:
|
||||
hardware_id: rtx-3090
|
||||
gpu_count: 4
|
||||
nvlink_active: false
|
||||
estate:
|
||||
- name: one
|
||||
compose: llamacpp/default
|
||||
gpus: [0]
|
||||
port: 8110
|
||||
- name: two
|
||||
compose: llamacpp/default
|
||||
gpus: [1]
|
||||
port: 8120
|
||||
- name: three
|
||||
compose: llamacpp/default
|
||||
gpus: [2]
|
||||
port: 8130
|
||||
- name: four
|
||||
compose: llamacpp/default
|
||||
gpus: [3]
|
||||
port: 8140
|
||||
YAML
|
||||
|
||||
FAKE_ESTATE_HELPER="${TMP_DIR}/estate-helper.py"
|
||||
cat > "$FAKE_ESTATE_HELPER" <<'PY'
|
||||
#!/usr/bin/env python3
|
||||
import sys
|
||||
print("ARGS " + " ".join(sys.argv[1:]))
|
||||
PY
|
||||
chmod +x "$FAKE_ESTATE_HELPER"
|
||||
|
||||
out="$(ESTATE_HELPER="$FAKE_ESTATE_HELPER" bash "${ROOT_DIR}/scripts/launch.sh" \
|
||||
--no-preflight \
|
||||
--estate-file "$ESTATE4" \
|
||||
--only one \
|
||||
--parallel \
|
||||
--parallel-jobs 3 \
|
||||
--parallel-stagger 0 2>&1)"
|
||||
assert_contains "$out" "ARGS boot --file $ESTATE4 --only one --parallel --parallel-jobs 3 --parallel-stagger 0"
|
||||
|
||||
out="$(
|
||||
cd "$ROOT_DIR"
|
||||
HOME="${TMP_DIR}/home-success" \
|
||||
CLUB3090_FAKE_GPUS="$FAKE_GPUS_4X" \
|
||||
CLUB3090_ESTATE_BOOT_LOG_DIR="${TMP_DIR}/logs-success" \
|
||||
python3 - "$ESTATE4" <<'PY'
|
||||
import argparse
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
|
||||
from scripts.lib.profiles import estate_cli as ec
|
||||
|
||||
state = {"active": 0, "max_active": 0}
|
||||
ready = []
|
||||
lock = threading.Lock()
|
||||
|
||||
|
||||
def fake_run_compose(inst, action, log_path=None):
|
||||
if log_path is not None:
|
||||
ec.append_log(log_path, f"compose {action} {inst.name}")
|
||||
with lock:
|
||||
state["active"] += 1
|
||||
state["max_active"] = max(state["max_active"], state["active"])
|
||||
time.sleep(0.05)
|
||||
with lock:
|
||||
state["active"] -= 1
|
||||
|
||||
|
||||
def fake_wait_ready_quiet(inst, timeout):
|
||||
ready.append(inst.name)
|
||||
return 1
|
||||
|
||||
|
||||
ec.run_compose = fake_run_compose
|
||||
ec.wait_ready_quiet = fake_wait_ready_quiet
|
||||
|
||||
rc = ec.command_boot(
|
||||
argparse.Namespace(
|
||||
file=sys.argv[1],
|
||||
only="",
|
||||
timeout=5,
|
||||
parallel=True,
|
||||
parallel_jobs=2,
|
||||
parallel_stagger=0.0,
|
||||
)
|
||||
)
|
||||
print(f"rc={rc}")
|
||||
print(f"max_active={state['max_active']}")
|
||||
print("ready=" + ",".join(sorted(ready)))
|
||||
for name in ("one", "two", "three", "four"):
|
||||
print(f"log:{name}={ec.instance_log_path(ec.InstanceSpec(name=name, compose_name='llamacpp/default', gpu_indices=(0,), port=8000)).exists()}")
|
||||
print(f"hard_cap={ec.effective_parallel_jobs(9, 8)}")
|
||||
PY
|
||||
)"
|
||||
assert_contains "$out" "[estate] parallel boot: 4 instance(s), jobs=2, stagger=0s"
|
||||
assert_contains "$out" "[estate] Summary: 4/4 healthy, 0 failed."
|
||||
assert_contains "$out" "rc=0"
|
||||
assert_contains "$out" "max_active=2"
|
||||
assert_contains "$out" "ready=four,one,three,two"
|
||||
assert_contains "$out" "log:one=True"
|
||||
assert_contains "$out" "log:four=True"
|
||||
assert_contains "$out" "hard_cap=4"
|
||||
|
||||
ESTATE2="${TMP_DIR}/estate2.yml"
|
||||
cat > "$ESTATE2" <<'YAML'
|
||||
schema_version: 1
|
||||
created: 2026-05-15T00:00:00Z
|
||||
rig:
|
||||
hardware_id: rtx-3090
|
||||
gpu_count: 2
|
||||
nvlink_active: false
|
||||
estate:
|
||||
- name: good
|
||||
compose: llamacpp/default
|
||||
gpus: [0]
|
||||
port: 8110
|
||||
- name: bad
|
||||
compose: llamacpp/default
|
||||
gpus: [1]
|
||||
port: 8120
|
||||
YAML
|
||||
|
||||
out="$(
|
||||
cd "$ROOT_DIR"
|
||||
HOME="${TMP_DIR}/home-fail" \
|
||||
CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:RTX_3090:24576:8.6' \
|
||||
CLUB3090_ESTATE_BOOT_LOG_DIR="${TMP_DIR}/logs-fail" \
|
||||
python3 - "$ESTATE2" <<'PY'
|
||||
import argparse
|
||||
import sys
|
||||
|
||||
from scripts.lib.profiles import estate_cli as ec
|
||||
|
||||
|
||||
def fake_run_compose(inst, action, log_path=None):
|
||||
if log_path is not None:
|
||||
ec.append_log(log_path, f"compose {action} {inst.name}")
|
||||
if inst.name == "bad":
|
||||
raise ec.EstateCliError("mock compose failed")
|
||||
|
||||
|
||||
def fake_wait_ready_quiet(inst, timeout):
|
||||
return 2
|
||||
|
||||
|
||||
ec.run_compose = fake_run_compose
|
||||
ec.wait_ready_quiet = fake_wait_ready_quiet
|
||||
|
||||
rc = ec.command_boot(
|
||||
argparse.Namespace(
|
||||
file=sys.argv[1],
|
||||
only="",
|
||||
timeout=5,
|
||||
parallel=True,
|
||||
parallel_jobs=2,
|
||||
parallel_stagger=0.0,
|
||||
)
|
||||
)
|
||||
print(f"rc={rc}")
|
||||
PY
|
||||
)"
|
||||
assert_contains "$out" "[estate] Summary: 1/2 healthy, 1 failed."
|
||||
assert_contains "$out" "bad ✗ failed"
|
||||
assert_contains "$out" "Failed instance: bad. See"
|
||||
assert_contains "$out" "rc=1"
|
||||
|
||||
out="$(
|
||||
cd "$ROOT_DIR"
|
||||
HOME="${TMP_DIR}/home-single" \
|
||||
CLUB3090_FAKE_GPUS="$FAKE_GPUS_4X" \
|
||||
CLUB3090_ESTATE_BOOT_LOG_DIR="${TMP_DIR}/logs-single" \
|
||||
python3 - "$ESTATE4" <<'PY'
|
||||
import argparse
|
||||
import sys
|
||||
|
||||
from scripts.lib.profiles import estate_cli as ec
|
||||
|
||||
events = []
|
||||
|
||||
|
||||
def fake_run_compose(inst, action):
|
||||
events.append(f"{action}:{inst.name}")
|
||||
|
||||
|
||||
def fake_wait_ready(inst, timeout):
|
||||
events.append(f"ready:{inst.name}")
|
||||
|
||||
|
||||
def fake_wait_ready_quiet(inst, timeout):
|
||||
raise AssertionError("single-instance --parallel should use sequential boot")
|
||||
|
||||
|
||||
ec.run_compose = fake_run_compose
|
||||
ec.wait_ready = fake_wait_ready
|
||||
ec.wait_ready_quiet = fake_wait_ready_quiet
|
||||
|
||||
rc = ec.command_boot(
|
||||
argparse.Namespace(
|
||||
file=sys.argv[1],
|
||||
only="one",
|
||||
timeout=5,
|
||||
parallel=True,
|
||||
parallel_jobs=2,
|
||||
parallel_stagger=0.0,
|
||||
)
|
||||
)
|
||||
print(f"rc={rc}")
|
||||
print("events=" + ",".join(events))
|
||||
PY
|
||||
)"
|
||||
assert_contains "$out" "[estate] all selected instances are healthy"
|
||||
assert_contains "$out" "rc=0"
|
||||
assert_contains "$out" "events=up:one,ready:one"
|
||||
|
||||
echo "test-parallel-boot: ok"
|
||||
@@ -24,11 +24,11 @@ run_test "load_profiles parses all profile groups" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles
|
||||
p = load_profiles()
|
||||
assert len(p.hardware) == 9
|
||||
assert len(p.models) == 2
|
||||
assert len(p.models) == 4
|
||||
assert len(p.workloads) == 5
|
||||
assert len(p.engines) == 6
|
||||
assert len(p.drafters) == 5
|
||||
assert len(p.calibration) == 2
|
||||
assert len(p.engines) == 7
|
||||
assert len(p.drafters) == 6
|
||||
assert len(p.calibration) == 4
|
||||
PY
|
||||
|
||||
run_test "fits() happy path: Qwen dual on 2x3090" <<'PY'
|
||||
@@ -49,6 +49,77 @@ assert r.recommended_kv_format == "turboquant_3bit_nc"
|
||||
assert r.diagnostics["constraints_skipped"] == ["C12"]
|
||||
PY
|
||||
|
||||
run_test "topology: single card classified" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"]])
|
||||
assert r == TopologyClass.SINGLE_CARD
|
||||
PY
|
||||
|
||||
run_test "topology: 2x3090 classified homogeneous" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"], p.hardware["rtx-3090"]])
|
||||
assert r == TopologyClass.HOMOGENEOUS
|
||||
PY
|
||||
|
||||
run_test "topology: 3090+4090 classified compute-mismatched" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"], p.hardware["rtx-4090"]])
|
||||
assert r == TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED
|
||||
PY
|
||||
|
||||
run_test "topology: 3090+3060 classified VRAM-mismatched" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"], p.hardware["rtx-3060-12gb"]])
|
||||
assert r == TopologyClass.VRAM_MISMATCHED
|
||||
PY
|
||||
|
||||
run_test "topology: VRAM cluster wins over mixed compute" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"], p.hardware["rtx-3060-12gb"], p.hardware["rtx-4090"]])
|
||||
assert r == TopologyClass.VRAM_MISMATCHED
|
||||
PY
|
||||
|
||||
run_test "C16 topology advisory emits note for compute mismatch" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, fits, TopologyClass
|
||||
p = load_profiles()
|
||||
r = fits(
|
||||
hardware=[p.hardware["rtx-3090"], p.hardware["rtx-4090"]],
|
||||
model=p.models["qwen3.6-27b"],
|
||||
workload=p.workloads["long-ctx-single"],
|
||||
engine=p.engines["vllm-nightly-mtp"],
|
||||
drafter=p.drafters["qwen-mtp-builtin"],
|
||||
tp=2,
|
||||
pp=1,
|
||||
project_vram=False,
|
||||
)
|
||||
assert r.topology_class == TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED
|
||||
assert "C16" in r.diagnostics["constraints_passed"]
|
||||
assert any("C16" in n and "vram_matched_compute_mismatched" in n for n in r.notes), r.notes
|
||||
PY
|
||||
|
||||
run_test "C16 topology advisory is silent for homogeneous GPUs" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, fits, TopologyClass
|
||||
p = load_profiles()
|
||||
r = fits(
|
||||
hardware=[p.hardware["rtx-3090"], p.hardware["rtx-3090"]],
|
||||
model=p.models["qwen3.6-27b"],
|
||||
workload=p.workloads["long-ctx-single"],
|
||||
engine=p.engines["vllm-nightly-mtp"],
|
||||
drafter=p.drafters["qwen-mtp-builtin"],
|
||||
tp=2,
|
||||
pp=1,
|
||||
project_vram=False,
|
||||
)
|
||||
assert r.topology_class == TopologyClass.HOMOGENEOUS
|
||||
assert "C16" in r.diagnostics["constraints_passed"]
|
||||
assert not any("C16" in n for n in r.notes), r.notes
|
||||
PY
|
||||
|
||||
run_test "C1 card count: world size mismatch rejected" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, fits
|
||||
p = load_profiles()
|
||||
@@ -92,12 +163,22 @@ assert not r.valid
|
||||
assert any(reason.startswith("C5:") for reason in r.reasons), r.reasons
|
||||
PY
|
||||
|
||||
run_test "C6 Genesis one-way implication: Qwen on non-Genesis vLLM rejected" <<'PY'
|
||||
run_test "C6 Genesis one-way: TQ3 KV on non-Genesis engine rejected (via C15)" <<'PY'
|
||||
# Under the TQ3-only Genesis policy, no model declares requires_genesis=true
|
||||
# (so C6 has no current model-level trigger). Genesis is enforced at the
|
||||
# *feature* level via C15: requesting turboquant_3bit_nc on an engine that
|
||||
# doesn't expose it (e.g. vllm-stable-next) fails C15. The previous
|
||||
# C6 assertion (Qwen on non-Genesis vLLM rejected) no longer holds —
|
||||
# Qwen 27B with fp8 is valid on non-Genesis engines.
|
||||
from scripts.lib.profiles.compat import load_profiles, fits
|
||||
p = load_profiles()
|
||||
# Positive: Qwen 27B + fp8 on non-Genesis engine is now valid.
|
||||
r = fits([p.hardware["rtx-3090"]], p.models["qwen3.6-27b"], p.workloads["long-ctx-single"], p.engines["vllm-stable-next"], kv_format="fp8_e5m2", tp=1, project_vram=False)
|
||||
assert r.valid, r.reasons
|
||||
# Negative: Qwen 27B + TQ3 on non-Genesis engine fails C15.
|
||||
r = fits([p.hardware["rtx-3090"]], p.models["qwen3.6-27b"], p.workloads["long-ctx-single"], p.engines["vllm-stable-next"], kv_format="turboquant_3bit_nc", tp=1, project_vram=False, required_engine_features=["turboquant_3bit_nc"])
|
||||
assert not r.valid
|
||||
assert any(reason.startswith("C6:") for reason in r.reasons), r.reasons
|
||||
assert any(reason.startswith("C15:") for reason in r.reasons), r.reasons
|
||||
PY
|
||||
|
||||
run_test "C7 drafter method: DFlash on MTP-only engine rejected" <<'PY'
|
||||
@@ -218,7 +299,7 @@ from scripts.lib.profiles.compat import load_profiles, to_compose_name
|
||||
p = load_profiles()
|
||||
name = to_compose_name(
|
||||
p.models["qwen3.6-27b"],
|
||||
p.engines["vllm-nightly-mtp"],
|
||||
p.engines["vllm-nightly-clean"],
|
||||
p.drafters["qwen-mtp-builtin"],
|
||||
"fp8_e5m2",
|
||||
2,
|
||||
@@ -236,7 +317,7 @@ from scripts.lib.profiles.compat import load_profiles, fits
|
||||
p = load_profiles()
|
||||
r = fits([p.hardware["rtx-3090"]], p.models["qwen3.6-27b"], p.workloads["long-ctx-single"], p.engines["vllm-nightly-mtp"], tp=1, project_vram=False)
|
||||
d = r.diagnostics
|
||||
assert d["constraints_evaluated"] == [f"C{i}" for i in range(1, 16)]
|
||||
assert d["constraints_evaluated"] == [f"C{i}" for i in range(1, 17)]
|
||||
assert "constraints_passed" in d and "constraints_failed" in d and "constraints_skipped" in d
|
||||
assert isinstance(d["elapsed_ms"], float)
|
||||
PY
|
||||
|
||||
@@ -143,6 +143,7 @@ out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:
|
||||
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
|
||||
--no-preflight --no-verify --model qwen3.6-27b --gpus 0,1 --no-projection 2>&1)"
|
||||
assert_contains "$out" "[launch] Tensor parallel TP=2"
|
||||
assert_not_contains "$out" "Topology:"
|
||||
assert_contains "$out" "SWITCHED vllm/dual CUDA=0,1 NVD=0,1 TP=2 PP=1"
|
||||
selected_count="$(grep -c "\[launch\] selected variant:" <<< "$out" || true)"
|
||||
if [[ "$selected_count" != "1" ]]; then
|
||||
@@ -179,6 +180,24 @@ if out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6
|
||||
fi
|
||||
assert_contains "$out" "Gemma 4 31B does not fit on a single 24 GB card today"
|
||||
|
||||
out="$(CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:RTX_3090:24576:8.6' \
|
||||
bash "${ROOT_DIR}/scripts/launch.sh" --topology 2>&1)"
|
||||
assert_contains "$out" "Topology class: homogeneous"
|
||||
assert_not_contains "$out" "Compute mismatch detected"
|
||||
|
||||
out="$(CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:RTX_4090:24576:8.9' \
|
||||
bash "${ROOT_DIR}/scripts/launch.sh" --topology 2>&1)"
|
||||
assert_contains "$out" "Topology class: vram_matched_compute_mismatched"
|
||||
assert_contains "$out" "Compute mismatch detected"
|
||||
assert_contains "$out" "Estate planner"
|
||||
|
||||
out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:RTX_4090:24576:8.9' \
|
||||
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
|
||||
--no-preflight --no-verify --model qwen3.6-27b --gpus 0,1 --no-projection 2>&1)"
|
||||
assert_contains "$out" "Topology: vram_matched_compute_mismatched"
|
||||
assert_contains "$out" "Compute mismatch detected"
|
||||
assert_contains "$out" "SWITCHED vllm/dual CUDA=0,1 NVD=0,1 TP=2 PP=1"
|
||||
|
||||
if out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:RTX_3090:24576:8.6,2:RTX_3090:24576:8.6,3:RTX_3090:24576:8.6,4:RTX_3090:24576:8.6,5:RTX_3090:24576:8.6' \
|
||||
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
|
||||
--no-preflight --no-verify --model qwen3.6-27b --gpus 0,1,2,3,4,5 --tp 6 --no-projection 2>&1)"; then
|
||||
|
||||
@@ -50,12 +50,18 @@ for dir in "${fixtures[@]}"; do
|
||||
}
|
||||
assert_contains "$row" "Report: \`results/rebench/${dir##*/}/REPORT.md\`"
|
||||
if [[ "$section" == "Gemma 4 31B (community-experimental)" ]]; then
|
||||
assert_columns "$row" 10
|
||||
assert_columns "$row" 11
|
||||
else
|
||||
assert_columns "$row" 8
|
||||
assert_columns "$row" 9
|
||||
fi
|
||||
done
|
||||
|
||||
out="$(BENCH_MOCK=1 RUNS=1 WARMUPS=0 bash scripts/bench.sh)"
|
||||
assert_contains "$out" "PP tok/s"
|
||||
out="$(BENCH_MOCK=1 PP=1 RUNS=1 WARMUPS=0 bash scripts/bench.sh)"
|
||||
assert_contains "$out" "summary [prompt-processing]"
|
||||
assert_contains "$out" "PP tok/s"
|
||||
|
||||
tag="qwen-int8-pth-n4-2026-05-10"
|
||||
rm -f "results/rebench/${tag}/BENCHMARKS-row.md" \
|
||||
"results/rebench/${tag}/PR-body.md" \
|
||||
|
||||
+134
-9
@@ -15,9 +15,11 @@ Predicts (per card, after TP split):
|
||||
- Total vs available VRAM
|
||||
- Verdict: PASS / TIGHT / FAIL
|
||||
|
||||
Two models modelled:
|
||||
Four models modelled:
|
||||
- Qwen 3.6 27B (DeltaNet hybrid: 16 full_attention + 48 GDN)
|
||||
- Qwen 3.6 35B-A3B (MoE + DeltaNet hybrid: 10 attention + 30 GDN)
|
||||
- Gemma 4 31B (SWA + dense MLP: 10 full_attention + 50 sliding_attention)
|
||||
- Gemma 4 26B-A4B (MoE + SWA: 5 full_attention + 25 sliding_attention)
|
||||
|
||||
vLLM rate-limits KV pool to fit available budget; this predictor models that
|
||||
capping behavior. When the requested KV pool exceeds what fits, the verdict
|
||||
@@ -39,7 +41,7 @@ Usage:
|
||||
bash tools/kv-calc.py --compose dual-turbo --vram 24 # Qwen (default model)
|
||||
bash tools/kv-calc.py --model gemma-4-31b --compose gemma-dual-int8 --vram 24
|
||||
bash tools/kv-calc.py --model gemma-4-31b --solve-max-ctx --kv-format int8_per_token_head --tp 2 --vram 24
|
||||
bash tools/kv-calc.py --calibration # both models, grouped per-model
|
||||
bash tools/kv-calc.py --calibration # all calibrated models, grouped per-model
|
||||
"""
|
||||
|
||||
import argparse
|
||||
@@ -90,16 +92,28 @@ def _weight_size(model, variant):
|
||||
|
||||
def _load_model_specs_from_yaml(profiles):
|
||||
qwen, gemma = profiles.models["qwen3.6-27b"], profiles.models["gemma-4-31b"]
|
||||
qwen_moe, gemma_moe = profiles.models["qwen3.6-35b-a3b"], profiles.models["gemma-4-26b-a4b"]
|
||||
q_fields = ("hidden_size", "num_hidden_layers", "num_gdn_layers", "num_attn_layers", "num_attn_heads", "num_kv_heads", "head_dim_attn", "linear_num_v_heads", "linear_num_k_heads", "linear_v_head_dim", "linear_k_head_dim", "linear_conv_kernel_dim", "max_ctx_supported", "attention_k_eq_v")
|
||||
g_fields = ("hidden_size", "intermediate_size", "num_hidden_layers", "num_full_attn_layers", "num_sliding_attn_layers", "num_attn_heads", "num_kv_heads", "head_dim_sliding", "global_head_dim", "sliding_window", "max_ctx_supported", "attention_k_eq_v")
|
||||
qspec = {"model_id": qwen.id, "model_family": qwen.family, **{k: getattr(qwen, k) for k in q_fields}, "valid_tp": list(qwen.valid_tp), "weights_total_gb": _weight_size(qwen, qwen.default_weight_variant), "mamba_state_bytes": 4, "chunk_size": 256}
|
||||
gspec = {"model_id": gemma.id, "model_family": gemma.family, **{k: getattr(gemma, k) for k in g_fields}, "valid_tp": list(gemma.valid_tp), "weights_int4_gb": _weight_size(gemma, "autoround_int4"), "weights_awq_gb": _weight_size(gemma, "awq"), "weights_bf16_gb": _weight_size(gemma, "bf16"), "drafter_mtp_gb": float(profiles.drafters["gemma-it-assistant"].vram_footprint_gb), "drafter_dflash_gb": float(profiles.drafters["gemma-dflash"].vram_footprint_gb)}
|
||||
return {"qwen3.6-27b": qspec, "gemma-4-31b": gspec}
|
||||
gm_fields = (*g_fields, "num_global_kv_heads", "num_experts", "num_experts_per_tok", "moe_intermediate_size", "active_params_b", "mtp_num_hidden_layers")
|
||||
qm_fields = (*q_fields, "num_experts", "num_experts_per_tok", "moe_intermediate_size", "shared_expert_intermediate_size", "active_params_b", "mtp_num_hidden_layers")
|
||||
qspec = {"model_id": qwen.id, "model_family": qwen.family, **{k: getattr(qwen, k) for k in q_fields}, "valid_tp": list(qwen.valid_tp), "weights_total_gb": _weight_size(qwen, qwen.default_weight_variant), "mamba_state_bytes": 4, "chunk_size": 256, "mtp_n_default": profiles.drafters["qwen-mtp-builtin"].n_default}
|
||||
qmspec = {"model_id": qwen_moe.id, "model_family": qwen_moe.family, **{k: getattr(qwen_moe, k) for k in qm_fields}, "valid_tp": list(qwen_moe.valid_tp), "weights_total_gb": _weight_size(qwen_moe, qwen_moe.default_weight_variant), "weights_gptq_gb": _weight_size(qwen_moe, "gptq_int4"), "mamba_state_bytes": 4, "chunk_size": 256, "mtp_n_default": profiles.drafters["qwen-mtp-builtin"].n_default}
|
||||
gspec = {"model_id": gemma.id, "model_family": gemma.family, **{k: getattr(gemma, k) for k in g_fields}, "valid_tp": list(gemma.valid_tp), "weights_int4_gb": _weight_size(gemma, "autoround_int4"), "weights_awq_gb": _weight_size(gemma, "awq"), "weights_bf16_gb": _weight_size(gemma, "bf16"), "drafter_mtp_gb": float(profiles.drafters["gemma-it-assistant"].vram_footprint_gb), "drafter_dflash_gb": float(profiles.drafters["gemma-dflash"].vram_footprint_gb), "mtp_n_default": profiles.drafters["gemma-it-assistant"].n_default}
|
||||
gmspec = {"model_id": gemma_moe.id, "model_family": gemma_moe.family, **{k: getattr(gemma_moe, k) for k in gm_fields}, "valid_tp": list(gemma_moe.valid_tp), "weights_int4_gb": _weight_size(gemma_moe, "autoround_int4_mixed"), "weights_awq_gb": _weight_size(gemma_moe, "awq_compressed_tensors"), "drafter_mtp_gb": float(profiles.drafters["gemma-26b-it-assistant"].vram_footprint_gb), "mtp_n_default": profiles.drafters["gemma-26b-it-assistant"].n_default}
|
||||
return {
|
||||
"qwen3.6-27b": qspec,
|
||||
"qwen3.6-35b-a3b": qmspec,
|
||||
"gemma-4-31b": gspec,
|
||||
"gemma-4-26b-a4b": gmspec,
|
||||
}
|
||||
|
||||
|
||||
MODEL_SPECS = _load_model_specs_from_yaml(PROFILES)
|
||||
QWEN36_27B = MODEL_SPECS["qwen3.6-27b"]
|
||||
QWEN36_35B_A3B = MODEL_SPECS["qwen3.6-35b-a3b"]
|
||||
GEMMA4_31B = MODEL_SPECS["gemma-4-31b"]
|
||||
GEMMA4_26B_A4B = MODEL_SPECS["gemma-4-26b-a4b"]
|
||||
|
||||
|
||||
# =============================================================================
|
||||
@@ -141,6 +155,25 @@ QWEN_GDN_ACTIVATION_COEF = {
|
||||
"turboquant_3bit_nc": 165,
|
||||
}
|
||||
|
||||
# ---- Qwen MoE activation + built-in MTP workspace ----
|
||||
# Path-B low-anchor fit from the two v0.7.3 preview rows. The per-token GDN
|
||||
# coefficient follows the dense-Qwen shape; the small constant captures MoE
|
||||
# expert dispatch/router buffers. Current vLLM TP preview effectively keeps
|
||||
# the quantized MoE weights resident per card, so weights are not divided by TP
|
||||
# for qwen3-next-moe in _weights_per_card_gb().
|
||||
QWEN_MOE_ACTIVATION_COEF = {
|
||||
"fp16": 110,
|
||||
"bf16": 110,
|
||||
"fp8_e5m2": 105,
|
||||
"fp8_e4m3": 105,
|
||||
"int8_per_token_head": 105,
|
||||
"q4_0": 130,
|
||||
"k8v4": 130,
|
||||
"turboquant_3bit_nc": 140,
|
||||
}
|
||||
QWEN_MOE_EXPERT_DISPATCH_GB = 0.20
|
||||
QWEN_MOE_BUILTIN_MTP_WORKSPACE_GB = 0.10
|
||||
|
||||
# ---- Gemma activation peak (mostly constant in ctx) ----
|
||||
# Unlike Qwen GDN, Gemma's activation peak comes from dense MLP forward +
|
||||
# SWA windowed-attention prefill, both bounded by chunked-prefill chunk_size.
|
||||
@@ -149,13 +182,22 @@ QWEN_GDN_ACTIVATION_COEF = {
|
||||
GEMMA_ACTIVATION_CONST_GB = 1.5 # per card at TP=1 — calibrated, ~scales as 1/TP
|
||||
GEMMA_ACTIVATION_PER_TOKEN_BYTES = 8 # tiny ctx scaling term to keep solver well-behaved
|
||||
|
||||
# ---- Gemma MoE activation peak ----
|
||||
# Low-anchor fit from awq.yml and awq-mtp.yml. MoE dispatch is folded into the
|
||||
# constant term; external assistant weights are modelled separately via
|
||||
# drafter_gb.
|
||||
GEMMA_MOE_ACTIVATION_CONST_GB = 1.8
|
||||
GEMMA_MOE_ACTIVATION_PER_TOKEN_BYTES = 12
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Compose presets (per-model)
|
||||
# =============================================================================
|
||||
COMPOSE_ALIAS_TEXT = {
|
||||
"qwen3.6-27b": "minimal=vllm/minimal long-text=vllm/long-text long-text-no-mtp=vllm/long-text-no-mtp long-vision=vllm/long-vision bounded-thinking=vllm/bounded-thinking tools-text=vllm/tools-text dual=vllm/dual dual-turbo=vllm/dual-turbo dual-dflash=vllm/dual-dflash dual-dflash-noviz=vllm/dual-dflash-noviz dual4=vllm/dual4 dual4-dflash=vllm/dual4-dflash",
|
||||
"qwen3.6-35b-a3b": "qwen-a3b-preview-single=vllm/qwen-a3b-preview-single qwen-a3b-preview=vllm/qwen-a3b-preview qwen-a3b-preview-mtp=vllm/qwen-a3b-preview-mtp",
|
||||
"gemma-4-31b": "gemma-dual=vllm/gemma-mtp gemma-dual-int8=vllm/gemma-int8 gemma-dual-int8-262k=vllm/gemma-int8-262k gemma-dual-bf16=vllm/gemma-bf16 gemma-dual-int8-tq3=vllm/gemma-int8-tq3 gemma-dual-dflash=vllm/gemma-dflash gemma-dual-dflash-int8=vllm/gemma-dflash-int8 gemma-dual-awq=vllm/gemma-awq gemma-single=vllm/gemma-mtp-tp1",
|
||||
"gemma-4-26b-a4b": "gemma-a4b-single=vllm/gemma-a4b-single gemma-a4b=vllm/gemma-a4b gemma-a4b-awq=vllm/gemma-a4b-awq gemma-a4b-awq-mtp=vllm/gemma-a4b-awq-mtp",
|
||||
}
|
||||
COMPOSE_ALIASES = {model: tuple(part.split("=", 1) for part in text.split()) for model, text in COMPOSE_ALIAS_TEXT.items()}
|
||||
|
||||
@@ -182,10 +224,18 @@ def _compose_cfg_from_registry(profiles, model_id, legacy_name, registry_name):
|
||||
cfg["mtp"] = drafter is not None and drafter.spec_method in ("mtp", "mtp_assistant")
|
||||
if drafter is not None and drafter.spec_method == "dflash":
|
||||
cfg.update({"mtp": False, "dflash_draft_gb": float(drafter.vram_footprint_gb)})
|
||||
if drafter is not None:
|
||||
cfg["mtp_n"] = int(drafter.n_default)
|
||||
if model_id == "gemma-4-31b" and drafter is not None:
|
||||
cfg["drafter_gb"] = float(drafter.vram_footprint_gb)
|
||||
if model_id == "gemma-4-31b":
|
||||
cfg["weights_variant"] = {"awq": "awq", "bf16": "bf16"}.get(entry["weights_variant"], "int4")
|
||||
if model_id == "gemma-4-26b-a4b":
|
||||
cfg["weights_variant"] = "awq" if entry["weights_variant"] == "awq_compressed_tensors" else "int4"
|
||||
if drafter is not None:
|
||||
cfg["drafter_gb"] = float(drafter.vram_footprint_gb)
|
||||
if model_id == "qwen3.6-35b-a3b":
|
||||
cfg["weights_variant"] = "gptq" if entry["weights_variant"] == "gptq_int4" else "default"
|
||||
cfg.update(COMPOSE_COMPAT_OVERRIDES.get((model_id, legacy_name), {}))
|
||||
return cfg
|
||||
|
||||
@@ -235,6 +285,12 @@ def _weights_per_card_gb(spec, tp, weights_variant="default"):
|
||||
"""Return per-card weights footprint in GB after TP split."""
|
||||
if spec["model_family"] == "qwen3-next-hybrid":
|
||||
return spec["weights_total_gb"] / tp
|
||||
elif spec["model_family"] == "qwen3-next-moe":
|
||||
# Current vLLM MoE preview keeps expert weights effectively resident
|
||||
# per TP rank; live 16K rows calibrate to full quant weight per card.
|
||||
if weights_variant == "gptq":
|
||||
return spec["weights_gptq_gb"]
|
||||
return spec["weights_total_gb"]
|
||||
elif spec["model_family"] == "gemma4-swa-dense":
|
||||
if weights_variant == "awq":
|
||||
return spec["weights_awq_gb"] / tp
|
||||
@@ -242,6 +298,12 @@ def _weights_per_card_gb(spec, tp, weights_variant="default"):
|
||||
return spec["weights_bf16_gb"] / tp
|
||||
else: # int4 default
|
||||
return spec["weights_int4_gb"] / tp
|
||||
elif spec["model_family"] == "gemma4-swa-moe":
|
||||
# Same MoE-residency assumption as Qwen A3B; validated by the
|
||||
# v0.7.3 AWQ rows where TP=2 still peaks near a full AWQ shard/card.
|
||||
if weights_variant == "int4":
|
||||
return spec["weights_int4_gb"]
|
||||
return spec["weights_awq_gb"]
|
||||
raise ValueError(f"Unknown model_family: {spec['model_family']}")
|
||||
|
||||
|
||||
@@ -278,6 +340,30 @@ def kv_pool_per_card_bytes(spec, kv_format, max_ctx, max_num_seqs, tp, mtp_n=0):
|
||||
growing = (per_token / tp) * effective_ctx * max_num_seqs
|
||||
return growing, 0.0
|
||||
|
||||
elif spec["model_family"] == "qwen3-next-moe":
|
||||
# K and V stored independently. GDN recurrent state is fixed-size and
|
||||
# per-stream, not context-linear.
|
||||
per_token = (
|
||||
spec["num_attn_layers"]
|
||||
* spec["num_kv_heads"]
|
||||
* spec["head_dim_attn"]
|
||||
* 2
|
||||
* bpe
|
||||
)
|
||||
effective_ctx = max_ctx + mtp_n * 32
|
||||
growing = (per_token / tp) * effective_ctx * max_num_seqs
|
||||
recurrent_per_stream = (
|
||||
spec["num_gdn_layers"]
|
||||
* (
|
||||
spec["linear_num_v_heads"] * spec["linear_v_head_dim"]
|
||||
+ spec["linear_num_k_heads"] * spec["linear_k_head_dim"]
|
||||
+ spec["linear_conv_kernel_dim"] * spec["hidden_size"]
|
||||
)
|
||||
* 2 # recurrent state kept in bf16 on this stack
|
||||
)
|
||||
recurrent_fixed = recurrent_per_stream * max_num_seqs
|
||||
return growing, recurrent_fixed
|
||||
|
||||
elif spec["model_family"] == "gemma4-swa-dense":
|
||||
# K==V tied → ×1 storage
|
||||
per_token_growing = (
|
||||
@@ -302,6 +388,28 @@ def kv_pool_per_card_bytes(spec, kv_format, max_ctx, max_num_seqs, tp, mtp_n=0):
|
||||
sliding_per_card = sliding_fixed_total / tp
|
||||
return growing, sliding_per_card
|
||||
|
||||
elif spec["model_family"] == "gemma4-swa-moe":
|
||||
# K==V tied. Global layers use their own KV-head count; sliding
|
||||
# layers keep the windowed KV head count.
|
||||
per_token_growing = (
|
||||
spec["num_full_attn_layers"]
|
||||
* spec["num_global_kv_heads"]
|
||||
* spec["global_head_dim"]
|
||||
* 1
|
||||
* bpe
|
||||
)
|
||||
growing = (per_token_growing / tp) * max_ctx * max_num_seqs
|
||||
sliding_fixed_total = (
|
||||
spec["num_sliding_attn_layers"]
|
||||
* spec["num_kv_heads"]
|
||||
* spec["head_dim_sliding"]
|
||||
* 1
|
||||
* bpe
|
||||
* spec["sliding_window"]
|
||||
)
|
||||
sliding_per_card = sliding_fixed_total / tp
|
||||
return growing, sliding_per_card
|
||||
|
||||
raise ValueError(f"Unknown model_family: {spec['model_family']}")
|
||||
|
||||
|
||||
@@ -320,11 +428,20 @@ def activation_peak_per_card_bytes(spec, kv_format, max_ctx, tp):
|
||||
coef = QWEN_GDN_ACTIVATION_COEF[kv_format]
|
||||
return (coef * spec["num_gdn_layers"] * max_ctx) / tp
|
||||
|
||||
elif spec["model_family"] == "qwen3-next-moe":
|
||||
coef = QWEN_MOE_ACTIVATION_COEF[kv_format]
|
||||
return (coef * spec["num_gdn_layers"] * max_ctx) / tp + QWEN_MOE_EXPERT_DISPATCH_GB * 1e9
|
||||
|
||||
elif spec["model_family"] == "gemma4-swa-dense":
|
||||
const_bytes = GEMMA_ACTIVATION_CONST_GB * 1e9
|
||||
per_token = GEMMA_ACTIVATION_PER_TOKEN_BYTES * max_ctx
|
||||
return (const_bytes + per_token) / tp
|
||||
|
||||
elif spec["model_family"] == "gemma4-swa-moe":
|
||||
const_bytes = GEMMA_MOE_ACTIVATION_CONST_GB * 1e9
|
||||
per_token = GEMMA_MOE_ACTIVATION_PER_TOKEN_BYTES * max_ctx
|
||||
return (const_bytes + per_token) / tp
|
||||
|
||||
raise ValueError(f"Unknown model_family: {spec['model_family']}")
|
||||
|
||||
|
||||
@@ -375,9 +492,10 @@ def predict(
|
||||
|
||||
weights_gb = _weights_per_card_gb(spec, tp, weights_variant)
|
||||
|
||||
mtp_n = int(spec.get("mtp_n_default", 3)) if mtp else 0
|
||||
growing_b, sliding_b = kv_pool_per_card_bytes(
|
||||
spec, kv_format, max_ctx, max_num_seqs, tp,
|
||||
mtp_n=3 if mtp else 0,
|
||||
mtp_n=mtp_n,
|
||||
)
|
||||
kv_pool_requested_gb = growing_b / 1e9
|
||||
kv_pool_sliding_fixed_gb = sliding_b / 1e9
|
||||
@@ -387,6 +505,8 @@ def predict(
|
||||
|
||||
# Drafter: prefer drafter_gb; fall back to legacy dflash_draft_gb.
|
||||
drafter_total = drafter_gb if drafter_gb > 0 else dflash_draft_gb
|
||||
if mtp and spec["model_family"] == "qwen3-next-moe":
|
||||
drafter_total += QWEN_MOE_BUILTIN_MTP_WORKSPACE_GB
|
||||
drafter_per_card = drafter_total / tp if tp > 1 else drafter_total
|
||||
|
||||
fixed_gb = weights_gb + activation_gb + overhead_gb + drafter_per_card + kv_pool_sliding_fixed_gb
|
||||
@@ -407,7 +527,10 @@ def predict(
|
||||
# concurrency reduced (BOOT OK, but `--max-num-seqs` may not be
|
||||
# honored at full max_ctx).
|
||||
# - PASS: requested KV fits with room to spare.
|
||||
MIN_KV_GB = 1.0 # vLLM needs at least ~1 GB for paged-attention blocks
|
||||
# The Qwen A3B preview is KV-light enough that the live 16K rows boot with
|
||||
# <0.1 GB requested growing KV. Keep the older 1 GB guard for dense/long-KV
|
||||
# models, but avoid false FAILs on this MoE family.
|
||||
MIN_KV_GB = 0.05 if spec["model_family"] == "qwen3-next-moe" else 1.0
|
||||
if available_for_kv < MIN_KV_GB:
|
||||
verdict = "FAIL"
|
||||
notes.append(
|
||||
@@ -434,6 +557,8 @@ def predict(
|
||||
notes.append("⚠ fp8_e4m3 on Ampere (sm_86): Triton `fp8e4nv` kernel unsupported; use int8_per_token_head instead (PR #40391 via #42102)")
|
||||
if spec["model_family"] == "gemma4-swa-dense" and tp == 1 and vram_gb < 32:
|
||||
notes.append("⚠ Gemma 4 31B TP=1 needs ≥32 GB VRAM; 24 GB Ampere boot-OOMs (model weights + drafter + min KV)")
|
||||
if spec["model_family"] in {"qwen3-next-moe", "gemma4-swa-moe"}:
|
||||
notes.append("MoE projection uses low-anchor calibration; add max_ctx/max_num_seqs A/B rows before treating this as production-grade.")
|
||||
if tp > 4:
|
||||
notes.append("TP > 4 predictions are extrapolated; report deltas via scripts/report.sh --bench")
|
||||
|
||||
@@ -554,7 +679,7 @@ def run_calibration():
|
||||
print()
|
||||
|
||||
total_c, total_n = 0, 0
|
||||
for model_key in ("qwen3.6-27b", "gemma-4-31b"):
|
||||
for model_key in MODEL_SPECS:
|
||||
c, n = _calibration_block(model_key)
|
||||
total_c += c
|
||||
total_n += n
|
||||
@@ -642,7 +767,7 @@ def main():
|
||||
help="(deprecated alias for --drafter-gb)")
|
||||
p.add_argument("--weights-variant", choices=["default", "int4", "awq", "bf16"], default=None,
|
||||
help="Gemma 4 only: which weight quant variant. Default: from --compose, or int4.")
|
||||
p.add_argument("--calibration", action="store_true", help="Print predicted vs measured for both models.")
|
||||
p.add_argument("--calibration", action="store_true", help="Print predicted vs measured for all calibrated models.")
|
||||
p.add_argument("--solve-max-ctx", action="store_true", help="Binary-search for the largest max_ctx that fits.")
|
||||
p.add_argument("--json", action="store_true", help="Output prediction as JSON.")
|
||||
args = p.parse_args()
|
||||
|
||||
Reference in New Issue
Block a user