Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d116ba9ba3 | ||
|
|
6775d10919 | ||
|
|
2a148d702b | ||
|
|
57eb269cd7 | ||
|
|
ce2617e0bc | ||
|
|
02249ab193 | ||
|
|
6bd8e2420c |
@@ -7,15 +7,6 @@ on:
|
||||
description: "Upstream vLLM image to vendor overlays into"
|
||||
required: false
|
||||
default: "vllm/vllm-openai:nightly-1acd67a795ebccdf9b9db7697ae9082058301657"
|
||||
smoke:
|
||||
description: "GPU smoke behavior"
|
||||
required: false
|
||||
default: "auto"
|
||||
type: choice
|
||||
options:
|
||||
- auto
|
||||
- skip
|
||||
- required
|
||||
schedule:
|
||||
- cron: "0 0 * * 0"
|
||||
push:
|
||||
@@ -36,14 +27,12 @@ concurrency:
|
||||
cancel-in-progress: false
|
||||
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
packages: write
|
||||
|
||||
env:
|
||||
IMAGE_NAME: ghcr.io/noonghunna/vllm-club3090
|
||||
DEFAULT_VLLM_BASE_IMAGE: vllm/vllm-openai:nightly-1acd67a795ebccdf9b9db7697ae9082058301657
|
||||
CANONICAL_COMPOSE: models/qwen3.6-27b/vllm/compose/dual/docker-compose.yml
|
||||
|
||||
jobs:
|
||||
build:
|
||||
@@ -111,80 +100,11 @@ jobs:
|
||||
org.opencontainers.image.version=${{ steps.meta.outputs.image_tag }}
|
||||
club3090.upstream_vllm_image=${{ steps.meta.outputs.upstream_image }}
|
||||
|
||||
detect-smoke-runner:
|
||||
name: Detect self-hosted GPU runner
|
||||
promote-aliases:
|
||||
name: Promote latest and nightly-stable
|
||||
needs: build
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
available: ${{ steps.detect.outputs.available }}
|
||||
smoke_mode: ${{ steps.detect.outputs.smoke_mode }}
|
||||
steps:
|
||||
- name: Detect online gpu-labeled runner
|
||||
id: detect
|
||||
uses: actions/github-script@v7
|
||||
env:
|
||||
SMOKE_MODE: ${{ inputs.smoke || 'auto' }}
|
||||
with:
|
||||
script: |
|
||||
const mode = process.env.SMOKE_MODE || "auto";
|
||||
core.setOutput("smoke_mode", mode);
|
||||
|
||||
if (mode === "skip") {
|
||||
core.notice("Smoke explicitly skipped. Dated image was pushed; aliases will not move.");
|
||||
core.setOutput("available", "false");
|
||||
return;
|
||||
}
|
||||
|
||||
const runners = await github.paginate(
|
||||
github.rest.actions.listSelfHostedRunnersForRepo,
|
||||
{
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
per_page: 100,
|
||||
},
|
||||
);
|
||||
|
||||
const available = runners.some((runner) => {
|
||||
const labels = runner.labels.map((label) => label.name.toLowerCase());
|
||||
return runner.status === "online" &&
|
||||
labels.includes("self-hosted") &&
|
||||
labels.includes("gpu");
|
||||
});
|
||||
|
||||
core.setOutput("available", available ? "true" : "false");
|
||||
if (!available) {
|
||||
const message = "No online self-hosted runner with label 'gpu' was found. Dated image was pushed; latest/nightly-stable were not moved.";
|
||||
if (mode === "required") {
|
||||
core.setFailed(message);
|
||||
} else {
|
||||
core.notice(message);
|
||||
}
|
||||
}
|
||||
|
||||
smoke:
|
||||
name: GPU smoke and alias promotion
|
||||
needs:
|
||||
- build
|
||||
- detect-smoke-runner
|
||||
if: needs.detect-smoke-runner.outputs.available == 'true'
|
||||
runs-on:
|
||||
- self-hosted
|
||||
- linux
|
||||
- x64
|
||||
- gpu
|
||||
timeout-minutes: 120
|
||||
env:
|
||||
IMAGE_REF: ${{ needs.build.outputs.image_ref }}
|
||||
IMAGE_NAME: ghcr.io/noonghunna/vllm-club3090
|
||||
COMPOSE_PROJECT_NAME: club3090-ci-vllm-dual
|
||||
COMPOSE_OVERRIDE: /tmp/club3090-ci-vllm-image.override.yml
|
||||
URL: http://localhost:8010
|
||||
MODEL: qwen3.6-27b-autoround
|
||||
CONTAINER: vllm-qwen36-27b-dual
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Log in to GHCR
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
@@ -192,125 +112,16 @@ jobs:
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Prepare canonical compose override
|
||||
- name: Promote :latest and :nightly-stable to dated tag
|
||||
shell: bash
|
||||
env:
|
||||
IMAGE_REF: ${{ needs.build.outputs.image_ref }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
docker pull "${IMAGE_REF}"
|
||||
cat > "${COMPOSE_OVERRIDE}" <<EOF
|
||||
services:
|
||||
vllm-qwen36-27b-dual:
|
||||
image: ${IMAGE_REF}
|
||||
EOF
|
||||
|
||||
- name: Stop prior club-3090 estate if present
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [[ -f "${HOME}/.club3090/estate.yml" ]]; then
|
||||
bash scripts/launch.sh --down-estate "${HOME}/.club3090/estate.yml" || true
|
||||
fi
|
||||
docker compose \
|
||||
-f "${CANONICAL_COMPOSE}" \
|
||||
-f "${COMPOSE_OVERRIDE}" \
|
||||
-p "${COMPOSE_PROJECT_NAME}" \
|
||||
down --remove-orphans || true
|
||||
|
||||
- name: Boot canonical dual vLLM compose
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
docker compose \
|
||||
-f "${CANONICAL_COMPOSE}" \
|
||||
-f "${COMPOSE_OVERRIDE}" \
|
||||
-p "${COMPOSE_PROJECT_NAME}" \
|
||||
up -d
|
||||
|
||||
- name: Wait for OpenAI endpoint
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
for _ in {1..120}; do
|
||||
if curl -sf -m 5 "${URL}/v1/models" >/dev/null; then
|
||||
exit 0
|
||||
fi
|
||||
sleep 10
|
||||
done
|
||||
docker compose \
|
||||
-f "${CANONICAL_COMPOSE}" \
|
||||
-f "${COMPOSE_OVERRIDE}" \
|
||||
-p "${COMPOSE_PROJECT_NAME}" \
|
||||
logs --tail=200
|
||||
exit 1
|
||||
|
||||
- name: Run verify-full
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
URL="${URL}" MODEL="${MODEL}" CONTAINER="${CONTAINER}" bash scripts/verify-full.sh
|
||||
|
||||
- name: Run 3-prompt smoke bench
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
python3 - <<'PY'
|
||||
import json
|
||||
import time
|
||||
import urllib.request
|
||||
|
||||
url = "http://localhost:8010"
|
||||
model = "qwen3.6-27b-autoround"
|
||||
prompts = [
|
||||
("narrative", "Write a concise paragraph explaining transformer attention.", 96),
|
||||
("code", "Write a small Python function that returns the nth Fibonacci number.", 96),
|
||||
("reasoning", "A train leaves at 08:00 traveling 60 km/h. Another leaves at 09:00 traveling 90 km/h. When does the second catch up?", 96),
|
||||
]
|
||||
|
||||
for label, prompt, max_tokens in prompts:
|
||||
body = json.dumps({
|
||||
"model": model,
|
||||
"messages": [{"role": "user", "content": prompt}],
|
||||
"max_tokens": max_tokens,
|
||||
"temperature": 0.3,
|
||||
"stream": False,
|
||||
"chat_template_kwargs": {"enable_thinking": False},
|
||||
}).encode()
|
||||
req = urllib.request.Request(
|
||||
f"{url}/v1/chat/completions",
|
||||
data=body,
|
||||
headers={"Content-Type": "application/json"},
|
||||
)
|
||||
start = time.time()
|
||||
with urllib.request.urlopen(req, timeout=300) as response:
|
||||
data = json.load(response)
|
||||
wall = time.time() - start
|
||||
usage = data.get("usage") or {}
|
||||
tokens = usage.get("completion_tokens") or 0
|
||||
text = data["choices"][0]["message"].get("content") or ""
|
||||
if not text.strip():
|
||||
raise SystemExit(f"{label}: empty completion")
|
||||
tps = tokens / wall if wall > 0 else 0
|
||||
print(f"{label}: wall={wall:.2f}s completion_tokens={tokens} wall_TPS={tps:.2f}")
|
||||
PY
|
||||
|
||||
- name: Promote latest aliases
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
docker tag "${IMAGE_REF}" "${IMAGE_NAME}:latest"
|
||||
docker tag "${IMAGE_REF}" "${IMAGE_NAME}:nightly-stable"
|
||||
docker push "${IMAGE_NAME}:latest"
|
||||
docker push "${IMAGE_NAME}:nightly-stable"
|
||||
|
||||
- name: Cleanup canonical compose
|
||||
if: always()
|
||||
shell: bash
|
||||
run: |
|
||||
docker compose \
|
||||
-f "${CANONICAL_COMPOSE}" \
|
||||
-f "${COMPOSE_OVERRIDE}" \
|
||||
-p "${COMPOSE_PROJECT_NAME}" \
|
||||
down --remove-orphans || true
|
||||
docker buildx imagetools create \
|
||||
-t "${IMAGE_NAME}:latest" \
|
||||
-t "${IMAGE_NAME}:nightly-stable" \
|
||||
"${IMAGE_REF}"
|
||||
|
||||
retention:
|
||||
name: Retain four weeks of dated nightlies
|
||||
|
||||
+67
-61
@@ -26,6 +26,12 @@ All `Narr / Code TPS` rows come from `bash scripts/bench.sh`, which runs:
|
||||
>
|
||||
> Sampling: `temperature=0.6, top_p=0.95, top_k=20, presence_penalty=0.0, enable_thinking=false`. Three warmups + five measured runs per prompt. Mean wall TPS reported.
|
||||
|
||||
`PP tok/s` is prompt-processing throughput. For vLLM rows, `bench.sh` scrapes
|
||||
the most recent `Avg prompt throughput` lines from container logs. For
|
||||
llama.cpp or engines without that log shape, run `PP=1 bash scripts/bench.sh`
|
||||
to add a single long-prompt fallback probe that computes prompt tokens over
|
||||
TTFT. Existing rows show `—` until re-benched with v0.7.1+ tooling.
|
||||
|
||||
Cross-rig numbers are comparable because the prompt + sampling are pinned. Variations against your rig usually trace back to power caps, PCIe lane counts, or pin (vLLM image SHA / Genesis commit) — see [`scripts/report.sh`](scripts/report.sh) which captures all three.
|
||||
|
||||
## How to add a row for your rig
|
||||
@@ -57,66 +63,66 @@ Primary serving model. Hybrid Qwen3-Next architecture (DeltaNet GDN + standard a
|
||||
|
||||
> ⚠️ **Cliff 2b open on `long-text*` / `long-vision` (2026-05-05)** — Genesis v7.72.2's PN59 streaming-GDN orchestrator doesn't engage on the chunked-prefill path 24 GB single-card configs are forced to take. Single-prompt prefill at >~50K may OOM. Filed at [Sandermage/genesis-vllm-patches#22](https://github.com/Sandermage/genesis-vllm-patches/issues/22). **Safe single-card paths**: `llamacpp/default` (no Cliff 2b) or single-prompt context capped at <50K. **TP=2 paths escape the cliff** entirely (see Dual-card section).
|
||||
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | Peak VRAM | Date | Notes |
|
||||
|---|---|---|---:|---:|---:|---|---|
|
||||
| `minimal.yml` (`mem-util 0.95 max-model-len 65536`) | @noonghunna (1× 3090, x16, 350 W) | TQ3 | 64K | ~32 / ~33 | ~22.4 GB | 2026-05-03 | no MTP. [stiggy2k16](https://github.com/noonghunna/club-3090/issues/43) cross-rig data point — short-prompt vLLM-safe path when llama.cpp is too slow. |
|
||||
| `long-vision.yml` | @noonghunna (1× 3090) | TQ3 | 145K | 50 / 66 | ~23.0 GB | 2026-04-30 | vision + tools + thinking. mem-util 0.95. |
|
||||
| `long-text.yml` ⭐ | @noonghunna (1× 3090) | TQ3 | 180K | 50 / 67 | ~22.3 GB | 2026-04-30 | text-only (vision tower dropped). MTP n=3. mem-util 0.93. **Default for RAG / IDE agents below 25K accumulated ctx**. |
|
||||
| `long-text.yml` | @laurimyllari (1× **4090**, AMD Ryzen 7 7800X3D, 230W cap) | TQ3 | **90K** (forced by KV-pool fit on 4090 — see Notes) | **102.96 / 103.09** | ~23.7 GB | 2026-05-05 | **First 4090 single-card vLLM bench** on club-3090. **Required `max-model-len` drop from 180K→90K** at default mem-util 0.92 (KV cache budget on his 24 GB 4090 is tighter than the 3090s the compose was calibrated against — likely 4090 driver/desktop overhead consumes more idle VRAM). MTP n=3 active, AL 3.34-3.45 narr / per-pos accept 92-95% / 79-84% / 62-67%. CV 2.2%/2.2%. Verify-stress hit Cliff 2b OOM at long-vision 50 MiB (sidesteps via long-text). [Issue #71](https://github.com/noonghunna/club-3090/issues/71) + [disc #62](https://github.com/noonghunna/club-3090/discussions/62#discussioncomment-16821619). |
|
||||
| `long-text-no-mtp.yml` | @noonghunna (1× 3090) | TQ3 | 200K | TBD | ~21.0 GB | — | max-context single-shot, no MTP. Slow decode but biggest ctx window. |
|
||||
| `bounded-thinking.yml` | @noonghunna (1× 3090) | TQ3 | 180K | 50 / 66 | ~21.7 GB | 2026-05-04 | structured-CoT FSM in reasoning channel; **recommended grammar: DeepSeek scratchpad** (PLAN/NOTE×0-15/VERDICT). Phase 3 final: **93.9% HE+ / 66.0% LCB v6** (87.4% combined, +1 net vs the andthattoo G/A/E baseline). Andthattoo G/A/E grammar also works (94.5% HE+ / 62.0% LCB / 86.9% combined, ~4× tighter think budget — pass via `extra_body`). See [STRUCTURED_COT.md](docs/STRUCTURED_COT.md). |
|
||||
| `tools-text.yml` | @noonghunna (1× 3090) | fp8 | 75K | TBD | TBD | — | IDE-agent path that escapes the long-text Cliff 1 mech B leak (see [#16](https://github.com/noonghunna/club-3090/issues/16)). |
|
||||
| `dual-dflash.yml`-shape forced TP=1 (DFlash N=5, fp8 KV, mem-util 0.96, custom_all_reduce disabled) | @efschu (1× **RTX 5090** 32 GB, AMD Ryzen 9 5950X, Debian trixie, PCIe x8, 575 W cap) | fp8 | 49K (KV-fit at 0.96 mem-util) | **126.53 / 200.11** (decode 127.98 / 204.80) | 31.5 GB | 2026-05-07 | **First single-5090 DFlash data point** on club-3090. AutoRound INT4 weights + DFlash N=5 draft. CV 3.0%/2.0%. **Code TPS 200 is the highest single-card number measured on the matrix** — beats single-3090 (50/67 long-text) by ~3× on code, single-4090 (102/103 at 90K) by ~2× code. Trade is ctx ceiling: 49K vs 90K-180K on 24 GB cards, due to KV-pool fit at fp8 + 32 GB total VRAM. vLLM `nightly-01d4d1ad3` (post-v7.72.2 uplift). [Issue #93](https://github.com/noonghunna/club-3090/issues/93). |
|
||||
| `vllm/default` (single, MAX_MODEL_LEN=48000, mem-util 0.92, MTP n=3) | @ygafarov (1× 3090 via **oculink eGPU on PCIe x4**, AMD Ryzen AI MAX+ 395 / Strix Halo miniPC, CachyOS, 124 GB RAM, 290W cap) | TQ3 | 48K | **68.86 / 91.70** (decode 69.27 / 92.76) | 23.6 GB | 2026-05-09 | **Soak: ⚠ borderline** (VRAM grew 240 MiB > 200 MiB threshold, 3 turns >30s; 100% TPS retention + 0 errors + 0 silent-empty turns — x4-PCIe accretion + bus-latency under prefill, not a leak. Threshold may need an "eGPU bus class" allowance.) **First Strix-Halo-miniPC + oculink-eGPU bench** on club-3090. Single 3090 over PCIe x4 (oculink) instead of x8/x16 internal. CV 1.6%/2.7%. MTP AL 3.31, accept 76.9% (per-pos 0.918/0.772/0.616). `scheduler_reserve_full_isl=False`. Driver 595.71.05 (very new, CUDA 13.2). vLLM `nightly-01d4d1ad3`. [Issue #113](https://github.com/noonghunna/club-3090/issues/113). |
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | PP tok/s | Peak VRAM | Date | Notes |
|
||||
| --- | --- | --- | ---: | ---: | ---: | ---: | --- | --- |
|
||||
| `minimal.yml` (`mem-util 0.95 max-model-len 65536`) | @noonghunna (1× 3090, x16, 350 W) | TQ3 | 64K | ~32 / ~33 | — | ~22.4 GB | 2026-05-03 | no MTP. [stiggy2k16](https://github.com/noonghunna/club-3090/issues/43) cross-rig data point — short-prompt vLLM-safe path when llama.cpp is too slow. |
|
||||
| `long-vision.yml` | @noonghunna (1× 3090) | TQ3 | 145K | 50 / 66 | — | ~23.0 GB | 2026-04-30 | vision + tools + thinking. mem-util 0.95. |
|
||||
| `long-text.yml` ⭐ | @noonghunna (1× 3090) | TQ3 | 180K | 50 / 67 | — | ~22.3 GB | 2026-04-30 | text-only (vision tower dropped). MTP n=3. mem-util 0.93. **Default for RAG / IDE agents below 25K accumulated ctx**. |
|
||||
| `long-text.yml` | @laurimyllari (1× **4090**, AMD Ryzen 7 7800X3D, 230W cap) | TQ3 | **90K** (forced by KV-pool fit on 4090 — see Notes) | **102.96 / 103.09** | — | ~23.7 GB | 2026-05-05 | **First 4090 single-card vLLM bench** on club-3090. **Required `max-model-len` drop from 180K→90K** at default mem-util 0.92 (KV cache budget on his 24 GB 4090 is tighter than the 3090s the compose was calibrated against — likely 4090 driver/desktop overhead consumes more idle VRAM). MTP n=3 active, AL 3.34-3.45 narr / per-pos accept 92-95% / 79-84% / 62-67%. CV 2.2%/2.2%. Verify-stress hit Cliff 2b OOM at long-vision 50 MiB (sidesteps via long-text). [Issue #71](https://github.com/noonghunna/club-3090/issues/71) + [disc #62](https://github.com/noonghunna/club-3090/discussions/62#discussioncomment-16821619). |
|
||||
| `long-text-no-mtp.yml` | @noonghunna (1× 3090) | TQ3 | 200K | TBD | — | ~21.0 GB | — | max-context single-shot, no MTP. Slow decode but biggest ctx window. |
|
||||
| `bounded-thinking.yml` | @noonghunna (1× 3090) | TQ3 | 180K | 50 / 66 | — | ~21.7 GB | 2026-05-04 | structured-CoT FSM in reasoning channel; **recommended grammar: DeepSeek scratchpad** (PLAN/NOTE×0-15/VERDICT). Phase 3 final: **93.9% HE+ / 66.0% LCB v6** (87.4% combined, +1 net vs the andthattoo G/A/E baseline). Andthattoo G/A/E grammar also works (94.5% HE+ / 62.0% LCB / 86.9% combined, ~4× tighter think budget — pass via `extra_body`). See [STRUCTURED_COT.md](docs/STRUCTURED_COT.md). |
|
||||
| `tools-text.yml` | @noonghunna (1× 3090) | fp8 | 75K | TBD | — | TBD | — | IDE-agent path that escapes the long-text Cliff 1 mech B leak (see [#16](https://github.com/noonghunna/club-3090/issues/16)). |
|
||||
| `dual-dflash.yml`-shape forced TP=1 (DFlash N=5, fp8 KV, mem-util 0.96, custom_all_reduce disabled) | @efschu (1× **RTX 5090** 32 GB, AMD Ryzen 9 5950X, Debian trixie, PCIe x8, 575 W cap) | fp8 | 49K (KV-fit at 0.96 mem-util) | **126.53 / 200.11** (decode 127.98 / 204.80) | — | 31.5 GB | 2026-05-07 | **First single-5090 DFlash data point** on club-3090. AutoRound INT4 weights + DFlash N=5 draft. CV 3.0%/2.0%. **Code TPS 200 is the highest single-card number measured on the matrix** — beats single-3090 (50/67 long-text) by ~3× on code, single-4090 (102/103 at 90K) by ~2× code. Trade is ctx ceiling: 49K vs 90K-180K on 24 GB cards, due to KV-pool fit at fp8 + 32 GB total VRAM. vLLM `nightly-01d4d1ad3` (post-v7.72.2 uplift). [Issue #93](https://github.com/noonghunna/club-3090/issues/93). |
|
||||
| `vllm/default` (single, MAX_MODEL_LEN=48000, mem-util 0.92, MTP n=3) | @ygafarov (1× 3090 via **oculink eGPU on PCIe x4**, AMD Ryzen AI MAX+ 395 / Strix Halo miniPC, CachyOS, 124 GB RAM, 290W cap) | TQ3 | 48K | **68.86 / 91.70** (decode 69.27 / 92.76) | — | 23.6 GB | 2026-05-09 | **Soak: ⚠ borderline** (VRAM grew 240 MiB > 200 MiB threshold, 3 turns >30s; 100% TPS retention + 0 errors + 0 silent-empty turns — x4-PCIe accretion + bus-latency under prefill, not a leak. Threshold may need an "eGPU bus class" allowance.) **First Strix-Halo-miniPC + oculink-eGPU bench** on club-3090. Single 3090 over PCIe x4 (oculink) instead of x8/x16 internal. CV 1.6%/2.7%. MTP AL 3.31, accept 76.9% (per-pos 0.918/0.772/0.616). `scheduler_reserve_full_isl=False`. Driver 595.71.05 (very new, CUDA 13.2). vLLM `nightly-01d4d1ad3`. [Issue #113](https://github.com/noonghunna/club-3090/issues/113). |
|
||||
|
||||
### Single-card (1× RTX 3090) — llama.cpp
|
||||
|
||||
| Compose | Rig | Quant | Max ctx | Narr / Code TPS | Peak VRAM | Date | Notes |
|
||||
|---|---|---|---:|---:|---:|---|---|
|
||||
| `llamacpp/default` | @noonghunna (1× 3090) | Unsloth Q5_K_XL | 262K | 21 / 21 | ~20 GB | 2026-04-21 | bulletproof — different engine, different memory allocator, no Cliff 1 / Cliff 2. Slow decode but cliff-immune. |
|
||||
| `llamacpp/concurrent` | @noonghunna (1× 3090) | Unsloth Q5_K_XL | 262K | TBD | TBD | — | concurrent-serving variant. |
|
||||
| llama.cpp PR [#22673](https://github.com/ggml-org/llama.cpp/pull/22673) MTP, custom build (`Qwen3.6-27B-MTP-Q4_K_M-GGUF` + `--spec-type mtp --spec-draft-n-max 3`) | @efschu (**2× Tesla V100-SXM2-16GB**, Xeon Gold 6154, Debian 13, custom-built llama-server docker) | Q4_K_M MTP | 100K | **49.96 / 62.46** | 15.6 GB/card (15,596 MiB at 100K ctx) | 2026-05-06 | **First V100 (sm_70 Volta) cross-rig data on the matrix** — only non-3090/4090/5090 GPU class tested. vLLM blocked (V100=CC 7.0, vLLM needs ≥7.5); fell back to llama.cpp via am17an's PR #22673 with a custom-built docker. **All 7 stress checks PASS including 90K NIAH** (Cliff 2 territory). 2× cards via tensor split (`-sm tensor`). MTP n=3, accept rates not in log. ~80 W/card (V100 max 300 W). [Issue #80](https://github.com/noonghunna/club-3090/issues/80). |
|
||||
| llama.cpp PR [#22673](https://github.com/ggml-org/llama.cpp/pull/22673) MTP, host build (`havenoammo/Qwen3.6-27B-MTP-UD-GGUF` + `--spec-type mtp --spec-draft-n-max 3` + q4_0 KV) | @lamentofhighborne (1× RTX 3090, PCIe x8, 350W) | UD-Q4_K_XL + Q8_0 MTP head | **131K** | **47.12 / 60.42** | ~23.1 GiB | 2026-05-07 | **First 1× 3090 llama.cpp MTP data point** on Qwen3.6-27B. Decode 47.60 / 61.71 TPS, TTFT 212 / 194 ms. **`verify-full-mtp.sh` PASS 8/8** (locally-adapted), **`verify-stress-mtp.sh` PASS 7/7 including 91K needle at 131K ctx** — pushes the documented llama.cpp MTP ctx ceiling from ~64-80K (q8_0 KV) to 131K (q4_0 KV). MTP acceptance 78.7%; recurrent 65-layer bug from froggeric's earlier MTP GGUF did **NOT** reproduce on havenoammo's UD GGUF. Native host build (no Docker), surfaced engine-coupling shortcomings in our verify/soak harness — see [Issue #85](https://github.com/noonghunna/club-3090/issues/85). |
|
||||
| llama.cpp PR [#22673](https://github.com/ggml-org/llama.cpp/pull/22673) MTP, host build (`froggeric/Qwen3.6-27B-MTP-GGUF` + `--spec-type mtp --spec-draft-n-max 3` + q4_0 KV) | @lamentofhighborne (1× RTX 3090, PCIe x8, 350 W) | Q4_K_M MTP | **164K** | **47.49 / 55.09** | ~22.2 GiB | 2026-05-07 | **Second 1× 3090 llama.cpp MTP data point on same rig** — froggeric's Q4_K_M MTP GGUF vs havenoammo's UD-Q4_K_XL above. Decode 47.91 / 55.81 TPS, TTFT 96 / 98 ms. `verify-full-mtp.sh` PASS 8/8, `verify-stress-mtp.sh` PASS 7/7 incl. 91K needle at 164K ctx. Functional MTP acceptance **86.7%**; canonical acceptance 55.3% narr / 71.2% code. **Ctx-fit ladder**: 262K OOMed MTP, 229K served without MTP, 196K initialized MTP but daemon died at 90K stress; 164K was the stable stress-passing ceiling on this rig. **Beats havenoammo on narr (47.49 vs 47.12, +0.8%) and ctx ceiling (164K vs 131K) but trails on code (55.09 vs 60.42, −9%)**. Manual long-context needles also passed at **120K** (39.39 decode TPS, 81% MTP accept) and **150K** (35.44 decode TPS, 80% MTP accept). MTP+vision incompat (per froggeric's model card); separate no-MTP+vision path passed 65K and 150K. [Issue #94](https://github.com/noonghunna/club-3090/issues/94). |
|
||||
| Compose | Rig | Quant | Max ctx | Narr / Code TPS | PP tok/s | Peak VRAM | Date | Notes |
|
||||
| --- | --- | --- | ---: | ---: | ---: | ---: | --- | --- |
|
||||
| `llamacpp/default` | @noonghunna (1× 3090) | Unsloth Q5_K_XL | 262K | 21 / 21 | — | ~20 GB | 2026-04-21 | bulletproof — different engine, different memory allocator, no Cliff 1 / Cliff 2. Slow decode but cliff-immune. |
|
||||
| `llamacpp/concurrent` | @noonghunna (1× 3090) | Unsloth Q5_K_XL | 262K | TBD | — | TBD | — | concurrent-serving variant. |
|
||||
| llama.cpp PR [#22673](https://github.com/ggml-org/llama.cpp/pull/22673) MTP, custom build (`Qwen3.6-27B-MTP-Q4_K_M-GGUF` + `--spec-type mtp --spec-draft-n-max 3`) | @efschu (**2× Tesla V100-SXM2-16GB**, Xeon Gold 6154, Debian 13, custom-built llama-server docker) | Q4_K_M MTP | 100K | **49.96 / 62.46** | — | 15.6 GB/card (15,596 MiB at 100K ctx) | 2026-05-06 | **First V100 (sm_70 Volta) cross-rig data on the matrix** — only non-3090/4090/5090 GPU class tested. vLLM blocked (V100=CC 7.0, vLLM needs ≥7.5); fell back to llama.cpp via am17an's PR #22673 with a custom-built docker. **All 7 stress checks PASS including 90K NIAH** (Cliff 2 territory). 2× cards via tensor split (`-sm tensor`). MTP n=3, accept rates not in log. ~80 W/card (V100 max 300 W). [Issue #80](https://github.com/noonghunna/club-3090/issues/80). |
|
||||
| llama.cpp PR [#22673](https://github.com/ggml-org/llama.cpp/pull/22673) MTP, host build (`havenoammo/Qwen3.6-27B-MTP-UD-GGUF` + `--spec-type mtp --spec-draft-n-max 3` + q4_0 KV) | @lamentofhighborne (1× RTX 3090, PCIe x8, 350W) | UD-Q4_K_XL + Q8_0 MTP head | **131K** | **47.12 / 60.42** | — | ~23.1 GiB | 2026-05-07 | **First 1× 3090 llama.cpp MTP data point** on Qwen3.6-27B. Decode 47.60 / 61.71 TPS, TTFT 212 / 194 ms. **`verify-full-mtp.sh` PASS 8/8** (locally-adapted), **`verify-stress-mtp.sh` PASS 7/7 including 91K needle at 131K ctx** — pushes the documented llama.cpp MTP ctx ceiling from ~64-80K (q8_0 KV) to 131K (q4_0 KV). MTP acceptance 78.7%; recurrent 65-layer bug from froggeric's earlier MTP GGUF did **NOT** reproduce on havenoammo's UD GGUF. Native host build (no Docker), surfaced engine-coupling shortcomings in our verify/soak harness — see [Issue #85](https://github.com/noonghunna/club-3090/issues/85). |
|
||||
| llama.cpp PR [#22673](https://github.com/ggml-org/llama.cpp/pull/22673) MTP, host build (`froggeric/Qwen3.6-27B-MTP-GGUF` + `--spec-type mtp --spec-draft-n-max 3` + q4_0 KV) | @lamentofhighborne (1× RTX 3090, PCIe x8, 350 W) | Q4_K_M MTP | **164K** | **47.49 / 55.09** | — | ~22.2 GiB | 2026-05-07 | **Second 1× 3090 llama.cpp MTP data point on same rig** — froggeric's Q4_K_M MTP GGUF vs havenoammo's UD-Q4_K_XL above. Decode 47.91 / 55.81 TPS, TTFT 96 / 98 ms. `verify-full-mtp.sh` PASS 8/8, `verify-stress-mtp.sh` PASS 7/7 incl. 91K needle at 164K ctx. Functional MTP acceptance **86.7%**; canonical acceptance 55.3% narr / 71.2% code. **Ctx-fit ladder**: 262K OOMed MTP, 229K served without MTP, 196K initialized MTP but daemon died at 90K stress; 164K was the stable stress-passing ceiling on this rig. **Beats havenoammo on narr (47.49 vs 47.12, +0.8%) and ctx ceiling (164K vs 131K) but trails on code (55.09 vs 60.42, −9%)**. Manual long-context needles also passed at **120K** (39.39 decode TPS, 81% MTP accept) and **150K** (35.44 decode TPS, 80% MTP accept). MTP+vision incompat (per froggeric's model card); separate no-MTP+vision path passed 65K and 150K. [Issue #94](https://github.com/noonghunna/club-3090/issues/94). |
|
||||
|
||||
### Dual-card (2× RTX 3090, TP=2)
|
||||
|
||||
> NVLink auto-detection: dual-card composes now detect NVLink presence automatically. The `dual-nvlink*.yml` files are deprecated stubs that extend the unified compose with `NVLINK_MODE=force_on`. All NVLink bench rows below were measured with NVLink enabled (either via auto-detection or the deprecated stub). PCIe rows used `NCCL_P2P_DISABLE=1`.
|
||||
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | Peak VRAM | Date | Notes |
|
||||
|---|---|---|---:|---:|---:|---|---|
|
||||
| `dual.yml` ⭐ | @noonghunna (2× 3090 PCIe, no NVLink) | fp8 | 262K (237K single-prompt verified) | 69 / 89 | ~23.6 GB | 2026-04-29 | tested 2-card baseline. fp8 KV, 2 streams, full feature set. **PASSES v2 continuous soak** (Cliff 2b clean). |
|
||||
| `dual-turbo.yml` | @noonghunna (2× 3090 PCIe) | TQ3 | 262K | 58 / 76 per-stream (**269 TPS aggregate at 4 streams**) | ~19.8 GB | 2026-04-29 | TQ3 KV — 4.67× concurrency for multi-tenant agent workloads. |
|
||||
| `dual-turbo.yml` ⭐ | @noonghunna (2× 3090 PCIe) | TQ3 | 262K | **81.21 / 108.20** single-stream | **20.0 GB** | 2026-05-05 | **v7.72.2 uplift**: Genesis pin `7b9fd319` + vLLM `01d4d1ad3` (Sander's PROD pin). 6 redundant local sidecars dropped (PN35/PN30/PN25/P78/PN34 supersede). 5 measured runs each, CV 2.3%/0.9%. AL 3.46. **VRAM −2.1 GB/card vs v7.69 baseline** (PN35 native + PN59 fold value). All 8/8 verify-full checks pass. |
|
||||
| `dual-dflash.yml` | @noonghunna (2× 3090 PCIe) | fp8 | 185K | 82 / **125** | ~23.6 GB | 2026-04-29 | DFlash N=5 + 1.75 GB draft / card. AL ~4.4. Fastest 2-card short-prompt code path. |
|
||||
| `dual-dflash.yml` | @apriori (2× 3090 + EPYC 7302P, Arch Linux, 230 W cap, NODE topology, no NVLink) | fp8 | 185K | **78.44 / 122.71** | ~24.0 GB | 2026-05-05 | **First EPYC + Arch cross-rig data on `dual-dflash`** — matches @noonghunna baseline within run-to-run CV (78/127 reference, narr drift +0.4 / code −3.4%). **PASSES continuous soak** (0 MiB VRAM growth, 0 errors, 0/25 silent-empty, 100% TPS retention) — first independent confirmation `dual-dflash` is Cliff 2b clean cross-rig. 3 turns >30s TTFT warning (informational). [Discussion #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16819551). |
|
||||
| `dual-dflash-noviz.yml` | @noonghunna (2× 3090 PCIe) | fp8 | 200K | 78 / **127** | ~23.8 GB | 2026-04-29 | DFlash + no vision tower. +15K ctx vs `dual-dflash`. |
|
||||
| `dual-dflash-noviz.yml` | @snoby (2× **4090** PCIe — 5-GPU rig, GPUs 2,3, no NVLink, [#46](https://github.com/noonghunna/club-3090/issues/46)) | fp8 | **180K** | 92.55 / **148.99** | ~21.8 GB | 2026-05-04 | First non-3090 cross-rig data. **Required `max-model-len` drop from 200K→180K** vs 3090 baseline (boot OOM at 200K) — 4090 ctx-ceiling gotcha pending investigation. +17% TPS lift vs same compose on 3090 (78→92.55 narr / 127→148.99 code). |
|
||||
| `dual-nvlink.yml` | @JusefPol (2× 3090 PCIe x8 + **NVLink 4× bonded**, i7-11700K, 365 W/card) | fp8 | 262K | **108.81 / 138.55** | ~23.7 GB | 2026-05-04 | First NVLink cross-rig data. **+58% narr / +56% code TPS vs `dual.yml` PCIe-only baseline (69 / 89)** — NVLink reduces the per-token NCCL allreduce latency floor; compounds at multi-stream. verify-stress 7/7 PASS incl. 91K needle. **PASSES v2 continuous soak** (5 sessions × 5 turns, 0 MiB growth, 100% TPS retention). MTP n=3, 65–98% per-position accept. PR [#31](https://github.com/noonghunna/club-3090/pull/31). |
|
||||
| `dual-nvlink-turbo.yml` ⭐ | @danbedford (2× 3090 NVLink, 230W cap) | TQ3 | 262K | **102.34 / 133.98** | ~22.3 GB | 2026-05-05 | **v7.72.2-rebench** (image `nightly-01d4d1ad3`). 4-stream TurboQuant KV + NVLink. **+11% narr / +12% code vs same-rig PCIe `dual-turbo` (#73 below)** — controlled A/B on identical hardware, only `NCCL_P2P_LEVEL` differs. Custom all-reduce ENABLED (disabled on PCIe). CV 3.1% narr / 1.8% code. PR [#56](https://github.com/noonghunna/club-3090/pull/56) + [Issue #69](https://github.com/noonghunna/club-3090/issues/69). |
|
||||
| `dual.yml` | @danbedford (2× 3090 NVLink-cable-attached, run as PCIe via `NCCL_P2P_DISABLE=1`, 230W cap) | fp8 | 262K | **89.24 / 114.57** | ~23.7 GB | 2026-05-06 | **First controlled PCIe-vs-NVLink A/B on same rig** — pair with `dual-nvlink.yml` row immediately above. **+15% narr / +15% code lift from NVLink** (#74 102/132 vs this 89/115). CV 3.8%/2.5%. **Note: this corrects the "+58% narr / +56% code" claim from JusefPol's row** — that comparison conflated NVLink lift with v7.72.2 lift (his baseline was 2026-04-29 dual.yml at 69/89 on the older image). On a strictly v7.72.2-controlled comparison NVLink adds ~15%, not ~58%. [Issue #77](https://github.com/noonghunna/club-3090/issues/77). |
|
||||
| `dual-turbo.yml` | @danbedford (2× 3090 NVLink-cable-attached, run as PCIe via `NCCL_P2P_DISABLE=1`, 230W cap) | TQ3 | 262K | 91.58 / 120.00 | ~22.0 GB | 2026-05-06 | Companion to `dual-nvlink-turbo` row above for the controlled A/B. NVLink lift on TQ3 path: **+11% / +12%**. CV 3.2%/1.9%. [Issue #73](https://github.com/noonghunna/club-3090/issues/73). |
|
||||
| `dual-nvlink.yml` | @danbedford (2× 3090 NVLink, 230W cap) | fp8 | 262K | **102.09 / 131.59** | ~24.0 GB | 2026-05-06 | Second cross-rig data on `dual-nvlink.yml` (vs JusefPol's earlier 108.81/138.55). Lower than JusefPol partly explained by his lower power cap (365 W/card vs 230) — on memory-bandwidth-bound decode, 2 GB/card more thermal headroom doesn't compound much, so close-but-lower at half the wattage is consistent. CV 2.6%/1.4%. [Issue #74](https://github.com/noonghunna/club-3090/issues/74). |
|
||||
| `dual-dflash.yml` | @danbedford (2× 3090 PCIe NVLink-cable-attached but `NCCL_P2P_DISABLE=1`, 230W cap) | FP16 | 185K | 86.62 / **141.02** | ~24.0 GB | 2026-05-06 | Third cross-rig DFlash data point (after @noonghunna 82/125 + @lolren 87/142). **Code TPS 141 ties lolren's 142** as the highest measured on club-3090. CV 2.4%/5.0%. [Issue #75](https://github.com/noonghunna/club-3090/issues/75). |
|
||||
| `dual-dflash-noviz.yml` | @danbedford (2× 3090 PCIe NVLink-cable-attached but `NCCL_P2P_DISABLE=1`, 230W cap) | FP16 | 200K | 88.31 / **142.79** | ~23.9 GB | 2026-05-06 | DFlash + no vision tower. Beats @noonghunna baseline 78/127 (+13%/+12%). CV 2.3%/2.9%. [Issue #76](https://github.com/noonghunna/club-3090/issues/76). |
|
||||
| `dual-nvlink-dflash.yml` ⭐ NEW | @danbedford (2× 3090 NVLink, 230W cap, i9-11900KF) | FP16 | 185K | **101.55 / 163.33** | 24.06 GB/card | 2026-05-07 | **First NVLink-enabled DFlash row.** Mirrors `dual-dflash.yml` shape but enables NCCL P2P over NVLink + custom_all_reduce. **+17% narr / +16% code over his own PCIe `dual-dflash` row above** (86.62 / 141.02 — same rig with `NCCL_P2P_DISABLE=1`). Decode 102.43 / 166.54 TPS, CV 1.8%/1.9%. **PASSES continuous soak** (0 errors, 0 silent-empty, 0 MiB growth, 100% TPS retention, p50 66.71). verify-full 8/8 + verify-stress 7/7 incl. 91K Cliff 2 needle. PR [#92](https://github.com/noonghunna/club-3090/pull/92). |
|
||||
| `dual-nvlink-dflash-noviz.yml` ⭐ NEW | @danbedford (2× 3090 NVLink, 230W cap) | FP16 | **188K** | **103.24 / 167.45** | ~23.97 GB/card | 2026-05-07 | **NVLink + DFlash + no vision** — pushes the with-vision 185K ctx ceiling to **188K** by dropping MoonViT (~0.78 GB freed). Empirically determined: 189K had only 1/3 success rate (flaky on freshly rebooted system), 188K is the stable ceiling. **+17% narr / +17% code over his own PCIe `dual-dflash-noviz` row above** (88.31 / 142.79). Decode 104.07 / 171.01 TPS, CV 2.2%/3.6%. **PASSES continuous soak** (p50 66.75, 100% retention). verify-full 8/8 + verify-stress 7/7. PR [#96](https://github.com/noonghunna/club-3090/pull/96). |
|
||||
| `dual.yml`-shape **+ patched P2P drivers** (no NVLink hardware) | @aaronlockhartdev (2× 3090 PCIe x16, EPYC 7F52, Arch Linux, custom Dockerfile via [Sam McLeod's guide](https://smcleod.net/2026/02/patching-nvidias-driver-and-vllm-to-enable-p2p-on-consumer-gpus/) — patched `aikitoria/open-gpu-kernel-modules` + vLLM `cuda.py` `return True` patch) | fp8 | 262K | **93 / 125** | n/a | 2026-05-07 | **First patched-driver P2P cross-rig data point** — answers the question raised in [disc #70](https://github.com/noonghunna/club-3090/discussions/70). Same-rig controlled A/B: unpatched baseline 91 narr / 114 code → patched P2P 93 / 125 = **+2% narr / +9% code**. Compared to NVLink hardware lift (+15% / +15% per @danbedford's controlled A/B): patched P2P captures **~60% of NVLink's code gain but ~13% of NVLink's narr gain** — code workloads (spec-decode K+1 verify is heavily cross-card matmul) benefit more from cross-card bandwidth than narr decode (more sequential per-token). For ~95% of dual-3090 owners without NVLink, the trade is small TPS lift vs custom kernel module + DKMS maintenance burden. [Issue #91](https://github.com/noonghunna/club-3090/issues/91). |
|
||||
| `dual-dflash-noviz.yml`-shape **+ patched P2P drivers** (no NVLink hardware, custom_all_reduce ENABLED) | @aaronlockhartdev (2× 3090 PCIe x16, EPYC 7F52, Arch Linux, patched kernel module + `NCCL_P2P_LEVEL=PHB`) | fp8 | 200K | **100.47 / 160.15** (decode 101.53 / 164.44) | ~22.2 GB/card | 2026-05-07 | **Second patched-P2P cross-rig data point** — extends [#91 dual.yml result](https://github.com/noonghunna/club-3090/issues/91) to the DFlash + no-vision path. Same-rig controlled A/B: unpatched baseline 82.55 narr / 134.45 code → patched P2P 100.47 / 160.15 = **+22% narr / +19% code**. **Significantly larger lift than `dual.yml`-shape** (+22%/+19% here vs +2%/+9% on `dual.yml`) — DFlash's K+1 cross-card verify pattern stresses peer-bandwidth more than fp8-only `dual.yml`. **Important methodology update**: `NCCL_P2P_LEVEL=PHB` alone with the default vLLM image produced the same lift as the full vLLM `cuda.py` patch — **the in-container vLLM source patch is unnecessary**, only the kernel module patch matters. CV 4.6%/2.4%. custom_all_reduce ENABLED (vs disabled on the `dual.yml` row). [Issue #95](https://github.com/noonghunna/club-3090/issues/95) + [disc #70](https://github.com/noonghunna/club-3090/discussions/70). |
|
||||
| `carnice-bf16mtp.yml` | @noonghunna (2× 3090 PCIe, no NVLink) | fp8 | 262K | **72** / **80** | ~22.25 GB | 2026-05-04 | **Carnice-V2-27B (Hermes agentic fine-tune) + BF16 MTP overlay**. Full 262K context, 2 streams. 71.75 narr / 80.35 code wall TPS (n=5 each, CV ~11%), MTP AL 3.02-3.14, TTFT 141ms. Patched chat template for Hermes JSON tool calls. verify-full 7/8 PASS. soak PASS. |
|
||||
| `dual.yml` ⭐ | @lolren (2× 3090 PCIe + Ryzen 9 5950X, **250W/card cap**) | fp8 | 262K | **89.78 / 117.60** | ~22.3 GB | 2026-05-05 | **First cross-rig data on the v7.72.2 uplift** (image `nightly-01d4d1ad3`, post-PR #59). +30% narr / +32% code over @noonghunna 2026-04-29 baseline (69/89 on older image) — confirms the v7.72.2 dividend cross-rig. CV 3.3%/2.0%. MTP AL ~3.5, per-pos accept 94/84/72%. [Disc #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16820303). |
|
||||
| `dual-dflash.yml` | @lolren (2× 3090 PCIe + Ryzen 9 5950X, 250W cap) | FP16 | 185K | 87.10 / **142.0** | ~22.1 GB | 2026-05-05 | Older image `nightly-7a1eb8ac2`. +6% narr / +14% code over @noonghunna baseline (82/125) — likely Ryzen 5950X advantage on prefill. DFlash AL ~4.5, per-pos accept 93/81/68/56/48%, avg accept 69%. [Disc #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16820303). |
|
||||
| `bounded-thinking.yml` | @lolren (2× 3090 PCIe + Ryzen 9 5950X, 250W cap, **MTP-disabled-suspected**) | TQ3 | 180K | 64.86 / 64.96 (CV **0.1%**) | ~22.3 GB | 2026-05-05 | **Anomaly:** lolren reports "no spec-decode" on this run despite `bounded-thinking.yml` shipping `--speculative-config mtp n=3` by default. Near-identical narr=code TPS + extreme CV stability (0.1%) suggests MTP was inactive — likely because his image was older `nightly-7a1eb8ac2` (pre-v7.72.2 + pre-PN35). Re-test on `nightly-01d4d1ad3` should restore MTP path → expect ~50/66 narr/code with normal CV. Tracked. [Disc #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16820303). |
|
||||
| `dual.yml` | @JDWarner (**Mixed RTX A5000 + RTX 3090**, both **Razer Core X eGPU enclosures over Thunderbolt 3**, Intel NUC11TNH i5-1135G7, **16 GB RAM**, headless, A5000=230W cap / 3090=290W cap, PCIe **x4 Gen 3** per card) | fp8 | 262K | **56.83 / 72.47** (soak p50 93.09) | ~23.6 GB/card | 2026-05-09 | **Soak: ✓ PASS** (5×5, 0 errors, 0 silent-empty, 100% TPS retention, 0 MiB growth). **First TB3 dual-eGPU + mixed-arch cross-rig data**. The setup that "shouldn't work": each card on a separate TB3 controller → ~3.94 GB/s effective per card vs ~32 GB/s on PCIe x16 Gen 4 (~8× cut), mixed Ampere SKUs (workstation A5000 + consumer 3090 with different mem bandwidth + clocks), 16 GB system RAM total. **Result: matches `dual.yml` PCIe x16 baseline within run-to-run noise** — confirms decode on Qwen3.6-27B is per-card-bandwidth bound, cross-card NCCL allreduce is small enough that even an 8× link cut doesn't dominate. Extends @aaronlockhartdev's #91/#95 finding (patched-P2P only +2%/+9% on `dual.yml`) in the opposite direction: even with 8× *less* cross-card bandwidth, decode holds. MTP AL 3.39-3.52, per-pos accept 0.93/0.83/0.70 (89% avg). verify-full + verify-stress all PASS. Genesis pin `7b9fd319` (v7.72.2). [Issue #107](https://github.com/noonghunna/club-3090/issues/107). |
|
||||
| `dual/docker-compose.yml` (default) | @ygafarov (**3090 via USB4 eGPU dock + 5070 Ti via OCuLink** — heterogeneous Ampere + Blackwell consumer dual-eGPU, AMD Ryzen AI MAX+ 395 / Strix Halo miniPC, CachyOS, 123 GB RAM, 290 W cap both cards, PCIe **x4** per card — USB4 ≈ 3.94 GB/s, OCuLink ≈ 7.88 GB/s) | fp8 | 200K | **65.10 / 85.81** | 17.1 / 15.7 GB | 2026-05-12 | **First heterogeneous Ampere + Blackwell consumer dual-eGPU on the matrix.** TP=2 bound by the slower USB4 link in allreduce + sm_86 kernels (5070 Ti spends back-half of step waiting — 91% util but only 125 W out of 290 W cap). KV pool 200K @ 1.00× concurrency — VRAM cap from the 5070 Ti's 16 GiB (model takes 13.8 GiB/card → only ~2.2 GiB left for KV on the smaller card). verify-stress 7/7 incl. **91K needle recall** (Cliff 2 clean). Soak ⚠ borderline (360 MiB > 200 MiB threshold — same eGPU-bus accretion as ygafarov's own #113 single-card row above at 240 MiB; 100% TPS retention + 0 silent-empty + 0 errors so not a leak). MTP AL 3.50, per-pos accept 0.94/0.86/0.70. CV 4.5%/1.8%. **Slower than ygafarov's own single-3090 #113 row** (68.86/91.70 at 48K) — on this rig the single-card path is recommended; the 5070 Ti adds VRAM cap pain without TPS gain. Driver 595.71.05, vLLM `nightly-1acd67a79`, no Genesis (Blackwell consumer not on allowlist). [Issue #120](https://github.com/noonghunna/club-3090/issues/120). |
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | PP tok/s | Peak VRAM | Date | Notes |
|
||||
| --- | --- | --- | ---: | ---: | ---: | ---: | --- | --- |
|
||||
| `dual.yml` ⭐ | @noonghunna (2× 3090 PCIe, no NVLink) | fp8 | 262K (237K single-prompt verified) | 69 / 89 | — | ~23.6 GB | 2026-04-29 | tested 2-card baseline. fp8 KV, 2 streams, full feature set. **PASSES v2 continuous soak** (Cliff 2b clean). |
|
||||
| `dual-turbo.yml` | @noonghunna (2× 3090 PCIe) | TQ3 | 262K | 58 / 76 per-stream (**269 TPS aggregate at 4 streams**) | — | ~19.8 GB | 2026-04-29 | TQ3 KV — 4.67× concurrency for multi-tenant agent workloads. |
|
||||
| `dual-turbo.yml` ⭐ | @noonghunna (2× 3090 PCIe) | TQ3 | 262K | **81.21 / 108.20** single-stream | — | **20.0 GB** | 2026-05-05 | **v7.72.2 uplift**: Genesis pin `7b9fd319` + vLLM `01d4d1ad3` (Sander's PROD pin). 6 redundant local sidecars dropped (PN35/PN30/PN25/P78/PN34 supersede). 5 measured runs each, CV 2.3%/0.9%. AL 3.46. **VRAM −2.1 GB/card vs v7.69 baseline** (PN35 native + PN59 fold value). All 8/8 verify-full checks pass. |
|
||||
| `dual-dflash.yml` | @noonghunna (2× 3090 PCIe) | fp8 | 185K | 82 / **125** | — | ~23.6 GB | 2026-04-29 | DFlash N=5 + 1.75 GB draft / card. AL ~4.4. Fastest 2-card short-prompt code path. |
|
||||
| `dual-dflash.yml` | @apriori (2× 3090 + EPYC 7302P, Arch Linux, 230 W cap, NODE topology, no NVLink) | fp8 | 185K | **78.44 / 122.71** | — | ~24.0 GB | 2026-05-05 | **First EPYC + Arch cross-rig data on `dual-dflash`** — matches @noonghunna baseline within run-to-run CV (78/127 reference, narr drift +0.4 / code −3.4%). **PASSES continuous soak** (0 MiB VRAM growth, 0 errors, 0/25 silent-empty, 100% TPS retention) — first independent confirmation `dual-dflash` is Cliff 2b clean cross-rig. 3 turns >30s TTFT warning (informational). [Discussion #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16819551). |
|
||||
| `dual-dflash-noviz.yml` | @noonghunna (2× 3090 PCIe) | fp8 | 200K | 78 / **127** | — | ~23.8 GB | 2026-04-29 | DFlash + no vision tower. +15K ctx vs `dual-dflash`. |
|
||||
| `dual-dflash-noviz.yml` | @snoby (2× **4090** PCIe — 5-GPU rig, GPUs 2,3, no NVLink, [#46](https://github.com/noonghunna/club-3090/issues/46)) | fp8 | **180K** | 92.55 / **148.99** | — | ~21.8 GB | 2026-05-04 | First non-3090 cross-rig data. **Required `max-model-len` drop from 200K→180K** vs 3090 baseline (boot OOM at 200K) — 4090 ctx-ceiling gotcha pending investigation. +17% TPS lift vs same compose on 3090 (78→92.55 narr / 127→148.99 code). |
|
||||
| `dual-nvlink.yml` | @JusefPol (2× 3090 PCIe x8 + **NVLink 4× bonded**, i7-11700K, 365 W/card) | fp8 | 262K | **108.81 / 138.55** | — | ~23.7 GB | 2026-05-04 | First NVLink cross-rig data. **+58% narr / +56% code TPS vs `dual.yml` PCIe-only baseline (69 / 89)** — NVLink reduces the per-token NCCL allreduce latency floor; compounds at multi-stream. verify-stress 7/7 PASS incl. 91K needle. **PASSES v2 continuous soak** (5 sessions × 5 turns, 0 MiB growth, 100% TPS retention). MTP n=3, 65–98% per-position accept. PR [#31](https://github.com/noonghunna/club-3090/pull/31). |
|
||||
| `dual-nvlink-turbo.yml` ⭐ | @danbedford (2× 3090 NVLink, 230W cap) | TQ3 | 262K | **102.34 / 133.98** | — | ~22.3 GB | 2026-05-05 | **v7.72.2-rebench** (image `nightly-01d4d1ad3`). 4-stream TurboQuant KV + NVLink. **+11% narr / +12% code vs same-rig PCIe `dual-turbo` (#73 below)** — controlled A/B on identical hardware, only `NCCL_P2P_LEVEL` differs. Custom all-reduce ENABLED (disabled on PCIe). CV 3.1% narr / 1.8% code. PR [#56](https://github.com/noonghunna/club-3090/pull/56) + [Issue #69](https://github.com/noonghunna/club-3090/issues/69). |
|
||||
| `dual.yml` | @danbedford (2× 3090 NVLink-cable-attached, run as PCIe via `NCCL_P2P_DISABLE=1`, 230W cap) | fp8 | 262K | **89.24 / 114.57** | — | ~23.7 GB | 2026-05-06 | **First controlled PCIe-vs-NVLink A/B on same rig** — pair with `dual-nvlink.yml` row immediately above. **+15% narr / +15% code lift from NVLink** (#74 102/132 vs this 89/115). CV 3.8%/2.5%. **Note: this corrects the "+58% narr / +56% code" claim from JusefPol's row** — that comparison conflated NVLink lift with v7.72.2 lift (his baseline was 2026-04-29 dual.yml at 69/89 on the older image). On a strictly v7.72.2-controlled comparison NVLink adds ~15%, not ~58%. [Issue #77](https://github.com/noonghunna/club-3090/issues/77). |
|
||||
| `dual-turbo.yml` | @danbedford (2× 3090 NVLink-cable-attached, run as PCIe via `NCCL_P2P_DISABLE=1`, 230W cap) | TQ3 | 262K | 91.58 / 120.00 | — | ~22.0 GB | 2026-05-06 | Companion to `dual-nvlink-turbo` row above for the controlled A/B. NVLink lift on TQ3 path: **+11% / +12%**. CV 3.2%/1.9%. [Issue #73](https://github.com/noonghunna/club-3090/issues/73). |
|
||||
| `dual-nvlink.yml` | @danbedford (2× 3090 NVLink, 230W cap) | fp8 | 262K | **102.09 / 131.59** | — | ~24.0 GB | 2026-05-06 | Second cross-rig data on `dual-nvlink.yml` (vs JusefPol's earlier 108.81/138.55). Lower than JusefPol partly explained by his lower power cap (365 W/card vs 230) — on memory-bandwidth-bound decode, 2 GB/card more thermal headroom doesn't compound much, so close-but-lower at half the wattage is consistent. CV 2.6%/1.4%. [Issue #74](https://github.com/noonghunna/club-3090/issues/74). |
|
||||
| `dual-dflash.yml` | @danbedford (2× 3090 PCIe NVLink-cable-attached but `NCCL_P2P_DISABLE=1`, 230W cap) | FP16 | 185K | 86.62 / **141.02** | — | ~24.0 GB | 2026-05-06 | Third cross-rig DFlash data point (after @noonghunna 82/125 + @lolren 87/142). **Code TPS 141 ties lolren's 142** as the highest measured on club-3090. CV 2.4%/5.0%. [Issue #75](https://github.com/noonghunna/club-3090/issues/75). |
|
||||
| `dual-dflash-noviz.yml` | @danbedford (2× 3090 PCIe NVLink-cable-attached but `NCCL_P2P_DISABLE=1`, 230W cap) | FP16 | 200K | 88.31 / **142.79** | — | ~23.9 GB | 2026-05-06 | DFlash + no vision tower. Beats @noonghunna baseline 78/127 (+13%/+12%). CV 2.3%/2.9%. [Issue #76](https://github.com/noonghunna/club-3090/issues/76). |
|
||||
| `dual-nvlink-dflash.yml` ⭐ NEW | @danbedford (2× 3090 NVLink, 230W cap, i9-11900KF) | FP16 | 185K | **101.55 / 163.33** | — | 24.06 GB/card | 2026-05-07 | **First NVLink-enabled DFlash row.** Mirrors `dual-dflash.yml` shape but enables NCCL P2P over NVLink + custom_all_reduce. **+17% narr / +16% code over his own PCIe `dual-dflash` row above** (86.62 / 141.02 — same rig with `NCCL_P2P_DISABLE=1`). Decode 102.43 / 166.54 TPS, CV 1.8%/1.9%. **PASSES continuous soak** (0 errors, 0 silent-empty, 0 MiB growth, 100% TPS retention, p50 66.71). verify-full 8/8 + verify-stress 7/7 incl. 91K Cliff 2 needle. PR [#92](https://github.com/noonghunna/club-3090/pull/92). |
|
||||
| `dual-nvlink-dflash-noviz.yml` ⭐ NEW | @danbedford (2× 3090 NVLink, 230W cap) | FP16 | **188K** | **103.24 / 167.45** | — | ~23.97 GB/card | 2026-05-07 | **NVLink + DFlash + no vision** — pushes the with-vision 185K ctx ceiling to **188K** by dropping MoonViT (~0.78 GB freed). Empirically determined: 189K had only 1/3 success rate (flaky on freshly rebooted system), 188K is the stable ceiling. **+17% narr / +17% code over his own PCIe `dual-dflash-noviz` row above** (88.31 / 142.79). Decode 104.07 / 171.01 TPS, CV 2.2%/3.6%. **PASSES continuous soak** (p50 66.75, 100% retention). verify-full 8/8 + verify-stress 7/7. PR [#96](https://github.com/noonghunna/club-3090/pull/96). |
|
||||
| `dual.yml`-shape **+ patched P2P drivers** (no NVLink hardware) | @aaronlockhartdev (2× 3090 PCIe x16, EPYC 7F52, Arch Linux, custom Dockerfile via [Sam McLeod's guide](https://smcleod.net/2026/02/patching-nvidias-driver-and-vllm-to-enable-p2p-on-consumer-gpus/) — patched `aikitoria/open-gpu-kernel-modules` + vLLM `cuda.py` `return True` patch) | fp8 | 262K | **93 / 125** | — | n/a | 2026-05-07 | **First patched-driver P2P cross-rig data point** — answers the question raised in [disc #70](https://github.com/noonghunna/club-3090/discussions/70). Same-rig controlled A/B: unpatched baseline 91 narr / 114 code → patched P2P 93 / 125 = **+2% narr / +9% code**. Compared to NVLink hardware lift (+15% / +15% per @danbedford's controlled A/B): patched P2P captures **~60% of NVLink's code gain but ~13% of NVLink's narr gain** — code workloads (spec-decode K+1 verify is heavily cross-card matmul) benefit more from cross-card bandwidth than narr decode (more sequential per-token). For ~95% of dual-3090 owners without NVLink, the trade is small TPS lift vs custom kernel module + DKMS maintenance burden. [Issue #91](https://github.com/noonghunna/club-3090/issues/91). |
|
||||
| `dual-dflash-noviz.yml`-shape **+ patched P2P drivers** (no NVLink hardware, custom_all_reduce ENABLED) | @aaronlockhartdev (2× 3090 PCIe x16, EPYC 7F52, Arch Linux, patched kernel module + `NCCL_P2P_LEVEL=PHB`) | fp8 | 200K | **100.47 / 160.15** (decode 101.53 / 164.44) | — | ~22.2 GB/card | 2026-05-07 | **Second patched-P2P cross-rig data point** — extends [#91 dual.yml result](https://github.com/noonghunna/club-3090/issues/91) to the DFlash + no-vision path. Same-rig controlled A/B: unpatched baseline 82.55 narr / 134.45 code → patched P2P 100.47 / 160.15 = **+22% narr / +19% code**. **Significantly larger lift than `dual.yml`-shape** (+22%/+19% here vs +2%/+9% on `dual.yml`) — DFlash's K+1 cross-card verify pattern stresses peer-bandwidth more than fp8-only `dual.yml`. **Important methodology update**: `NCCL_P2P_LEVEL=PHB` alone with the default vLLM image produced the same lift as the full vLLM `cuda.py` patch — **the in-container vLLM source patch is unnecessary**, only the kernel module patch matters. CV 4.6%/2.4%. custom_all_reduce ENABLED (vs disabled on the `dual.yml` row). [Issue #95](https://github.com/noonghunna/club-3090/issues/95) + [disc #70](https://github.com/noonghunna/club-3090/discussions/70). |
|
||||
| `carnice-bf16mtp.yml` | @noonghunna (2× 3090 PCIe, no NVLink) | fp8 | 262K | **72** / **80** | — | ~22.25 GB | 2026-05-04 | **Carnice-V2-27B (Hermes agentic fine-tune) + BF16 MTP overlay**. Full 262K context, 2 streams. 71.75 narr / 80.35 code wall TPS (n=5 each, CV ~11%), MTP AL 3.02-3.14, TTFT 141ms. Patched chat template for Hermes JSON tool calls. verify-full 7/8 PASS. soak PASS. |
|
||||
| `dual.yml` ⭐ | @lolren (2× 3090 PCIe + Ryzen 9 5950X, **250W/card cap**) | fp8 | 262K | **89.78 / 117.60** | — | ~22.3 GB | 2026-05-05 | **First cross-rig data on the v7.72.2 uplift** (image `nightly-01d4d1ad3`, post-PR #59). +30% narr / +32% code over @noonghunna 2026-04-29 baseline (69/89 on older image) — confirms the v7.72.2 dividend cross-rig. CV 3.3%/2.0%. MTP AL ~3.5, per-pos accept 94/84/72%. [Disc #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16820303). |
|
||||
| `dual-dflash.yml` | @lolren (2× 3090 PCIe + Ryzen 9 5950X, 250W cap) | FP16 | 185K | 87.10 / **142.0** | — | ~22.1 GB | 2026-05-05 | Older image `nightly-7a1eb8ac2`. +6% narr / +14% code over @noonghunna baseline (82/125) — likely Ryzen 5950X advantage on prefill. DFlash AL ~4.5, per-pos accept 93/81/68/56/48%, avg accept 69%. [Disc #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16820303). |
|
||||
| `bounded-thinking.yml` | @lolren (2× 3090 PCIe + Ryzen 9 5950X, 250W cap, **MTP-disabled-suspected**) | TQ3 | 180K | 64.86 / 64.96 (CV **0.1%**) | — | ~22.3 GB | 2026-05-05 | **Anomaly:** lolren reports "no spec-decode" on this run despite `bounded-thinking.yml` shipping `--speculative-config mtp n=3` by default. Near-identical narr=code TPS + extreme CV stability (0.1%) suggests MTP was inactive — likely because his image was older `nightly-7a1eb8ac2` (pre-v7.72.2 + pre-PN35). Re-test on `nightly-01d4d1ad3` should restore MTP path → expect ~50/66 narr/code with normal CV. Tracked. [Disc #18](https://github.com/noonghunna/club-3090/discussions/18#discussioncomment-16820303). |
|
||||
| `dual.yml` | @JDWarner (**Mixed RTX A5000 + RTX 3090**, both **Razer Core X eGPU enclosures over Thunderbolt 3**, Intel NUC11TNH i5-1135G7, **16 GB RAM**, headless, A5000=230W cap / 3090=290W cap, PCIe **x4 Gen 3** per card) | fp8 | 262K | **56.83 / 72.47** (soak p50 93.09) | — | ~23.6 GB/card | 2026-05-09 | **Soak: ✓ PASS** (5×5, 0 errors, 0 silent-empty, 100% TPS retention, 0 MiB growth). **First TB3 dual-eGPU + mixed-arch cross-rig data**. The setup that "shouldn't work": each card on a separate TB3 controller → ~3.94 GB/s effective per card vs ~32 GB/s on PCIe x16 Gen 4 (~8× cut), mixed Ampere SKUs (workstation A5000 + consumer 3090 with different mem bandwidth + clocks), 16 GB system RAM total. **Result: matches `dual.yml` PCIe x16 baseline within run-to-run noise** — confirms decode on Qwen3.6-27B is per-card-bandwidth bound, cross-card NCCL allreduce is small enough that even an 8× link cut doesn't dominate. Extends @aaronlockhartdev's #91/#95 finding (patched-P2P only +2%/+9% on `dual.yml`) in the opposite direction: even with 8× *less* cross-card bandwidth, decode holds. MTP AL 3.39-3.52, per-pos accept 0.93/0.83/0.70 (89% avg). verify-full + verify-stress all PASS. Genesis pin `7b9fd319` (v7.72.2). [Issue #107](https://github.com/noonghunna/club-3090/issues/107). |
|
||||
| `dual/docker-compose.yml` (default) | @ygafarov (**3090 via USB4 eGPU dock + 5070 Ti via OCuLink** — heterogeneous Ampere + Blackwell consumer dual-eGPU, AMD Ryzen AI MAX+ 395 / Strix Halo miniPC, CachyOS, 123 GB RAM, 290 W cap both cards, PCIe **x4** per card — USB4 ≈ 3.94 GB/s, OCuLink ≈ 7.88 GB/s) | fp8 | 200K | **65.10 / 85.81** | — | 17.1 / 15.7 GB | 2026-05-12 | **First heterogeneous Ampere + Blackwell consumer dual-eGPU on the matrix.** TP=2 bound by the slower USB4 link in allreduce + sm_86 kernels (5070 Ti spends back-half of step waiting — 91% util but only 125 W out of 290 W cap). KV pool 200K @ 1.00× concurrency — VRAM cap from the 5070 Ti's 16 GiB (model takes 13.8 GiB/card → only ~2.2 GiB left for KV on the smaller card). verify-stress 7/7 incl. **91K needle recall** (Cliff 2 clean). Soak ⚠ borderline (360 MiB > 200 MiB threshold — same eGPU-bus accretion as ygafarov's own #113 single-card row above at 240 MiB; 100% TPS retention + 0 silent-empty + 0 errors so not a leak). MTP AL 3.50, per-pos accept 0.94/0.86/0.70. CV 4.5%/1.8%. **Slower than ygafarov's own single-3090 #113 row** (68.86/91.70 at 48K) — on this rig the single-card path is recommended; the 5070 Ti adds VRAM cap pain without TPS gain. Driver 595.71.05, vLLM `nightly-1acd67a79`, no Genesis (Blackwell consumer not on allowlist). [Issue #120](https://github.com/noonghunna/club-3090/issues/120). |
|
||||
|
||||
### Quad-card (4× RTX 3090, TP=4)
|
||||
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | Peak VRAM | Date | Notes |
|
||||
|---|---|---|---:|---:|---:|---|---|
|
||||
| `multi4.yml` | @whamp (4× 3090 PCIe x4/x16/x8/x16, 300 W cap, no NVLink) | fp8 | 262K | 63 / 76 | ~23.5 GB | 2026-05-03 | TP=4 capacity king. **6.77× concurrency at 262K**. PASSES v2 continuous soak (20 sessions, 0 MiB growth, 90.8% TPS retention). PR [#44](https://github.com/noonghunna/club-3090/pull/44). |
|
||||
| `multi4.yml` | [@alanspires #127](https://github.com/noonghunna/club-3090/issues/127) (**6× 3090 VFIO-passthrough**, AMD EPYC 7313 host, all cards 250W cap, no NVLink) | fp8 | 262K | **74.93 / 92.85** | ~21.7 GB | 2026-05-14 | TP=4 on a 6-card rig — GPUs 4-5 free (Qwen `num_kv_heads=4` doesn't divide 6, so TP=6 invalid). **First VFIO-passthrough + virtualized data point** — scripts (verify-full / verify-stress / soak) all PASS on the virt envelope without modification. KV pool 1,774,963 tokens at 6.77× concurrency. Soak p50 122.84 / p95 161.14 across 5 multi-turn sessions, 0 errors. |
|
||||
| `multi4-dflash.yml` | @whamp (4× 3090 PCIe x4/x16/x8/x16, 300 W cap) | fp8 | 262K | 64 / **104** | ~22.0 GB | 2026-05-03 | TP=4 + DFlash. 2.27× concurrency at 262K. PASSES v2 continuous soak (5 sessions, 0 MiB growth, 100% TPS retention). **Bench-vs-soak inversion**: bench shows DFlash wins by 37% on short-prompt code, soak shows DFlash *loses* by 47% on multi-turn agent — DFlash AL likely collapses on mixed prompts. PR [#44](https://github.com/noonghunna/club-3090/pull/44). |
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | PP tok/s | Peak VRAM | Date | Notes |
|
||||
| --- | --- | --- | ---: | ---: | ---: | ---: | --- | --- |
|
||||
| `multi4.yml` | @whamp (4× 3090 PCIe x4/x16/x8/x16, 300 W cap, no NVLink) | fp8 | 262K | 63 / 76 | — | ~23.5 GB | 2026-05-03 | TP=4 capacity king. **6.77× concurrency at 262K**. PASSES v2 continuous soak (20 sessions, 0 MiB growth, 90.8% TPS retention). PR [#44](https://github.com/noonghunna/club-3090/pull/44). |
|
||||
| `multi4.yml` | [@alanspires #127](https://github.com/noonghunna/club-3090/issues/127) (**6× 3090 VFIO-passthrough**, AMD EPYC 7313 host, all cards 250W cap, no NVLink) | fp8 | 262K | **74.93 / 92.85** | — | ~21.7 GB | 2026-05-14 | TP=4 on a 6-card rig — GPUs 4-5 free (Qwen `num_kv_heads=4` doesn't divide 6, so TP=6 invalid). **First VFIO-passthrough + virtualized data point** — scripts (verify-full / verify-stress / soak) all PASS on the virt envelope without modification. KV pool 1,774,963 tokens at 6.77× concurrency. Soak p50 122.84 / p95 161.14 across 5 multi-turn sessions, 0 errors. |
|
||||
| `multi4-dflash.yml` | @whamp (4× 3090 PCIe x4/x16/x8/x16, 300 W cap) | fp8 | 262K | 64 / **104** | — | ~22.0 GB | 2026-05-03 | TP=4 + DFlash. 2.27× concurrency at 262K. PASSES v2 continuous soak (5 sessions, 0 MiB growth, 100% TPS retention). **Bench-vs-soak inversion**: bench shows DFlash wins by 37% on short-prompt code, soak shows DFlash *loses* by 47% on multi-turn agent — DFlash AL likely collapses on mixed prompts. PR [#44](https://github.com/noonghunna/club-3090/pull/44). |
|
||||
|
||||
### Verify-stress + soak-continuous matrix
|
||||
|
||||
@@ -228,18 +234,18 @@ Tested via `URL=http://localhost:8004 MODEL=luce-dflash bash scripts/verify-stre
|
||||
|
||||
Cross-rig data on Google's official Gemma 4 MTP "assistant" drafter (released 2026-05-05). PR [#41745](https://github.com/vllm-project/vllm/pull/41745) merged 2026-05-06 → today's nightly contains it natively (overlay dropped 2026-05-08). The companion compose [`dual/int8.yml`](models/gemma-4-31b/vllm/compose/dual/int8.yml) (added 2026-05-08) vendors PR [#40391](https://github.com/vllm-project/vllm/pull/40391) (rebased) + PR #42006 + PR #41991 to unlock per-token-head INT8 KV → 8.2× context lift on Ampere (32K → 262K). See announcement [discussion #67](https://github.com/noonghunna/club-3090/discussions/67) for the original Gemma 4 setup story; Phase 2 INT8 PTH validation in progress 2026-05-08.
|
||||
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | AL | Per-pos accept (code) | Peak VRAM | Date | Notes |
|
||||
|---|---|---|---:|---:|---:|---|---:|---|---|
|
||||
| `dual.yml` (TP=2) | @noonghunna (2× 3090 PCIe, no NVLink, 230W cap) | bf16 | 32K | **108.87 / 142.25** | **3.94-4.04** | 92 / 79 / 68 / 59 % | 22.5 GB/card | 2026-05-05 | First Ampere consumer cross-rig data on Google MTP drafters. **+1.79× narr / +2.31× code** over baseline (61 TPS no-spec-decode same TP). **PASSES continuous soak** (100 turns, 0 errors / 0 silent-empty / 0 MiB growth, 98.3% TPS retention). bf16 KV (fp8 blocked on Ampere — see TP=1 row). PR [#41745](https://github.com/vllm-project/vllm/pull/41745) overlay + transformers 5.8.0 entrypoint. |
|
||||
| `dual.yml` (TP=2) re-bench post-#41745 merge | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | 32K | **105.91 / 141.11** | 3.94 | (warming) | 21.5 GB/card | 2026-05-08 | Re-validated on post-merge nightly `1acd67a795...` (PR #41745 overlay dropped, transformers entrypoint upgrade dropped). Within CV of 109/142 baseline → cleanup is parity-clean. KV pool 99K tokens, 3.03× concurrency at 32K. |
|
||||
| **`dual-int8.yml` (TP=2, max-num-seqs=4)** ⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | **int8_per_token_head** | **98K** | **96.16 / 127.11** | 3.79 | (warming) | 22.2 GB/card | 2026-05-08 | **3.07× context lift over bf16 ceiling on Ampere — INT8 PTH KV unblocks Gemma 4 long-context.** Vendors PR #40391 rebased + PR #42006 + PR #41991 stacked (see `models/gemma-4-31b/vllm/patches/`). KV pool **354K tokens, 3.6× concurrency**. ~10% TPS cost vs bf16 / 32K. **PASSES verify-stress 7/7** incl. 91K Cliff-2 needle. PR #40391's per-token-head page-size fix routes via `get_padded_attention_kv_cache_shape()`; INT8 (not fp8) is the right Ampere dtype because Triton `fp8e4nv` kernel is not supported on sm_86 (Ada/Blackwell only). |
|
||||
| **`dual-int8.yml` (TP=2, max-num-seqs=1, MAX_MODEL_LEN=262144)** ⭐⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | int8_per_token_head | **262K (model native max)** | **95.27 / 125.93** | 3.93 | (warming) | 22.1 GB/card | 2026-05-08 | **8.2× context lift vs dual.yml — full Gemma 4 native context (262144) unblocked on dual 3090 Ampere.** KV pool 455K tokens, 1.74× concurrency at full 262K. **PASSES verify-stress 7/7** + **137K NIAH PASS** (correctly recalled needle from 137,557-token prompt, 5min wall, ~458 prefill TPS). Per-token TPS preserved at full max-model-len (95/126 at 262K vs 96/127 at 98K — bench prompt size dominates, not max-model-len). Override `MAX_MODEL_LEN=262144 MAX_NUM_SEQS=1`. |
|
||||
| `dual-dflash.yml` (TP=2, n=7) | @noonghunna (2× 3090 PCIe, no NVLink, 230W cap) | bf16 | 32K | **95.16 / 167.55** | **~3.0 narr / 5.23 code** | 89 / 78 / 66 / 57 / 50 / 43 / 39 % | 22.7 GB/card | 2026-05-06 | First Ampere consumer cross-rig data on **z-lab Gemma 4 DFlash** block-diffusion drafter (vLLM PR [#41703](https://github.com/vllm-project/vllm/pull/41703) — Codex-rebased onto upstream/main). **+2.74× code / +1.56× narr** over baseline. **PASSES continuous soak** (100 turns, 0 errors / 0 silent-empty / 0 MiB growth, 98.6% TPS retention, p50 55.8 TPS — 5.8% higher than n=5). DFlash dominates MTP on **code (+18%)**; MTP wins on narrative (+15%). n-sweep: n=5 109/141 (best narr) → n=6 99/161 (knee) → **n=7 95/168 (code-optimal default)** → n=8 91/167 (dominated) → n=15 82/172 (past knee). PR #41703 overlay (12 RO-mounted files) + transformers 5.8.0 + nightly `e47c98ef`. |
|
||||
| `dual-dflash.yml` (TP=2, n=7) re-bench | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | 32K | **104.48 / 176.66** (CV 2.1% / 3.6%) | 2.85 narr / 4.11-4.94 code | (warm) | 22.3 GB/card | 2026-05-08 | Re-validated after Phase 2 INT8 PTH session. Same overlay + same `e47c98ef` pin — modest uplift over 2026-05-06 (warm-cache + ambient variance — CV ranges overlap at +1σ). KV pool 42,848 tokens, 1.31× concurrency at 32K. Ampere upper-bound for Gemma 4 + DFlash: code-optimal at 177 TPS. Combining DFlash drafter with PR #40391 INT8 PTH KV (Phase 3, dual-dflash-int8.yml) is the next structural step — would unlock long-context code-optimal. |
|
||||
| **`dual-dflash-int8.yml` (TP=2, n=7, MAX_MODEL_LEN=262144, MAX_NUM_SEQS=1)** ⭐⭐⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | int8_per_token_head + drafter bf16 | **262K (model native max)** | **86.86 / 145.96** (CV 0.9% / 2.2%) | 5.0-5.3 long-ctx code | (warm) | 22.0 GB/card | 2026-05-08 | **8.2× context lift over `dual-dflash.yml` 32K bf16 baseline — DFlash + INT8 PTH KV unblocked on Ampere via [vLLM PR #42102](https://github.com/vllm-project/vllm/pull/42102) (our patch).** Matches the 32K bf16 baseline's code TPS within CV at 8× more context (146 vs 168 = -13% perf cost for 8× ctx). KV pool 168,178 tokens, 0.64× concurrency at full 262K — effective single-stream serving ceiling ~168K. **NIAH PASS at 98,444 tokens** (`bronze octopus 17` recalled cleanly, 157s wall = ~625 effective prefill TPS). DFlash drafter uses BF16 KV in independent pool (target uses INT8 PTH); the patch partitions them at unify-time, drafter cache_dtype overridden to "auto" in qwen3_dflash.py, FA metadata scheduler reads per-spec dtype. **n-sweep at 262K config 2026-05-08**: n=5 81.91/138.87 (-6/-5%), n=7 86.86/145.96 (default), n=8 86.63/152.06 (+0/+4% but CV 5.7% — within noise). n=7 retained as default — sweet spot didn't shift meaningfully from the 32K bf16 baseline. **DFlash code-optimal advantage preserved at long ctx**: 146 code TPS vs dual-int8.yml's 126 at 262K = +16% code (offset: -10% narr). Pin nightly `e47c98ef`; needs `dual-dflash-int8.yml` compose (still ⚠️ flagged DOES NOT BOOT until PR #42102 lands — currently requires the vllm-src patch mounts). |
|
||||
| `single.yml` (TP=1) | @noonghunna (1× 3090) | bf16 / fp8 | — | **boot OOM** | — | — | — | 2026-05-05 | **Upstream-blocked on Ampere consumer.** bf16 KV: weights+drafter+profiling at 8K ctx + mem-util 0.95 leaves zero KV pool ("No available memory for the cache blocks"). fp8 KV: Triton `fp8e4nv not supported in this architecture` on sm_86 (Ampere supports `fp8e4b15`/`fp8e5` only); but `fp8_e5m2` is rejected by `gemma4_mm.py:1336` allowlist. Compose preserved for re-test when (a) vLLM adds Ampere-aware fp8 dispatch OR (b) PR #41745 relaxes the assert. Gemma 4 26B-A4B MoE single-card is the obvious follow-up. |
|
||||
| **`dual-awq.yml` (TP=2, MAX_NUM_SEQS=4, MTP n=4)** ⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | **65K** | **104.59 / 130.56** (CV 1.7% / 0.4%) | 3.07 narr / 3.55-3.88 code | 0.79 / 0.59 / 0.43 / 0.31 | 19.8 GB/card | 2026-05-08 | **Cross-rig reproducer of @3dluvr's [#103 bench](https://github.com/noonghunna/club-3090/issues/103) — AWQ-4bit weights instead of AutoRound INT4.** Bypasses PR #40391 (per-token-head bug) entirely because AWQ doesn't use FP8 KV — trades weight quant precision for ctx instead of trading KV precision. Vendor: `cyankiwi/gemma-4-31B-it-AWQ-4bit` (~17 GB on disk, AWQ-pack-quantized group_size=32, asymmetric, MSE observer; routed via vLLM compressed-tensors loader → Marlin kernel). KV pool 89,228 tokens, 1.36× concurrency at 65K. Numbers comparable to dual.yml (BF16 INT4 AutoRound, 105.91/141.11) on narrative; slightly slower code at this n. Default config — multi-stream agent / RAG. |
|
||||
| **`dual-awq.yml` (TP=2, MAX_NUM_SEQS=1, MTP n=8, MAX_MODEL_LEN=118304)** ⭐⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | **118K** | **101.16 / 141.90** (CV 2.5% / 0.4%) | 3.7 narr / 5.13 code | 0.89 / 0.77 / 0.64 / 0.54 / 0.45 / 0.37 / 0.26 / 0.21 | 19.8 GB/card | 2026-05-08 | **3.7× context lift over dual.yml's BF16 32K ceiling.** Closer match to @3dluvr's #103 anchor (113/163 at 195K with `--dtype half --async-scheduling --cudagraph_capture_sizes [9]`). Our config: `--dtype bfloat16`, default cudagraph capture sizes `[1,2,4,8,16]` (auto-clamped to single-stream). vLLM auto-estimated 195K won't fit at 0.85 mem-util (10.18 GiB KV needed vs 7.25 GiB available); 118K is the achievable ceiling at this mem-util / cudagraph budget. **NIAH PASS at 88K** (recalled "bronze octopus 17" from 88,527-token prompt, 11-tok completion in 127.1s wall, ~696 effective prefill TPS). KV pool 118,304 tokens, 1.00× concurrency. Code AL 5.13 (n=8 saturates well on code, less so on narr where AL is 3.7). **n=8 vs n=4 trade**: code +9% (130 → 142), narrative -3% (104 → 101) — n=8 dominates n=4 for code. Override `MAX_MODEL_LEN=118304 MAX_NUM_SEQS=1 MTP_N=8`. **A/B finding 2026-05-08** — three tuning flags tested individually on this rig (with bench n=3 each, sync-baseline 101.16/141.90):
|
||||
| Compose | Rig | KV | Max ctx | Narr / Code TPS | PP tok/s | AL | Per-pos accept (code) | Peak VRAM | Date | Notes |
|
||||
| --- | --- | --- | ---: | ---: | ---: | ---: | --- | ---: | --- | --- |
|
||||
| `dual.yml` (TP=2) | @noonghunna (2× 3090 PCIe, no NVLink, 230W cap) | bf16 | 32K | **108.87 / 142.25** | — | **3.94-4.04** | 92 / 79 / 68 / 59 % | 22.5 GB/card | 2026-05-05 | First Ampere consumer cross-rig data on Google MTP drafters. **+1.79× narr / +2.31× code** over baseline (61 TPS no-spec-decode same TP). **PASSES continuous soak** (100 turns, 0 errors / 0 silent-empty / 0 MiB growth, 98.3% TPS retention). bf16 KV (fp8 blocked on Ampere — see TP=1 row). PR [#41745](https://github.com/vllm-project/vllm/pull/41745) overlay + transformers 5.8.0 entrypoint. |
|
||||
| `dual.yml` (TP=2) re-bench post-#41745 merge | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | 32K | **105.91 / 141.11** | — | 3.94 | (warming) | 21.5 GB/card | 2026-05-08 | Re-validated on post-merge nightly `1acd67a795...` (PR #41745 overlay dropped, transformers entrypoint upgrade dropped). Within CV of 109/142 baseline → cleanup is parity-clean. KV pool 99K tokens, 3.03× concurrency at 32K. |
|
||||
| **`dual-int8.yml` (TP=2, max-num-seqs=4)** ⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | **int8_per_token_head** | **98K** | **96.16 / 127.11** | — | 3.79 | (warming) | 22.2 GB/card | 2026-05-08 | **3.07× context lift over bf16 ceiling on Ampere — INT8 PTH KV unblocks Gemma 4 long-context.** Vendors PR #40391 rebased + PR #42006 + PR #41991 stacked (see `models/gemma-4-31b/vllm/patches/`). KV pool **354K tokens, 3.6× concurrency**. ~10% TPS cost vs bf16 / 32K. **PASSES verify-stress 7/7** incl. 91K Cliff-2 needle. PR #40391's per-token-head page-size fix routes via `get_padded_attention_kv_cache_shape()`; INT8 (not fp8) is the right Ampere dtype because Triton `fp8e4nv` kernel is not supported on sm_86 (Ada/Blackwell only). |
|
||||
| **`dual-int8.yml` (TP=2, max-num-seqs=1, MAX_MODEL_LEN=262144)** ⭐⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | int8_per_token_head | **262K (model native max)** | **95.27 / 125.93** | — | 3.93 | (warming) | 22.1 GB/card | 2026-05-08 | **8.2× context lift vs dual.yml — full Gemma 4 native context (262144) unblocked on dual 3090 Ampere.** KV pool 455K tokens, 1.74× concurrency at full 262K. **PASSES verify-stress 7/7** + **137K NIAH PASS** (correctly recalled needle from 137,557-token prompt, 5min wall, ~458 prefill TPS). Per-token TPS preserved at full max-model-len (95/126 at 262K vs 96/127 at 98K — bench prompt size dominates, not max-model-len). Override `MAX_MODEL_LEN=262144 MAX_NUM_SEQS=1`. |
|
||||
| `dual-dflash.yml` (TP=2, n=7) | @noonghunna (2× 3090 PCIe, no NVLink, 230W cap) | bf16 | 32K | **95.16 / 167.55** | — | **~3.0 narr / 5.23 code** | 89 / 78 / 66 / 57 / 50 / 43 / 39 % | 22.7 GB/card | 2026-05-06 | First Ampere consumer cross-rig data on **z-lab Gemma 4 DFlash** block-diffusion drafter (vLLM PR [#41703](https://github.com/vllm-project/vllm/pull/41703) — Codex-rebased onto upstream/main). **+2.74× code / +1.56× narr** over baseline. **PASSES continuous soak** (100 turns, 0 errors / 0 silent-empty / 0 MiB growth, 98.6% TPS retention, p50 55.8 TPS — 5.8% higher than n=5). DFlash dominates MTP on **code (+18%)**; MTP wins on narrative (+15%). n-sweep: n=5 109/141 (best narr) → n=6 99/161 (knee) → **n=7 95/168 (code-optimal default)** → n=8 91/167 (dominated) → n=15 82/172 (past knee). PR #41703 overlay (12 RO-mounted files) + transformers 5.8.0 + nightly `e47c98ef`. |
|
||||
| `dual-dflash.yml` (TP=2, n=7) re-bench | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | 32K | **104.48 / 176.66** (CV 2.1% / 3.6%) | — | 2.85 narr / 4.11-4.94 code | (warm) | 22.3 GB/card | 2026-05-08 | Re-validated after Phase 2 INT8 PTH session. Same overlay + same `e47c98ef` pin — modest uplift over 2026-05-06 (warm-cache + ambient variance — CV ranges overlap at +1σ). KV pool 42,848 tokens, 1.31× concurrency at 32K. Ampere upper-bound for Gemma 4 + DFlash: code-optimal at 177 TPS. Combining DFlash drafter with PR #40391 INT8 PTH KV (Phase 3, dual-dflash-int8.yml) is the next structural step — would unlock long-context code-optimal. |
|
||||
| **`dual-dflash-int8.yml` (TP=2, n=7, MAX_MODEL_LEN=262144, MAX_NUM_SEQS=1)** ⭐⭐⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | int8_per_token_head + drafter bf16 | **262K (model native max)** | **86.86 / 145.96** (CV 0.9% / 2.2%) | — | 5.0-5.3 long-ctx code | (warm) | 22.0 GB/card | 2026-05-08 | **8.2× context lift over `dual-dflash.yml` 32K bf16 baseline — DFlash + INT8 PTH KV unblocked on Ampere via [vLLM PR #42102](https://github.com/vllm-project/vllm/pull/42102) (our patch).** Matches the 32K bf16 baseline's code TPS within CV at 8× more context (146 vs 168 = -13% perf cost for 8× ctx). KV pool 168,178 tokens, 0.64× concurrency at full 262K — effective single-stream serving ceiling ~168K. **NIAH PASS at 98,444 tokens** (`bronze octopus 17` recalled cleanly, 157s wall = ~625 effective prefill TPS). DFlash drafter uses BF16 KV in independent pool (target uses INT8 PTH); the patch partitions them at unify-time, drafter cache_dtype overridden to "auto" in qwen3_dflash.py, FA metadata scheduler reads per-spec dtype. **n-sweep at 262K config 2026-05-08**: n=5 81.91/138.87 (-6/-5%), n=7 86.86/145.96 (default), n=8 86.63/152.06 (+0/+4% but CV 5.7% — within noise). n=7 retained as default — sweet spot didn't shift meaningfully from the 32K bf16 baseline. **DFlash code-optimal advantage preserved at long ctx**: 146 code TPS vs dual-int8.yml's 126 at 262K = +16% code (offset: -10% narr). Pin nightly `e47c98ef`; needs `dual-dflash-int8.yml` compose (still ⚠️ flagged DOES NOT BOOT until PR #42102 lands — currently requires the vllm-src patch mounts). |
|
||||
| `single.yml` (TP=1) | @noonghunna (1× 3090) | bf16 / fp8 | — | **boot OOM** | — | — | — | — | 2026-05-05 | **Upstream-blocked on Ampere consumer.** bf16 KV: weights+drafter+profiling at 8K ctx + mem-util 0.95 leaves zero KV pool ("No available memory for the cache blocks"). fp8 KV: Triton `fp8e4nv not supported in this architecture` on sm_86 (Ampere supports `fp8e4b15`/`fp8e5` only); but `fp8_e5m2` is rejected by `gemma4_mm.py:1336` allowlist. Compose preserved for re-test when (a) vLLM adds Ampere-aware fp8 dispatch OR (b) PR #41745 relaxes the assert. Gemma 4 26B-A4B MoE single-card is the obvious follow-up. |
|
||||
| **`dual-awq.yml` (TP=2, MAX_NUM_SEQS=4, MTP n=4)** ⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | **65K** | **104.59 / 130.56** (CV 1.7% / 0.4%) | — | 3.07 narr / 3.55-3.88 code | 0.79 / 0.59 / 0.43 / 0.31 | 19.8 GB/card | 2026-05-08 | **Cross-rig reproducer of @3dluvr's [#103 bench](https://github.com/noonghunna/club-3090/issues/103) — AWQ-4bit weights instead of AutoRound INT4.** Bypasses PR #40391 (per-token-head bug) entirely because AWQ doesn't use FP8 KV — trades weight quant precision for ctx instead of trading KV precision. Vendor: `cyankiwi/gemma-4-31B-it-AWQ-4bit` (~17 GB on disk, AWQ-pack-quantized group_size=32, asymmetric, MSE observer; routed via vLLM compressed-tensors loader → Marlin kernel). KV pool 89,228 tokens, 1.36× concurrency at 65K. Numbers comparable to dual.yml (BF16 INT4 AutoRound, 105.91/141.11) on narrative; slightly slower code at this n. Default config — multi-stream agent / RAG. |
|
||||
| **`dual-awq.yml` (TP=2, MAX_NUM_SEQS=1, MTP n=8, MAX_MODEL_LEN=118304)** ⭐⭐ | @noonghunna (2× 3090 PCIe, 230W cap) | bf16 | **118K** | **101.16 / 141.90** (CV 2.5% / 0.4%) | — | 3.7 narr / 5.13 code | 0.89 / 0.77 / 0.64 / 0.54 / 0.45 / 0.37 / 0.26 / 0.21 | 19.8 GB/card | 2026-05-08 | **3.7× context lift over dual.yml's BF16 32K ceiling.** Closer match to @3dluvr's #103 anchor (113/163 at 195K with `--dtype half --async-scheduling --cudagraph_capture_sizes [9]`). Our config: `--dtype bfloat16`, default cudagraph capture sizes `[1,2,4,8,16]` (auto-clamped to single-stream). vLLM auto-estimated 195K won't fit at 0.85 mem-util (10.18 GiB KV needed vs 7.25 GiB available); 118K is the achievable ceiling at this mem-util / cudagraph budget. **NIAH PASS at 88K** (recalled "bronze octopus 17" from 88,527-token prompt, 11-tok completion in 127.1s wall, ~696 effective prefill TPS). KV pool 118,304 tokens, 1.00× concurrency. Code AL 5.13 (n=8 saturates well on code, less so on narr where AL is 3.7). **n=8 vs n=4 trade**: code +9% (130 → 142), narrative -3% (104 → 101) — n=8 dominates n=4 for code. Override `MAX_MODEL_LEN=118304 MAX_NUM_SEQS=1 MTP_N=8`. **A/B finding 2026-05-08** — three tuning flags tested individually on this rig (with bench n=3 each, sync-baseline 101.16/141.90): |
|
||||
| Flag | Narr Δ | Code Δ | Verdict |
|
||||
|---|---:|---:|---|
|
||||
| `--dtype half` (vs bfloat16) | -3% | -2% | Ampere has NO fp16 hardware accel on sm_86 — bfloat16 is canonical |
|
||||
|
||||
@@ -16,6 +16,73 @@ history; SemVer takes over from `v0.3.0` onward.
|
||||
|
||||
---
|
||||
|
||||
## v0.7.1 — 2026-05-15
|
||||
|
||||
|
||||
### ✨ Features
|
||||
|
||||
- feat(bench): surface prompt processing throughput ([2a148d7](https://github.com/noonghunna/club-3090/commit/2a148d702b9415129d4c4ec9d3e7d30765927aa4))
|
||||
- feat(llamacpp): expose batch tuning knobs ([02249ab](https://github.com/noonghunna/club-3090/commit/02249ab1939f354ac062d343efefe32677203174))
|
||||
|
||||
|
||||
### 🐛 Bug fixes
|
||||
|
||||
- fix(ci): simplify vllm image workflow, drop smoke-gate (#135) ([ce2617e](https://github.com/noonghunna/club-3090/commit/ce2617e0bc0f56d42caf64e96847d966380be80c))
|
||||
|
||||
|
||||
### 📝 Documentation
|
||||
|
||||
- docs(upstream): PR #42102 closed-as-slop; local overlay permanent ([57eb269](https://github.com/noonghunna/club-3090/commit/57eb269cd70935fc3069b85e46ead8f0f0af13dc))
|
||||
|
||||
|
||||
|
||||
[Pin: `git checkout v0.7.1`] · [Full diff](https://github.com/noonghunna/club-3090/compare/v0.7.0...v0.7.1)
|
||||
## v0.7.0 — 2026-05-14
|
||||
|
||||
|
||||
### ✨ Features
|
||||
|
||||
- feat(scripts): add diagnose-profile triage ([c2adb39](https://github.com/noonghunna/club-3090/commit/c2adb3970dbdcb3bf227cfc7a1ac58a2de3930a4))
|
||||
- feat(compose): use profile-sourced vllm image pins ([e6e33ab](https://github.com/noonghunna/club-3090/commit/e6e33ab37235cdfa2d47987e16e00d617564e55d))
|
||||
- feat(launch): export profile vllm pins ([c306383](https://github.com/noonghunna/club-3090/commit/c3063838bfde3f7811db3edbb9541814aea8f09d))
|
||||
- feat(profiles): resolve vllm nightly pins ([40f1ef7](https://github.com/noonghunna/club-3090/commit/40f1ef78f8e3ab80a5424ce196c7715df85a0d1a))
|
||||
- feat(launch): add estate planner orchestration ([c9b153f](https://github.com/noonghunna/club-3090/commit/c9b153f91f3dd4a076395008ad3bfecdd226a452))
|
||||
- feat(launch): validate single-model profiles ([a142b1c](https://github.com/noonghunna/club-3090/commit/a142b1ce1085c78e8752fa52718af66f7a49f237))
|
||||
- feat(compat): add profile validator and estate self-test ([6581ccc](https://github.com/noonghunna/club-3090/commit/6581ccca9e8613594dba876ba2472e78d48c9eed))
|
||||
- feat(compose): accept ESTATE_GPUS and ESTATE_PORT overrides ([a57596e](https://github.com/noonghunna/club-3090/commit/a57596e0764babc6fe91e96af8b6dd746da45e09))
|
||||
- feat(profiles): ship v0.7.0 data layer ([69825d7](https://github.com/noonghunna/club-3090/commit/69825d7dae0fe1387a04879b31e0d225143b3684))
|
||||
|
||||
|
||||
### 🐛 Bug fixes
|
||||
|
||||
- fix(tools): resolve profile image pins in audit ([98535dc](https://github.com/noonghunna/club-3090/commit/98535dc81453e54be03a1f5915bb2367ebab5c27))
|
||||
- fix(tools): bump engine nightly profiles ([d1acde0](https://github.com/noonghunna/club-3090/commit/d1acde0b28eccb123338d2a9d94e4ef6ff75cc4b))
|
||||
- fix(ci): keep vllm base arg in image metadata ([1abe65f](https://github.com/noonghunna/club-3090/commit/1abe65fe2810be594d6890ce0f26a3f0a554075f))
|
||||
- fix(launch): persist estate source of truth ([52e4347](https://github.com/noonghunna/club-3090/commit/52e43470c6ee111dfa97be81e13e229ede715073))
|
||||
|
||||
|
||||
### 📝 Documentation
|
||||
|
||||
- docs: document profile-sourced vllm pins ([86445be](https://github.com/noonghunna/club-3090/commit/86445be3e8c24ad829d45c4f1702a0bccf27e4dc))
|
||||
- docs: document club vllm image pin ([2ae8303](https://github.com/noonghunna/club-3090/commit/2ae8303833497f20d4079a12226ca23c83f71337))
|
||||
- docs: expand KV_MATH + add ADDING_MODELS workflow ([1f8aaa2](https://github.com/noonghunna/club-3090/commit/1f8aaa2acc72a7cff763fa81ae938d552fffffb2))
|
||||
- docs(hardware): clarify 3090 stock TDP varies by board SKU ([0d59f94](https://github.com/noonghunna/club-3090/commit/0d59f949e472095e3ecb83ce133eb103d10588d9))
|
||||
|
||||
|
||||
### 🧹 Maintenance
|
||||
|
||||
- chore(vllm): use club3090 image in composes ([aebc4f3](https://github.com/noonghunna/club-3090/commit/aebc4f321c3536943bfbd554d2c21ed6824742f9))
|
||||
- chore(ci): build club vllm image ([e88a2a8](https://github.com/noonghunna/club-3090/commit/e88a2a8d21efd3556739fcb0fb1a26327b7d38cd))
|
||||
- refactor(kv-calc): consume profile data ([9ccde62](https://github.com/noonghunna/club-3090/commit/9ccde62abe360bfb9170fc88102623dc5e87597e))
|
||||
|
||||
|
||||
### 🧹 Other
|
||||
|
||||
- Revert "chore(vllm): use club3090 image in composes" ([c7c40bd](https://github.com/noonghunna/club-3090/commit/c7c40bdf1232ec2a2f8e5b1d98249df33026f46b))
|
||||
|
||||
|
||||
|
||||
[Pin: `git checkout v0.7.0`] · [Full diff](https://github.com/noonghunna/club-3090/compare/v0.6.3...v0.7.0)
|
||||
## v0.6.3 — 2026-05-14
|
||||
|
||||
|
||||
|
||||
+2
-2
@@ -93,7 +93,7 @@ For contributors who know the `BENCHMARKS.md` section structure and want to prop
|
||||
```bash
|
||||
bash scripts/bench.sh
|
||||
```
|
||||
Drop the run-by-run output in the PR — `wall_TPS`, `decode_TPS`, `TTFT`, MTP `AL` (where applicable). Mean + CV + n=5 minimum.
|
||||
Drop the run-by-run output in the PR — `wall_TPS`, `decode_TPS`, `PP tok/s`, `TTFT`, MTP `AL` (where applicable). Mean + CV + n=5 minimum.
|
||||
5. **Open the PR with a description that answers four questions:**
|
||||
- What problem does this solve? (One paragraph.)
|
||||
- What's the measured impact? (Numbers.)
|
||||
@@ -127,7 +127,7 @@ Or run the steps individually if you'd rather:
|
||||
CONTAINER=<container-name> ENDPOINT=<http://localhost:port> \
|
||||
bash scripts/soak-test.sh
|
||||
```
|
||||
5. **`bench.sh` run** — 3 warmups + 5 measured runs of narrative + code prompts. Report `wall_TPS`, `decode_TPS`, `TTFT`, peak VRAM/card per run, MTP/DFlash AL where applicable.
|
||||
5. **`bench.sh` run** — 3 warmups + 5 measured runs of narrative + code prompts. Report `wall_TPS`, `decode_TPS`, `PP tok/s`, `TTFT`, peak VRAM/card per run, MTP/DFlash AL where applicable. For llama.cpp or other engines without vLLM prompt-throughput logs, include `PP=1 bash scripts/bench.sh` so the long-prompt fallback captures prompt-processing throughput.
|
||||
6. **BENCHMARKS row** — under the appropriate model section, mirroring existing column shape (incl. `Rig` column). Attribution is automatic.
|
||||
7. **CHANGELOG entry** — in `models/<model>/CHANGELOG.md`.
|
||||
|
||||
|
||||
@@ -55,7 +55,7 @@ More models coming. The repo structure scales — when we add Qwen3.5-27B / GLM-
|
||||
|
||||

|
||||
|
||||
Bench protocol: 3 warm + 5 measured runs of the canonical narrative + code prompts. Substrate: vLLM nightly `0.20.1rc1.dev16+g7a1eb8ac2` + Genesis v7.69 dev tip (commit `2db18df`), with local backports `patch_inputs_embeds_optional.py` (vllm#35975) and `patch_tolist_cudagraph.py`. llama.cpp mainline `0d0764dfd`, RTX 3090 sm_86 PCIe-only at 230 W. Per-config details + run-by-run numbers + VRAM + AL/accept rates: [models/qwen3.6-27b/CHANGELOG.md](models/qwen3.6-27b/CHANGELOG.md) (per-model history) and [scripts/bench.sh](scripts/bench.sh) (canonical bench).
|
||||
Bench protocol: 3 warm + 5 measured runs of the canonical narrative + code prompts. `scripts/bench.sh` reports wall TPS, decode TPS, TTFT, and prompt-processing throughput (`PP tok/s`; vLLM log scrape, `PP=1` long-prompt fallback for llama.cpp). Substrate: vLLM nightly `0.20.1rc1.dev16+g7a1eb8ac2` + Genesis v7.69 dev tip (commit `2db18df`), with local backports `patch_inputs_embeds_optional.py` (vllm#35975) and `patch_tolist_cudagraph.py`. llama.cpp mainline `0d0764dfd`, RTX 3090 sm_86 PCIe-only at 230 W. Per-config details + run-by-run numbers + VRAM + AL/accept rates: [models/qwen3.6-27b/CHANGELOG.md](models/qwen3.6-27b/CHANGELOG.md) (per-model history) and [scripts/bench.sh](scripts/bench.sh) (canonical bench).
|
||||
|
||||
---
|
||||
|
||||
|
||||
+88
-99
@@ -1,127 +1,116 @@
|
||||
# Club-3090 CI GPU Runner Setup
|
||||
# Club-3090 vLLM Image — Build + Distribution
|
||||
|
||||
The vLLM image workflow builds on GitHub-hosted Ubuntu, then optionally smokes the
|
||||
fresh image on a self-hosted GPU runner. If no runner is registered, the workflow
|
||||
still pushes the dated `nightly-YYYYMMDD-clubXXXX` image and leaves `latest` /
|
||||
`nightly-stable` untouched.
|
||||
The vLLM image workflow (`.github/workflows/build-vllm-image.yml`) builds on
|
||||
GitHub-hosted Ubuntu, applies vendored overlays via Dockerfile, pushes a dated
|
||||
nightly tag to GHCR, then promotes the `:latest` and `:nightly-stable` aliases
|
||||
to point at the just-built dated tag. No self-hosted runner required.
|
||||
|
||||
## Runner Requirements
|
||||
The release-pinned tag `:club-vX.Y.Z` is created automatically when the workflow
|
||||
runs on a Git tag push (e.g. `v0.7.0`).
|
||||
|
||||
- Linux x86_64 host with Docker Engine and Docker Compose v2.
|
||||
- NVIDIA driver and NVIDIA Container Toolkit installed.
|
||||
- At least two 24 GB NVIDIA GPUs for the canonical `qwen3.6-27b/vllm/dual`
|
||||
smoke. The production validation target is 2x RTX 3090.
|
||||
- Enough local storage for the vLLM image, model cache, Docker layers, and
|
||||
compile caches. Plan for at least 250 GB free.
|
||||
- A dedicated runner host. Do not run untrusted pull-request jobs on this
|
||||
machine.
|
||||
## Tag conventions
|
||||
|
||||
## Labels
|
||||
| Tag | Mutable? | What it represents | Recommended for |
|
||||
|---|---|---|---|
|
||||
| `nightly-YYYYMMDD-clubNNNN` | Immutable | A specific build of upstream nightly + our overlays | Reproducibility; pin in `VLLM_IMAGE` to lock to a known state |
|
||||
| `:latest` | Mutable | The most-recent dated nightly | Users who want bleeding-edge; expect occasional breakage |
|
||||
| `:nightly-stable` | Mutable | Same target as `:latest` today; reserved for future divergence | Same as `:latest` for now |
|
||||
| `:club-vX.Y.Z` | Immutable (per release) | Built on the v0.7.0 tag push and never moves | Users who want verified releases; this is the recommended path |
|
||||
|
||||
Register the runner with the normal self-hosted labels plus `gpu`:
|
||||
## Why no smoke-gating?
|
||||
|
||||
```text
|
||||
self-hosted
|
||||
linux
|
||||
x64
|
||||
gpu
|
||||
```
|
||||
The workflow originally tried to gate `:latest` promotion on a self-hosted GPU
|
||||
runner running `verify-full.sh` + a 3-prompt smoke. That added:
|
||||
|
||||
The workflow checks for an online runner with `self-hosted` and `gpu`; the smoke
|
||||
job itself targets `[self-hosted, linux, x64, gpu]`.
|
||||
- A dependency on infra (the runner) we don't maintain.
|
||||
- A permission requirement (`administration: read` to list runners) the default `GITHUB_TOKEN` lacks.
|
||||
- A failure mode where `:latest` never gets promoted if no runner is registered.
|
||||
|
||||
## Registration
|
||||
We dropped smoke-gating in v0.7.1 (issue #135) because:
|
||||
|
||||
1. Open the GitHub repository.
|
||||
2. Go to **Settings -> Actions -> Runners -> New self-hosted runner**.
|
||||
3. Choose Linux x64 and follow GitHub's generated commands.
|
||||
4. Add the `gpu` label during configuration, or add it later from the runner UI.
|
||||
5. Install the runner as a service:
|
||||
1. The build step itself catches the most common failure modes (missing patch
|
||||
source, invalid Dockerfile, overlay path mismatches).
|
||||
2. Users who want verified images use `:club-vX.Y.Z` release tags, not `:latest`.
|
||||
3. The Docker Hub convention is that `:latest` = "most recent, no guarantees".
|
||||
|
||||
If a self-hosted GPU runner ever becomes available, smoke-gating can be layered
|
||||
on top of this workflow as a separate post-build job — the simpler design today
|
||||
doesn't preclude it.
|
||||
|
||||
## Using the GHCR image
|
||||
|
||||
The pre-built GHCR image is **opt-in**. Default launches use the upstream vLLM
|
||||
nightly SHA resolved from `scripts/lib/profiles/engines/<engine-id>.yml →
|
||||
install.spec` (the standard `vllm/vllm-openai:nightly-<sha>` ref).
|
||||
|
||||
To use the verified release image:
|
||||
|
||||
```bash
|
||||
sudo ./svc.sh install
|
||||
sudo ./svc.sh start
|
||||
VLLM_IMAGE=ghcr.io/noonghunna/vllm-club3090:club-v0.7.0 \
|
||||
bash scripts/launch.sh --variant vllm/dual
|
||||
```
|
||||
|
||||
The runner user must be able to run Docker commands. On a typical Ubuntu host:
|
||||
To use the bleeding-edge `:latest` (rebuilt on every overlay change + weekly):
|
||||
|
||||
```bash
|
||||
sudo usermod -aG docker "$USER"
|
||||
newgrp docker
|
||||
VLLM_IMAGE=ghcr.io/noonghunna/vllm-club3090:latest \
|
||||
bash scripts/launch.sh --variant vllm/dual
|
||||
```
|
||||
|
||||
Restart the runner service after changing group membership.
|
||||
|
||||
## Host Preflight
|
||||
|
||||
Run these on the runner host before enabling the smoke job:
|
||||
|
||||
```bash
|
||||
nvidia-smi
|
||||
docker compose version
|
||||
docker run --rm --gpus all nvidia/cuda:12.8.0-base-ubuntu24.04 nvidia-smi
|
||||
```
|
||||
|
||||
Then clone this repository at the path used by the runner workspace once and
|
||||
make sure the model cache is present or mounted at the compose default:
|
||||
|
||||
```bash
|
||||
ls -ld models-cache
|
||||
```
|
||||
|
||||
If your cache lives elsewhere, set `MODEL_DIR` in the runner service
|
||||
environment. The canonical compose reads `${MODEL_DIR:-../../../../../models-cache}`.
|
||||
|
||||
## What The Smoke Job Does
|
||||
|
||||
On a green build, the workflow:
|
||||
|
||||
1. Pulls `ghcr.io/noonghunna/vllm-club3090:nightly-YYYYMMDD-clubXXXX`.
|
||||
2. Boots `models/qwen3.6-27b/vllm/compose/dual/docker-compose.yml` with a
|
||||
temporary compose override that points at the dated image.
|
||||
3. Waits for `http://localhost:8010/v1/models`.
|
||||
4. Runs `bash scripts/verify-full.sh`.
|
||||
5. Runs a three-prompt OpenAI-compatible smoke bench.
|
||||
6. Only then retags the image as `latest` and `nightly-stable`.
|
||||
|
||||
If any smoke step fails, the dated image remains available for debugging and the
|
||||
rolling aliases do not move.
|
||||
|
||||
## Using The GHCR Image
|
||||
|
||||
The pre-built GHCR image is opt-in. Normal launches use the upstream vLLM
|
||||
nightly SHA resolved from `scripts/lib/profiles/engines/<engine-id>.yml`.
|
||||
|
||||
To force a verified club image after this workflow has moved `latest` forward:
|
||||
|
||||
```bash
|
||||
VLLM_IMAGE=ghcr.io/noonghunna/vllm-club3090:latest bash scripts/launch.sh --variant vllm/dual
|
||||
```
|
||||
|
||||
`VLLM_IMAGE` is a full image reference override. The launcher still exports
|
||||
`VLLM_IMAGE` is a full image-reference override. The launcher still exports
|
||||
`VLLM_NIGHTLY_SHA` from the matching EngineProfile, but Docker Compose uses
|
||||
`VLLM_IMAGE` first.
|
||||
`VLLM_IMAGE` first when present.
|
||||
|
||||
## Existing Containers
|
||||
## Workflow triggers
|
||||
|
||||
The runner should be dedicated to CI. Before booting the canonical compose, the
|
||||
workflow tears down the default club-3090 estate if `~/.club3090/estate.yml`
|
||||
exists, then runs `docker compose down` for the CI project name. Avoid running
|
||||
manual workloads on the same host while the workflow is active.
|
||||
The workflow runs on:
|
||||
|
||||
## Registry Permissions
|
||||
- **`workflow_dispatch`** — manual `gh workflow run build-vllm-image.yml` for ad-hoc rebuilds
|
||||
- **`schedule`** — weekly, Sunday 00:00 UTC
|
||||
- **`push`** to master when `.github/workflows/build-vllm-image.yml`, `docker/vllm-club3090/**`, or `models/*/vllm/patches/**` change
|
||||
- **`push`** of any tag matching `v0.7.*`, `v0.[8-9].*`, or `v[1-9]*` — adds the `:club-vX.Y.Z` release tag
|
||||
|
||||
The workflow uses `GITHUB_TOKEN` with `packages: write` to push GHCR images and
|
||||
move aliases. No personal access token is required for the repository-owned
|
||||
package.
|
||||
## Workflow permissions
|
||||
|
||||
`GITHUB_TOKEN` with default `contents: read` + `packages: write` is sufficient.
|
||||
No `administration: read` is needed (we removed the runner detection that
|
||||
required it). No personal access tokens required.
|
||||
|
||||
## Manual `:latest` bootstrap (recovery)
|
||||
|
||||
If `:latest` ever gets out of sync with the most-recent dated nightly (e.g.
|
||||
during this workflow's redesign migration), you can manually re-tag a dated
|
||||
nightly as `:latest` without re-building:
|
||||
|
||||
```bash
|
||||
docker login ghcr.io --username "${GITHUB_USERNAME}" --password-stdin <<< "${GITHUB_TOKEN}"
|
||||
|
||||
docker buildx imagetools create \
|
||||
-t ghcr.io/noonghunna/vllm-club3090:latest \
|
||||
-t ghcr.io/noonghunna/vllm-club3090:nightly-stable \
|
||||
ghcr.io/noonghunna/vllm-club3090:nightly-YYYYMMDD-clubXXXX
|
||||
```
|
||||
|
||||
Replace `nightly-YYYYMMDD-clubXXXX` with the dated tag you want to bless as
|
||||
`:latest`. List recent tags via:
|
||||
|
||||
```bash
|
||||
gh api -H 'Accept: application/vnd.github+json' \
|
||||
/users/noonghunna/packages/container/vllm-club3090/versions | \
|
||||
jq -r '.[] | .metadata.container.tags[]' | head -20
|
||||
```
|
||||
|
||||
(This requires `read:packages` scope on the token. Without it, list tags via the
|
||||
GHCR web UI: https://github.com/noonghunna/club-3090/pkgs/container/vllm-club3090.)
|
||||
|
||||
## Retention
|
||||
|
||||
The scheduled workflow keeps:
|
||||
|
||||
- `latest`
|
||||
- `nightly-stable`
|
||||
- every `club-v*` release tag
|
||||
- dated `nightly-YYYYMMDD-clubXXXX` tags from the last four weeks
|
||||
- `:latest`
|
||||
- `:nightly-stable`
|
||||
- every `:club-v*` release tag
|
||||
- dated `nightly-YYYYMMDD-clubNNNN` tags from the last four weeks
|
||||
|
||||
Older dated nightly package versions are deleted by the retention job.
|
||||
Older dated nightly versions are deleted by the retention job (runs only on the
|
||||
weekly schedule + `workflow_dispatch`, not on every push).
|
||||
|
||||
@@ -42,6 +42,57 @@ isn't your topology.
|
||||
|
||||
---
|
||||
|
||||
## Topology classification
|
||||
|
||||
The launcher classifies your selected hardware and emits strategy guidance
|
||||
when the cards are not matched. You can run the classifier without booting
|
||||
anything:
|
||||
|
||||
```bash
|
||||
bash scripts/launch.sh --topology
|
||||
```
|
||||
|
||||
Use `--gpus 0,1` or `--cards 2` with `--topology` if you only want advice
|
||||
for a subset.
|
||||
|
||||
| Class | What it means | Example | Recommended |
|
||||
|---|---|---|---|
|
||||
| `single_card` | 1 GPU detected | 1x RTX 3090 | Use the largest single-card compose that fits (`vllm/default`, `vllm/long-text`, `llamacpp/default`). |
|
||||
| `homogeneous` | All cards have matched VRAM and matched SM | 2x RTX 3090 | TP=N is the optimal default; use the shipped `vllm/dual*` or `vllm/dual4*` composes. |
|
||||
| `vram_matched_compute_mismatched` | Same VRAM, different compute tier | RTX 3090 + RTX 4090 | TP=N works correctly, but faster cards wait at NCCL allreduce. Estate planner is better for multi-model workloads. |
|
||||
| `vram_mismatched` | Different VRAM sizes | RTX 3060 12 GB + RTX 3090 24 GB | Prefer llama.cpp `--tensor-split`, manual PP=N experiments, or estate planner. Avoid TP=N across the full mismatched set. |
|
||||
| `heterogeneous_mixed` | Multiple VRAM and compute tiers | RTX 3060 + RTX 3090 + RTX 4090 | Manual selection. Run one model on the largest matched subset or use estate planner for separate endpoints. |
|
||||
|
||||
### Why TP=N is poor on VRAM-mismatched cards
|
||||
|
||||
Tensor parallelism splits weights evenly across cards. If one card has 24 GB
|
||||
and another has 12 GB, TP=2 still puts roughly half the model on each card.
|
||||
The smaller card becomes the hard ceiling for weights, KV cache, activations,
|
||||
and fragmentation. For Qwen 3.6 27B INT4, that usually leaves too little KV
|
||||
headroom to be useful.
|
||||
|
||||
For mismatched VRAM, the practical paths are:
|
||||
|
||||
- llama.cpp `--tensor-split` for weighted layer placement.
|
||||
- PP=N as a manual vLLM flag flip (`--pipeline-parallel-size N`) when you are
|
||||
deliberately experimenting. club-3090 does not ship a PP compose today.
|
||||
- Estate planner: `bash scripts/launch.sh --estate` runs different models on
|
||||
different card subsets without forcing one model across uneven VRAM.
|
||||
|
||||
### When compute-mismatched TP is fine
|
||||
|
||||
Matched VRAM with different SM, such as RTX 3090 + RTX 4090, is a different
|
||||
trade-off. TP=2 works because both cards have enough memory for the same model
|
||||
shard and KV budget. The cost is throughput: the faster card waits at NCCL
|
||||
allreduce barriers, so effective pair speed caps near the slower card. You
|
||||
preserve per-card VRAM capacity, but waste some compute on the faster card.
|
||||
|
||||
That is acceptable for one-model serving. If your goal is maximum aggregate
|
||||
throughput from two different cards, estate planner usually wins because each
|
||||
card runs its own model at full speed.
|
||||
|
||||
---
|
||||
|
||||
## Valid TP values for Qwen3.6-27B
|
||||
|
||||
vLLM's tensor parallelism splits attention heads across cards. The TP
|
||||
|
||||
+5
-4
@@ -57,14 +57,15 @@ image: ${VLLM_IMAGE:-vllm/vllm-openai:nightly-${VLLM_NIGHTLY_SHA}}
|
||||
`scripts/launch.sh`, `scripts/switch.sh`, and estate boot resolve
|
||||
`VLLM_NIGHTLY_SHA` from `scripts/lib/profiles/engines/<engine-id>.yml →
|
||||
install.spec`. `VLLM_IMAGE` is a full-image override for users who want to opt
|
||||
into the pre-built GHCR image after the CI smoke has moved `latest` forward.
|
||||
into the pre-built GHCR image. The `:club-vX.Y.Z` release tags are the
|
||||
recommended pinned target; `:latest` follows the most-recent dated nightly.
|
||||
|
||||
| Pin source | Composes using it | Reason for pin | Retirement candidate? |
|
||||
|---|---|---|---|
|
||||
| `scripts/lib/profiles/engines/vllm-nightly-mtp.yml` → `vllm/vllm-openai:nightly-1acd67a7...` | MTP vLLM composes | Post-#41745 nightly for Qwen and Gemma MTP paths. | Bump this YAML when a newer upstream nightly absorbs the required local fixes. |
|
||||
| `scripts/lib/profiles/engines/vllm-nightly-dflash.yml` → `vllm/vllm-openai:nightly-e47c98ef...` | DFlash vLLM composes | DFlash overlay baseline. | Bump this YAML after DFlash overlay drift is revalidated. |
|
||||
| `scripts/lib/profiles/engines/vllm-nightly-full.yml` → `vllm/vllm-openai:nightly-e47c98ef...` | full-overlay vLLM composes | INT8 PTH + DFlash coexistence overlay baseline. | Bump this YAML when #42102/#41703/#35936/#40361 land and the overlay surface shrinks. |
|
||||
| `VLLM_IMAGE=ghcr.io/noonghunna/vllm-club3090:latest` | opt-in override for any vLLM compose | v0.7.0 pre-built image path. CI vendors the overlay set and only moves `latest` / `nightly-stable` after GPU smoke passes. | Optional. It is not the default boot path. |
|
||||
| `scripts/lib/profiles/engines/vllm-nightly-full.yml` → `vllm/vllm-openai:nightly-e47c98ef...` | full-overlay vLLM composes | INT8 PTH + DFlash coexistence overlay baseline. **#42102 closed-as-slop 2026-05-15 → DFlash+quant-KV overlay is now permanent.** | Bump this YAML when #41703 / #35936 / #40361 land and the overlay surface shrinks. The #42102/#41559 coexistence patch stays vendored indefinitely. |
|
||||
| `VLLM_IMAGE=ghcr.io/noonghunna/vllm-club3090:club-vX.Y.Z` (or `:latest`) | opt-in override for any vLLM compose | Pre-built image with vendored overlays baked in. Release tags (`:club-v0.7.0`, etc.) are immutable; `:latest` follows the most-recent dated nightly. See `docs/CI_RUNNER_SETUP.md`. | Optional. It is not the default boot path. |
|
||||
| `ghcr.io/ggml-org/llama.cpp:server-cuda` | 2 (Qwen 3.6-27B llama-cpp) | Stable tag, no hash drift on upstream side. No patches mounted. | Not a retirement candidate — drift-free. Capture digest if reproducibility matters. |
|
||||
|
||||
**Retirement workflow:** see [`NIGHTLY_BUMP_RUNBOOK.md`](./NIGHTLY_BUMP_RUNBOOK.md).
|
||||
@@ -89,7 +90,7 @@ into the pre-built GHCR image after the CI smoke has moved `latest` forward.
|
||||
| [#40914](https://github.com/vllm-project/vllm/pull/40914) — Sandermage K+1 verify routing | 🟡 Open, ❌ negative on our Qwen3.6-27B stack | **Reframed 2026-05-11:** the synthetic `seq_lens` K+1 route is not the P67-equivalent we need here. Local rebase on post-#41434 nightly made MTP acceptance look perfect (AL=4.0 / ~100%) but produced `!`-flood needle corruption plus tool/multi-turn timeouts. Dropping it improved verify-stress from 3/7 to 5/7, but TQ3/TQ4/k8v4 + MTP still fail long-context needles. | Do not ship Genesis-free TQ+MTP on #40914 alone. Use `dual/tq3-nomtp.yml` without Genesis, or `dual/tq3-mtp-genesis.yml` with Genesis P67/P67b. |
|
||||
| [#40334](https://github.com/vllm-project/vllm/pull/40334) — DFlash `combine_hidden_states` dtype mismatch | 🟡 Open | All `dual-dflash*.yml` need `--dtype bfloat16` flag to work around. | Composes set `--dtype bfloat16`. Drop when this lands. |
|
||||
| [#40382](https://github.com/vllm-project/vllm/issues/40382) — Gemma-4 + DFlash unservable on Ampere | 🟠 Open, no fix in progress | Blocks DFlash on Gemma-4 family. Not directly our problem (we serve Qwen3.6) but tracked because future model adds may hit it. | None — different attention backend selection. |
|
||||
| **[#41559](https://github.com/vllm-project/vllm/issues/41559) — DFlash spec-decode incompatible with all KV cache quantization** (seantechco, filed 2026-05-03) | 🟢 **OUR FIX PR OPEN: [#42102](https://github.com/vllm-project/vllm/pull/42102)** (filed 2026-05-08) | **REFRAMED 2026-05-08 PM via Codex investigation**: original allowlist-gating framing was partially outdated on current main. Current state: FLASH_ATTN gates dynamically via `flash_attn_supports_fp8()` (FA3-only); FLEX_ATTENTION raises `NotImplementedError` on quantized KV at impl construction; TRITON_ATTN remains causal-only via `assert causal` at `triton_unified_attention.py:542`. The KV-quant write path itself (`triton_reshape_and_cache_flash_per_token_head_quant`) IS causal-mask-independent — but no current backend actually executes both quantized KV AND non-causal attention. **Sharper framing for the common case (BF16 DFlash drafter alongside quantized target KV)**: don't need any backend to "support quantized KV in non-causal mode" — just need the engine to stop forcing target+drafter to share a single page-size unify pass. Three-layer local fix at `/opt/ai/engines/vllm/primary` branch `dflash-noncausal-kv-quant` (commit `cfb8f711`, 4 files, +333/-35): (1) `vllm/v1/core/kv_cache_utils.py` partition DFlash drafter specs into independent KV groups before unify, allocator extended to size isolated tensors by their own page_size; (2) `vllm/model_executor/models/qwen3_dflash.py` override drafter cache_dtype to "auto" when engine global is quantized; (3) `vllm/v1/attention/backends/flash_attn.py` FA metadata scheduler uses per-spec dtype when spec's kv_quant_mode is NONE. **Validated end-to-end on dual 3090 Ampere**: Gemma 4 + z-lab DFlash drafter + INT8 PTH KV target boots HEALTHY at 65K, Paris smoke clean, narrative 95.89 / code 168.09 TPS (matches bf16 32K baseline within CV — long-context unlocked at zero perf cost), AL 5.0-5.3 long-ctx code preserved, NIAH PASS at 32K prompt, KV pool 149,345 tokens (4× lift over baseline). | Local commit `cfb8f711` ready for review + push to `noonghunna/vllm` fork + upstream PR submission. PR description draft at `/tmp/dflash-int8-pr-description.md` (covers non-duplication checks, AI-assistance disclosure, validation matrix). Forensic Phase 3a/3b stacks remain at `models/gemma-4-31b/vllm/patches/vllm-gemma4-dflash-int8/` as historical record (the wrong-fix path that helped diagnose). Container artifacts cleaned up; Qwen production restored. |
|
||||
| **[#41559](https://github.com/vllm-project/vllm/issues/41559) — DFlash spec-decode incompatible with all KV cache quantization** (seantechco, filed 2026-05-03) | ❌ **OUR FIX PR #42102 CLOSED AS SLOP** by @benchislett on 2026-05-15 (no comment, just `closed-as-slop` label). Issue #41559 still OPEN upstream. | Local fix preserved: three-layer patch (4 files, +333/-35) on branch `dflash-noncausal-kv-quant` (commits `cfb8f711` + `5cb61c60`). (1) `vllm/v1/core/kv_cache_utils.py` partitions DFlash drafter specs into independent KV groups before unify, allocator extended to size isolated tensors by their own page_size; (2) `vllm/model_executor/models/qwen3_dflash.py` overrides drafter cache_dtype to "auto" when engine global is quantized; (3) `vllm/v1/attention/backends/flash_attn.py` FA metadata scheduler uses per-spec dtype when spec's kv_quant_mode is NONE. **Validated locally on dual 3090 Ampere**: Gemma 4 + z-lab DFlash drafter + INT8 PTH KV target boots HEALTHY at 65K, narrative 95.89 / code 168.09 TPS, AL 5.0-5.3 preserved, NIAH PASS at 32K, KV pool 149,345 tokens (4× lift). | **Vendor permanently.** Patch lives at `models/gemma-4-31b/vllm/patches/vllm-gemma4-dflash-int8/` and is baked into `vllm-nightly-full` + `vllm-nightly-dflash` EngineProfiles. Re-engagement with upstream NOT recommended (vLLM has hardened anti-AI-PR policy). Watch issue #41559 for any newer maintainer-blessed PR; drop our overlay then. |
|
||||
| [#40354](https://github.com/vllm-project/vllm/issues/40354) — Marlin TP=2 W4A16 < 64 | ✅ Same root-cause as #40361 | Our PR #40361 resolves this. | See #40361 row. |
|
||||
| [#39931](https://github.com/vllm-project/vllm/issues/39931) — DeltaNet rollback support | 🔴 Open, architectural | Blocks **all** spec-decode (EAGLE / DFlash) on Qwen3-Next family across engines. The reason "speculative decoding doesn't work" on this stack. | Use MTP (no rollback needed) until this lands. |
|
||||
| [#40124](https://github.com/vllm-project/vllm/issues/40124) — related architectural | 🔴 Open | Pairs with #39931 for DeltaNet rollback. | Same as above. |
|
||||
|
||||
@@ -39,6 +39,19 @@ Memory budget: 14.5 GB (Q3_K_XL) + 4.5 GB KV @ 262K + 0.8 GB mmproj ≈ 20 GB /
|
||||
|
||||
Trade max context for parallelism. Same image, `--parallel 4` + smaller ctx pool.
|
||||
|
||||
### Tuning knobs
|
||||
|
||||
Both Docker composes expose llama.cpp's batch-size controls without editing YAML:
|
||||
|
||||
| Env var | llama.cpp flag | Default | Sensible range on 24 GB | Notes |
|
||||
|---|---|---:|---:|---|
|
||||
| `BATCH_SIZE` | `-b` | `4096` | `2048`-`8192` | Logical prompt-processing batch. Higher can improve prefill throughput if VRAM headroom allows. |
|
||||
| `UBATCH_SIZE` | `-ub` | `2048` | `1024`-`4096` | Physical microbatch. Lower this first if long prompts OOM during prefill. |
|
||||
|
||||
These are throughput-tuning knobs inside llama.cpp. They are orthogonal to
|
||||
`ESTATE_GPUS` and `ESTATE_PORT`, which only isolate GPU assignment and host port
|
||||
when `scripts/launch.sh --estate` boots multiple instances.
|
||||
|
||||
---
|
||||
|
||||
## Recipes (host-binary alternative)
|
||||
|
||||
@@ -59,6 +59,8 @@ services:
|
||||
-m /models/${GGUF_FILE:-qwen3.6-27b-gguf/unsloth-q3kxl/Qwen3.6-27B-UD-Q3_K_XL.gguf}
|
||||
--mmproj /models/${MMPROJ_FILE:-qwen3.6-27b-gguf/mmproj-F16.gguf}
|
||||
-c ${CTX_SIZE:-192000}
|
||||
-b ${BATCH_SIZE:-4096}
|
||||
-ub ${UBATCH_SIZE:-2048}
|
||||
-ngl 99
|
||||
-fa on
|
||||
--cache-type-k ${KV_TYPE:-q4_0}
|
||||
|
||||
@@ -56,6 +56,8 @@
|
||||
# GGUF_FILE path under /models (default: qwen3.6-27b-gguf/unsloth-q3kxl/Qwen3.6-27B-UD-Q3_K_XL.gguf)
|
||||
# MMPROJ_FILE path under /models (default: qwen3.6-27b-gguf/mmproj-F16.gguf)
|
||||
# CTX_SIZE total KV pool (default: 262144)
|
||||
# BATCH_SIZE llama.cpp -b (default: 4096)
|
||||
# UBATCH_SIZE llama.cpp -ub (default: 2048)
|
||||
# KV_TYPE K and V quant type (default: q4_0)
|
||||
# REASONING_FORMAT reasoning channel routing (default: none — for opencode/IDE-agent compat)
|
||||
# Override to `auto` to get separate `reasoning_content` field
|
||||
@@ -118,6 +120,10 @@ services:
|
||||
- /models/${MMPROJ_FILE:-qwen3.6-27b-gguf/mmproj-F16.gguf}
|
||||
- -c
|
||||
- ${CTX_SIZE:-262144}
|
||||
- -b
|
||||
- ${BATCH_SIZE:-4096}
|
||||
- -ub
|
||||
- ${UBATCH_SIZE:-2048}
|
||||
- -ngl
|
||||
- "99"
|
||||
- -fa
|
||||
|
||||
+134
-5
@@ -7,6 +7,7 @@
|
||||
# - per-run: wall time, TTFT (via streaming), completion tokens,
|
||||
# wall_TPS (= comp / wall), decode_TPS (= comp / (wall - TTFT))
|
||||
# - per-prompt summary: mean / std / CV for both TPS metrics + mean TTFT
|
||||
# + prompt-processing throughput (`PP tok/s`)
|
||||
# - shows MTP SpecDecoding metrics from docker logs at the end
|
||||
#
|
||||
# Why two TPS metrics:
|
||||
@@ -35,10 +36,15 @@
|
||||
# MAX_TOKENS_CODE Default: 800
|
||||
# ONLY Set to "narr" or "code" to skip the other. Default: both
|
||||
# QUIET Set to 1 to skip per-run lines (just print summary)
|
||||
# PP Set to 1 to add the long-prompt PP fallback probe.
|
||||
# llama.cpp containers enable this automatically.
|
||||
# PP_FALLBACK_TOKENS Approximate filler-token target for PP=1. Default: 10000
|
||||
# PP_MAX_TOKENS Completion cap for the PP fallback request. Default: 16
|
||||
#
|
||||
# Usage:
|
||||
# bash scripts/bench.sh
|
||||
# ONLY=code bash scripts/bench.sh
|
||||
# PP=1 bash scripts/bench.sh
|
||||
# RUNS=10 bash scripts/bench.sh
|
||||
|
||||
set -euo pipefail
|
||||
@@ -49,7 +55,7 @@ ROOT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
if [[ -f "${ROOT_DIR}/scripts/preflight.sh" ]]; then
|
||||
# shellcheck source=preflight.sh
|
||||
source "${ROOT_DIR}/scripts/preflight.sh"
|
||||
preflight_autodetect_endpoint
|
||||
preflight_autodetect_endpoint || true
|
||||
fi
|
||||
URL="${URL:-http://localhost:8020}"
|
||||
MODEL="${MODEL:-qwen3.6-27b-autoround}"
|
||||
@@ -62,6 +68,9 @@ PROMPT_NARR="${PROMPT_NARR:-Write a detailed 800-word essay explaining transform
|
||||
PROMPT_CODE="${PROMPT_CODE:-Write a Python implementation of quicksort with comments explaining each step.}"
|
||||
ONLY="${ONLY:-both}"
|
||||
QUIET="${QUIET:-0}"
|
||||
PP="${PP:-0}"
|
||||
PP_FALLBACK_TOKENS="${PP_FALLBACK_TOKENS:-10000}"
|
||||
PP_MAX_TOKENS="${PP_MAX_TOKENS:-16}"
|
||||
|
||||
need() {
|
||||
command -v "$1" >/dev/null 2>&1 || { echo "ERROR: '$1' not in PATH." >&2; exit 1; }
|
||||
@@ -69,6 +78,51 @@ need() {
|
||||
need curl
|
||||
need python3
|
||||
|
||||
ENGINE_KIND="${ENGINE_KIND:-unknown}"
|
||||
if [[ "$ENGINE_KIND" == "unknown" && "${CONTAINER:-}" != "none" ]] && command -v docker >/dev/null 2>&1 && docker inspect "${CONTAINER}" >/dev/null 2>&1; then
|
||||
container_image="$(docker inspect --format '{{.Config.Image}}' "${CONTAINER}" 2>/dev/null || true)"
|
||||
container_name="$(docker inspect --format '{{.Name}}' "${CONTAINER}" 2>/dev/null || true)"
|
||||
if [[ "${container_image} ${container_name}" == *"llama.cpp"* || "${container_image} ${container_name}" == *"llama-cpp"* ]]; then
|
||||
ENGINE_KIND="llamacpp"
|
||||
elif [[ "${container_image} ${container_name}" == *"vllm"* ]]; then
|
||||
ENGINE_KIND="vllm"
|
||||
fi
|
||||
fi
|
||||
|
||||
PP_MODE="log"
|
||||
if [[ "$PP" == "1" || "$ENGINE_KIND" == "llamacpp" ]]; then
|
||||
PP_MODE="fallback"
|
||||
fi
|
||||
|
||||
if [[ "${BENCH_MOCK:-0}" == "1" ]]; then
|
||||
if [[ "$PP_MODE" == "fallback" ]]; then
|
||||
cat <<'EOF'
|
||||
|
||||
========== PROMPT-PROCESSING (fallback target=10000 prompt tokens, max_tokens=16) ==========
|
||||
=== measured (1) ===
|
||||
run-1 wall= 3.20s ttft= 2500ms prompt_toks= 9876 PP_tok/s=3950.40
|
||||
|
||||
=== summary [prompt-processing] (n=1) ===
|
||||
PP tok/s mean=3950.40 std= 0.00 CV= 0.0% min=3950.40 max=3950.40
|
||||
TTFT mean= 2500ms std= 0ms min=2500ms max=2500ms
|
||||
EOF
|
||||
else
|
||||
cat <<'EOF'
|
||||
|
||||
========== NARRATIVE (prompt=61 chars, max_tokens=1000) ==========
|
||||
=== measured (1) ===
|
||||
run-1 wall= 4.20s ttft= 120ms toks=1000 wall_TPS=238.10 decode_TPS=245.10
|
||||
|
||||
=== summary [narrative] (n=1) ===
|
||||
wall_TPS mean= 238.10 std= 0.00 CV= 0.0% min=238.10 max=238.10
|
||||
decode_TPS mean= 245.10 std= 0.00 CV= 0.0% min=245.10 max=245.10
|
||||
TTFT mean= 120ms std= 0ms min=120ms max=120ms
|
||||
PP tok/s mean=2843.21 std= 0.00 CV= 0.0% min=2843.21 max=2843.21
|
||||
EOF
|
||||
fi
|
||||
exit 0
|
||||
fi
|
||||
|
||||
if ! curl -sf "${URL}/v1/models" >/dev/null; then
|
||||
echo "ERROR: service not reachable at ${URL}/v1/models" >&2
|
||||
echo " Start with: cd compose && docker compose up -d" >&2
|
||||
@@ -76,14 +130,17 @@ if ! curl -sf "${URL}/v1/models" >/dev/null; then
|
||||
fi
|
||||
|
||||
python3 - "$URL" "$MODEL" "$WARMUPS" "$RUNS" "$QUIET" "$ONLY" \
|
||||
"$CONTAINER" "$PP_MODE" "$PP_FALLBACK_TOKENS" "$PP_MAX_TOKENS" \
|
||||
"$PROMPT_NARR" "$MAX_TOKENS_NARR" \
|
||||
"$PROMPT_CODE" "$MAX_TOKENS_CODE" << 'PYEOF'
|
||||
import json, sys, time, urllib.request, statistics as s
|
||||
import json, re, shutil, subprocess, sys, time, urllib.request, statistics as s
|
||||
|
||||
(URL, MODEL, WARMUPS, RUNS, QUIET, ONLY,
|
||||
CONTAINER, PP_MODE, PP_FALLBACK_TOKENS, PP_MAX_TOKENS,
|
||||
PROMPT_NARR, MAX_NARR, PROMPT_CODE, MAX_CODE) = sys.argv[1:]
|
||||
WARMUPS = int(WARMUPS); RUNS = int(RUNS); QUIET = int(QUIET) == 1
|
||||
MAX_NARR = int(MAX_NARR); MAX_CODE = int(MAX_CODE)
|
||||
PP_FALLBACK_TOKENS = int(PP_FALLBACK_TOKENS); PP_MAX_TOKENS = int(PP_MAX_TOKENS)
|
||||
|
||||
def run_once(prompt, max_tokens):
|
||||
body = json.dumps({
|
||||
@@ -101,6 +158,7 @@ def run_once(prompt, max_tokens):
|
||||
t_send = time.time()
|
||||
ttft = None
|
||||
completion_tokens = 0
|
||||
prompt_tokens = 0
|
||||
with urllib.request.urlopen(req, timeout=600) as r:
|
||||
for line in r:
|
||||
line = line.decode("utf-8", errors="ignore").rstrip()
|
||||
@@ -122,11 +180,14 @@ def run_once(prompt, max_tokens):
|
||||
usage = chunk.get("usage")
|
||||
if usage:
|
||||
completion_tokens = usage.get("completion_tokens", completion_tokens)
|
||||
prompt_tokens = usage.get("prompt_tokens", prompt_tokens)
|
||||
t_end = time.time()
|
||||
wall = t_end - t_send
|
||||
if ttft is None:
|
||||
ttft = wall
|
||||
return wall, ttft, completion_tokens
|
||||
if not prompt_tokens:
|
||||
prompt_tokens = max(1, len(prompt.split()))
|
||||
return wall, ttft, completion_tokens, prompt_tokens
|
||||
|
||||
def fmt(label, wall, ttft, toks):
|
||||
decode_t = max(wall - ttft, 1e-6)
|
||||
@@ -135,18 +196,44 @@ def fmt(label, wall, ttft, toks):
|
||||
line = f" {label:<10s} wall={wall:6.2f}s ttft={ttft*1000:6.0f}ms toks={toks:>4d} wall_TPS={wtps:6.2f} decode_TPS={dtps:6.2f}"
|
||||
return wtps, dtps, ttft, line
|
||||
|
||||
def fmt_pp(label, wall, ttft, prompt_tokens):
|
||||
pp = prompt_tokens / max(ttft, 1e-6)
|
||||
line = f" {label:<10s} wall={wall:6.2f}s ttft={ttft*1000:6.0f}ms prompt_toks={prompt_tokens:>6d} PP_tok/s={pp:7.2f}"
|
||||
return pp, ttft, line
|
||||
|
||||
def stats(name, xs, unit=""):
|
||||
m = s.mean(xs)
|
||||
sd = s.stdev(xs) if len(xs) > 1 else 0
|
||||
cv = (sd / m * 100) if m > 0 else 0
|
||||
return f" {name:<14s} mean={m:7.2f}{unit} std={sd:6.2f} CV={cv:4.1f}% min={min(xs):.2f} max={max(xs):.2f}"
|
||||
|
||||
def scrape_prompt_throughput(container, n):
|
||||
if not container or container == "none" or shutil.which("docker") is None:
|
||||
return []
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["docker", "logs", container],
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
text=True,
|
||||
errors="replace",
|
||||
timeout=10,
|
||||
check=False,
|
||||
)
|
||||
except Exception:
|
||||
return []
|
||||
vals = [
|
||||
float(m.group(1))
|
||||
for m in re.finditer(r"Avg prompt throughput:\s*([0-9]+(?:\.[0-9]+)?)\s*tokens/s", proc.stdout)
|
||||
]
|
||||
return vals[-max(n, 1):]
|
||||
|
||||
def run_set(label, prompt, max_tokens):
|
||||
print(f"\n========== {label.upper()} (prompt={len(prompt)} chars, max_tokens={max_tokens}) ==========")
|
||||
print(f"=== warmups ({WARMUPS}) ===")
|
||||
for i in range(WARMUPS):
|
||||
try:
|
||||
w, t, k = run_once(prompt, max_tokens)
|
||||
w, t, k, _ = run_once(prompt, max_tokens)
|
||||
_, _, _, line = fmt(f"warm-{i+1}", w, t, k)
|
||||
if not QUIET:
|
||||
print(line)
|
||||
@@ -156,7 +243,7 @@ def run_set(label, prompt, max_tokens):
|
||||
walls, decodes, ttfts = [], [], []
|
||||
for i in range(RUNS):
|
||||
try:
|
||||
w, t, k = run_once(prompt, max_tokens)
|
||||
w, t, k, _ = run_once(prompt, max_tokens)
|
||||
wtps, dtps, ttft, line = fmt(f"run-{i+1}", w, t, k)
|
||||
if not QUIET:
|
||||
print(line)
|
||||
@@ -168,11 +255,53 @@ def run_set(label, prompt, max_tokens):
|
||||
print(stats("wall_TPS", walls))
|
||||
print(stats("decode_TPS", decodes))
|
||||
print(f" TTFT mean={s.mean(ttfts)*1000:6.0f}ms std={s.stdev(ttfts)*1000 if len(ttfts) > 1 else 0:5.0f}ms min={min(ttfts)*1000:.0f}ms max={max(ttfts)*1000:.0f}ms")
|
||||
if PP_MODE == "log":
|
||||
pp_vals = scrape_prompt_throughput(CONTAINER, len(walls))
|
||||
if pp_vals:
|
||||
print(stats("PP tok/s", pp_vals))
|
||||
else:
|
||||
print(" PP tok/s n/a (vLLM log scrape unavailable; use PP=1 for long-prompt fallback)")
|
||||
else:
|
||||
print(" PP tok/s n/a (long-prompt fallback below)")
|
||||
|
||||
def long_prompt(target_tokens):
|
||||
filler = (
|
||||
"club3090 prompt processing calibration filler with stable token shape. "
|
||||
"This sentence is intentionally plain so tokenizer variance stays modest. "
|
||||
)
|
||||
words_per_chunk = max(len(filler.split()), 1)
|
||||
chunks = max(1, target_tokens // words_per_chunk)
|
||||
return (
|
||||
"Read the following calibration text. Reply with one concise sentence summarizing its purpose.\n\n"
|
||||
+ filler * chunks
|
||||
)
|
||||
|
||||
def run_pp_fallback():
|
||||
prompt = long_prompt(PP_FALLBACK_TOKENS)
|
||||
print(
|
||||
f"\n========== PROMPT-PROCESSING "
|
||||
f"(fallback target={PP_FALLBACK_TOKENS} prompt tokens, max_tokens={PP_MAX_TOKENS}) =========="
|
||||
)
|
||||
print("=== measured (1) ===")
|
||||
pp_vals, ttfts = [], []
|
||||
try:
|
||||
w, t, _k, prompt_tokens = run_once(prompt, PP_MAX_TOKENS)
|
||||
pp, ttft, line = fmt_pp("run-1", w, t, prompt_tokens)
|
||||
print(line)
|
||||
pp_vals.append(pp); ttfts.append(ttft)
|
||||
except Exception as e:
|
||||
print(f" run-1 FAIL: {e}")
|
||||
if pp_vals:
|
||||
print("\n=== summary [prompt-processing] (n=1) ===")
|
||||
print(stats("PP tok/s", pp_vals))
|
||||
print(f" TTFT mean={s.mean(ttfts)*1000:6.0f}ms std= 0ms min={min(ttfts)*1000:.0f}ms max={max(ttfts)*1000:.0f}ms")
|
||||
|
||||
if ONLY in ("both", "narr"):
|
||||
run_set("narrative", PROMPT_NARR, MAX_NARR)
|
||||
if ONLY in ("both", "code"):
|
||||
run_set("code", PROMPT_CODE, MAX_CODE)
|
||||
if PP_MODE == "fallback":
|
||||
run_pp_fallback()
|
||||
PYEOF
|
||||
|
||||
# GPU state
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
# bash scripts/launch.sh --estate-file <path> # boot an existing estate plan
|
||||
# bash scripts/launch.sh --validate-estate <path> # validate estate.yml, no boot
|
||||
# bash scripts/launch.sh --down-estate <path> # stop estate instances
|
||||
# bash scripts/launch.sh --topology # print GPU topology advisory, no boot
|
||||
# bash scripts/launch.sh --model qwen3.6-27b --gpus 0,1
|
||||
# bash scripts/launch.sh --engine vllm --cards 1 # deprecated; prefer --gpus
|
||||
# bash scripts/launch.sh --workload long-ctx-single # profile-aware filter
|
||||
@@ -66,6 +67,7 @@ ESTATE_FILE=""
|
||||
VALIDATE_ESTATE=""
|
||||
DOWN_ESTATE=""
|
||||
ONLY_NAMES=""
|
||||
TOPOLOGY_ONLY=0
|
||||
CARDS=""
|
||||
VARIANT=""
|
||||
MODEL_NAME=""
|
||||
@@ -87,6 +89,7 @@ while [[ $# -gt 0 ]]; do
|
||||
--validate-estate) VALIDATE_ESTATE="$2"; shift 2 ;;
|
||||
--down-estate) DOWN_ESTATE="$2"; shift 2 ;;
|
||||
--only) ONLY_NAMES="$2"; shift 2 ;;
|
||||
--topology) TOPOLOGY_ONLY=1; SKIP_PREFLIGHT=1; shift ;;
|
||||
--engine) ENGINE="$2"; shift 2 ;;
|
||||
--workload) WORKLOAD_ID="$2"; shift 2 ;;
|
||||
--drafter) DRAFTER_ID="$2"; shift 2 ;;
|
||||
@@ -602,6 +605,77 @@ selected_gpu_profile_spec() {
|
||||
printf '%s' "$joined"
|
||||
}
|
||||
|
||||
select_topology_gpus() {
|
||||
GPU_LINES="$(compose_hw_detect_gpus 2>/dev/null || true)"
|
||||
[[ -n "$GPU_LINES" ]] || return 1
|
||||
CARD_INDICES=()
|
||||
CARD_NAMES=()
|
||||
CARD_MEM_MIB=()
|
||||
CARD_SM=()
|
||||
|
||||
if [[ -n "$GPU_ARG" && "$GPU_ARG" != "all" ]]; then
|
||||
IFS=',' read -ra _launch_topology_tokens <<< "$GPU_ARG"
|
||||
local idx
|
||||
for idx in "${_launch_topology_tokens[@]}"; do
|
||||
idx="$(_compose_meta_trim "$idx")"
|
||||
[[ -z "$idx" ]] && continue
|
||||
gpu_exists "$idx" || { echo "[launch] ERROR: requested GPU ${idx}, but it was not detected." >&2; exit 1; }
|
||||
append_selected_gpu "$idx"
|
||||
done
|
||||
elif [[ -n "$CARDS" ]]; then
|
||||
[[ "$CARDS" =~ ^[0-9]+$ && "$CARDS" -ge 1 ]] || { echo "[launch] ERROR: --cards expects a positive integer." >&2; exit 1; }
|
||||
local idx name mem_mib sm selected=0
|
||||
while IFS=$'\t' read -r idx name mem_mib sm; do
|
||||
[[ -z "$idx" ]] && continue
|
||||
append_selected_gpu "$idx"
|
||||
selected=$((selected + 1))
|
||||
(( selected >= CARDS )) && break
|
||||
done <<< "$GPU_LINES"
|
||||
(( selected == CARDS )) || { echo "[launch] ERROR: --cards ${CARDS} requested, but only ${selected} GPU(s) were detected." >&2; exit 1; }
|
||||
else
|
||||
local idx name mem_mib sm
|
||||
while IFS=$'\t' read -r idx name mem_mib sm; do
|
||||
[[ -z "$idx" ]] && continue
|
||||
append_selected_gpu "$idx"
|
||||
done <<< "$GPU_LINES"
|
||||
fi
|
||||
|
||||
[[ "${#CARD_INDICES[@]}" -gt 0 ]] || return 1
|
||||
SELECTED_GPU_CSV="$(IFS=','; echo "${CARD_INDICES[*]}")"
|
||||
summarize_selected_vram >/dev/null
|
||||
return 0
|
||||
}
|
||||
|
||||
print_topology_advisory() {
|
||||
local output
|
||||
output="$(python3 "$LAUNCH_PROFILE" topology --gpu-spec "$(selected_gpu_profile_spec)" --format wizard 2>&1)" || {
|
||||
echo "$output" >&2
|
||||
exit 2
|
||||
}
|
||||
if [[ -n "$output" ]]; then
|
||||
echo "$output" >&2
|
||||
fi
|
||||
}
|
||||
|
||||
print_topology_and_exit() {
|
||||
local output
|
||||
if ! select_topology_gpus; then
|
||||
echo "Detected hardware:"
|
||||
echo " no NVIDIA GPUs detected"
|
||||
echo ""
|
||||
echo "Topology class: unavailable"
|
||||
echo ""
|
||||
echo "For details, see docs/MULTI_CARD.md."
|
||||
exit 0
|
||||
fi
|
||||
output="$(python3 "$LAUNCH_PROFILE" topology --gpu-spec "$(selected_gpu_profile_spec)" --format standalone 2>&1)" || {
|
||||
echo "$output" >&2
|
||||
exit 0
|
||||
}
|
||||
echo "$output"
|
||||
exit 0
|
||||
}
|
||||
|
||||
launch_nvlink_active() {
|
||||
if [[ "${#CARD_INDICES[@]}" -ne 2 ]]; then
|
||||
printf '0'
|
||||
@@ -932,6 +1006,10 @@ if [[ "$ESTATE_MODE" -eq 1 || -n "$ESTATE_FILE" ]]; then
|
||||
exit $?
|
||||
fi
|
||||
|
||||
if [[ "$TOPOLOGY_ONLY" -eq 1 ]]; then
|
||||
print_topology_and_exit
|
||||
fi
|
||||
|
||||
# --- wizard ---
|
||||
if [[ -z "$VARIANT" ]]; then
|
||||
echo "" >&2
|
||||
@@ -939,6 +1017,7 @@ if [[ -z "$VARIANT" ]]; then
|
||||
echo "(Use --variant <name> next time to skip the wizard.)" >&2
|
||||
choose_model
|
||||
choose_gpus
|
||||
print_topology_advisory
|
||||
pick_parallelism
|
||||
if [[ "$MODEL_NAME" == "gemma-4-31b" && "${#CARD_INDICES[@]}" -eq 1 && "$MIN_VRAM_GB" -lt 32 ]]; then
|
||||
gemma_single_24gb_guidance
|
||||
|
||||
@@ -266,6 +266,26 @@ def peak_vram(internal: dict[str, Any], gpu_count: int) -> str:
|
||||
return f"{max(vals) / 1024:.1f} GB{suffix}"
|
||||
|
||||
|
||||
def pp_display(bench: dict[str, Any]) -> str:
|
||||
vals: list[float] = []
|
||||
for kind in ("narrative", "code"):
|
||||
try:
|
||||
val = bench.get(kind, {}).get("pp_tps_mean")
|
||||
if val is not None:
|
||||
vals.append(float(val))
|
||||
except Exception:
|
||||
pass
|
||||
if vals:
|
||||
return f"{sum(vals) / len(vals):.0f}"
|
||||
try:
|
||||
val = bench.get("prompt_processing", {}).get("pp_tps_mean")
|
||||
if val is not None:
|
||||
return f"{float(val):.0f}"
|
||||
except Exception:
|
||||
pass
|
||||
return "—"
|
||||
|
||||
|
||||
def parse_jsonish(value: str) -> dict[str, Any]:
|
||||
try:
|
||||
parsed = json.loads(value)
|
||||
@@ -370,6 +390,7 @@ def format_row(data: dict[str, Any]) -> str:
|
||||
kv = kv_display(c["kv"], c["served"])
|
||||
max_ctx = fmt_ctx(c["max_ctx"])
|
||||
tps = f"**{fmt_tps(narrative.get('wall_tps_mean'))} / {fmt_tps(code.get('wall_tps_mean'))}**"
|
||||
pp = pp_display(bench)
|
||||
peak = peak_vram(internal, gpu_count)
|
||||
spec_n = c["spec"].get("num_speculative_tokens")
|
||||
notes = [soak_note(soak), f"verify-stress {verify}"]
|
||||
@@ -393,12 +414,12 @@ def format_row(data: dict[str, Any]) -> str:
|
||||
per_pos = str(mtp.get("per_position") or "—")
|
||||
return (
|
||||
f"| `{compose}` | {rig_cell(rig)} | {kv} | {max_ctx} | {tps} | "
|
||||
f"{al} | {per_pos} | {peak} | {data['date']} | {note_cell} |"
|
||||
f"{pp} | {al} | {per_pos} | {peak} | {data['date']} | {note_cell} |"
|
||||
)
|
||||
|
||||
return (
|
||||
f"| `{compose}` | {rig_cell(rig)} | {kv} | {max_ctx} | {tps} | "
|
||||
f"{peak} | {data['date']} | {note_cell} |"
|
||||
f"{pp} | {peak} | {data['date']} | {note_cell} |"
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -12,6 +12,7 @@ import os
|
||||
import subprocess
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
|
||||
@@ -26,7 +27,7 @@ from .compose_registry import COMPOSE_REGISTRY
|
||||
SUPPORTED_SCHEMA_VERSIONS = {1}
|
||||
PROFILE_ROOT = Path(__file__).resolve().parent
|
||||
REPO_ROOT = Path(__file__).resolve().parents[3]
|
||||
CONSTRAINT_IDS = [f"C{i}" for i in range(1, 16)]
|
||||
CONSTRAINT_IDS = [f"C{i}" for i in range(1, 17)]
|
||||
ESTATE_CONSTRAINT_IDS = [f"E{i}" for i in range(1, 5)]
|
||||
|
||||
|
||||
@@ -42,6 +43,37 @@ class CrossReferenceError(ProfileError):
|
||||
"""Raised when a profile references a missing profile id."""
|
||||
|
||||
|
||||
class TopologyClass(str, Enum):
|
||||
SINGLE_CARD = "single_card"
|
||||
HOMOGENEOUS = "homogeneous"
|
||||
VRAM_MATCHED_COMPUTE_MISMATCHED = "vram_matched_compute_mismatched"
|
||||
VRAM_MISMATCHED = "vram_mismatched"
|
||||
HETEROGENEOUS_MIXED = "heterogeneous_mixed"
|
||||
|
||||
|
||||
TOPOLOGY_ADVISORY = {
|
||||
TopologyClass.SINGLE_CARD: None,
|
||||
TopologyClass.HOMOGENEOUS: None,
|
||||
TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED: (
|
||||
"Compute mismatch detected (VRAM matched). TP=N works fine but the faster card "
|
||||
"waits at every NCCL allreduce — effective throughput caps at slower card's speed "
|
||||
"(~30% of faster card idle at allreduce). Full per-card VRAM capacity preserved. "
|
||||
"Alternative: estate planner (--estate) to run different models per card at full speed."
|
||||
),
|
||||
TopologyClass.VRAM_MISMATCHED: (
|
||||
"VRAM mismatch detected. TP=N would cap to smaller card's usable model size. "
|
||||
"Recommended paths: (a) llama.cpp `--tensor-split` for weighted layer split, "
|
||||
"(b) PP=N (manual flag flip — `--pipeline-parallel-size N` on a vllm/dual compose; "
|
||||
"no shipping PP compose), (c) estate planner (--estate) to run different models per card."
|
||||
),
|
||||
TopologyClass.HETEROGENEOUS_MIXED: (
|
||||
"Heterogeneous hardware detected (multiple VRAM and compute tiers). Manual selection "
|
||||
"recommended. Consider the estate planner (--estate) to put different models on "
|
||||
"different card subsets, or run a single model on the largest matched subset."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def _logger() -> logging.Logger:
|
||||
logger = logging.getLogger("compat")
|
||||
if not logger.handlers:
|
||||
@@ -92,6 +124,30 @@ class HardwareProfile:
|
||||
notes: Optional[str] = None
|
||||
|
||||
|
||||
def classify_hardware_topology(hardware: list[HardwareProfile]) -> TopologyClass:
|
||||
"""Classify selected GPUs for TP-vs-PP/estate advisory output."""
|
||||
if not hardware:
|
||||
raise ProfileError("classify_hardware_topology requires at least one HardwareProfile")
|
||||
if len(hardware) == 1:
|
||||
return TopologyClass.SINGLE_CARD
|
||||
|
||||
vrams = sorted(hw.vram_gb for hw in hardware)
|
||||
sms = {hw.sm for hw in hardware}
|
||||
|
||||
vram_clusters = 1
|
||||
for i in range(1, len(vrams)):
|
||||
if vrams[i] - vrams[i - 1] > 1.0:
|
||||
vram_clusters += 1
|
||||
|
||||
if vram_clusters == 1 and len(sms) == 1:
|
||||
return TopologyClass.HOMOGENEOUS
|
||||
if vram_clusters == 1 and len(sms) > 1:
|
||||
return TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED
|
||||
if vram_clusters > 1:
|
||||
return TopologyClass.VRAM_MISMATCHED
|
||||
return TopologyClass.HETEROGENEOUS_MIXED
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ModelProfile:
|
||||
schema_version: int
|
||||
@@ -203,6 +259,7 @@ class FitsResult:
|
||||
world_size: Optional[int] = None
|
||||
bottleneck_vram_gb: Optional[float] = None
|
||||
homogeneous: Optional[bool] = None
|
||||
topology_class: Optional[TopologyClass] = None
|
||||
kv_projection: Optional[dict[str, Any]] = None
|
||||
compose_name: Optional[str] = None
|
||||
weights_variant: Optional[str] = None
|
||||
@@ -686,6 +743,7 @@ def fits(
|
||||
effective_max_num_seqs = max_num_seqs if max_num_seqs is not None else int(workload.defaults.get("max_num_seqs", 1))
|
||||
effective_weights = resolve_weights_variant(model, engine, weights_variant)
|
||||
homogeneous = len({hw.id for hw in hardware}) <= 1
|
||||
topology_class = classify_hardware_topology(hardware) if hardware else None
|
||||
bottleneck = min((hw.vram_gb for hw in hardware), default=None)
|
||||
effective_cudagraph = _cudagraph_mode(hardware)
|
||||
|
||||
@@ -830,6 +888,14 @@ def fits(
|
||||
else:
|
||||
ok("C12")
|
||||
|
||||
if topology_class is None:
|
||||
skip("C16", "Topology advisory not run; no hardware profiles provided.")
|
||||
else:
|
||||
ok("C16")
|
||||
advisory = TOPOLOGY_ADVISORY.get(topology_class)
|
||||
if advisory:
|
||||
notes.append(f"C16 topology={topology_class.value}: {advisory}")
|
||||
|
||||
diagnostics = {
|
||||
"constraints_evaluated": list(CONSTRAINT_IDS),
|
||||
"constraints_passed": passed,
|
||||
@@ -850,6 +916,7 @@ def fits(
|
||||
world_size=world_size,
|
||||
bottleneck_vram_gb=bottleneck,
|
||||
homogeneous=homogeneous,
|
||||
topology_class=topology_class,
|
||||
kv_projection=kv_projection,
|
||||
weights_variant=effective_weights,
|
||||
diagnostics=diagnostics,
|
||||
|
||||
@@ -19,7 +19,16 @@ if str(REPO_ROOT) not in sys.path:
|
||||
|
||||
os.environ.setdefault("CLUB3090_LOG_LEVEL", "ERROR")
|
||||
|
||||
from scripts.lib.profiles.compat import FitsResult, ProfileError, fits, load_profiles, to_compose_name # noqa: E402
|
||||
from scripts.lib.profiles.compat import ( # noqa: E402
|
||||
TOPOLOGY_ADVISORY,
|
||||
FitsResult,
|
||||
ProfileError,
|
||||
TopologyClass,
|
||||
classify_hardware_topology,
|
||||
fits,
|
||||
load_profiles,
|
||||
to_compose_name,
|
||||
)
|
||||
from scripts.lib.profiles.compose_registry import COMPOSE_REGISTRY # noqa: E402
|
||||
|
||||
|
||||
@@ -103,6 +112,26 @@ def _parse_gpu_specs(value: str, profiles) -> list:
|
||||
return hardware
|
||||
|
||||
|
||||
def _parse_gpu_specs_with_indices(value: str, profiles) -> list[tuple[str, object]]:
|
||||
hardware = []
|
||||
for raw in value.split(";"):
|
||||
raw = raw.strip()
|
||||
if not raw:
|
||||
continue
|
||||
try:
|
||||
idx, name, mem_mib, sm = raw.split("|", 3)
|
||||
except ValueError as exc:
|
||||
raise LaunchCompatError(f"invalid --gpu-spec entry `{raw}`") from exc
|
||||
hardware_id = _hardware_id_from_gpu(name, int(mem_mib), float(sm))
|
||||
try:
|
||||
hardware.append((idx, profiles.hardware[hardware_id]))
|
||||
except KeyError as exc:
|
||||
raise LaunchCompatError(f"hardware profile `{hardware_id}` is not installed") from exc
|
||||
if not hardware:
|
||||
raise LaunchCompatError("no GPU specs were provided for topology classification")
|
||||
return hardware
|
||||
|
||||
|
||||
def _engine_family(engine_type: str) -> str:
|
||||
return "llamacpp" if engine_type == "llama.cpp" else engine_type
|
||||
|
||||
@@ -364,6 +393,90 @@ def command_resolve_variant_pin(args: argparse.Namespace) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
def _hardware_line(index: str, hardware) -> str:
|
||||
return f" GPU {index}: {hardware.display_name} ({hardware.vram_gb:g} GB, sm {hardware.sm:g})"
|
||||
|
||||
|
||||
def _standalone_recommendation(topology: TopologyClass, count: int) -> list[str]:
|
||||
if topology == TopologyClass.SINGLE_CARD:
|
||||
return [
|
||||
"Recommended:",
|
||||
" 1. Use the largest single-card compose your model fits.",
|
||||
" 2. Add another matched card for TP=2 when long-context concurrency matters.",
|
||||
]
|
||||
if topology == TopologyClass.HOMOGENEOUS:
|
||||
return [
|
||||
"Recommended:",
|
||||
f" 1. TP={count} is the default path for matched cards; use the shipped vllm/dual* or multi-card composes.",
|
||||
" 2. Estate planner remains useful when you want separate models/endpoints instead of one larger TP instance.",
|
||||
]
|
||||
if topology == TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED:
|
||||
return [
|
||||
"Recommended:",
|
||||
f" 1. TP={count} works as-is. Compute mismatch means the faster card waits at every NCCL allreduce; effective throughput caps at the slower card's speed (~30% of faster card idle). Full per-card VRAM capacity preserved.",
|
||||
" 2. Estate planner — `bash scripts/launch.sh --estate` runs different models per card, each at full speed.",
|
||||
"",
|
||||
"Not recommended:",
|
||||
" - PP=N: possible as a manual flag flip (`--pipeline-parallel-size N`) on a vllm/dual compose, but no PP compose ships today.",
|
||||
]
|
||||
if topology == TopologyClass.VRAM_MISMATCHED:
|
||||
return [
|
||||
"Recommended:",
|
||||
" 1. llama.cpp `--tensor-split` for weighted layer split on mismatched VRAM.",
|
||||
" 2. PP=N as a manual vLLM flag flip (`--pipeline-parallel-size N`) if you are deliberately experimenting.",
|
||||
" 3. Estate planner — run different models per card or use the largest matched subset.",
|
||||
"",
|
||||
"Not recommended:",
|
||||
" - TP=N on the full mismatched set: the smaller card caps usable model size and KV headroom.",
|
||||
]
|
||||
return [
|
||||
"Recommended:",
|
||||
" 1. Manual selection. Use the largest matched subset for one model.",
|
||||
" 2. Estate planner — put different models on different card subsets.",
|
||||
]
|
||||
|
||||
|
||||
def command_topology(args: argparse.Namespace) -> int:
|
||||
_quiet_compat_logger()
|
||||
profiles = load_profiles()
|
||||
indexed_hardware = _parse_gpu_specs_with_indices(args.gpu_spec, profiles)
|
||||
hardware = [item[1] for item in indexed_hardware]
|
||||
topology = classify_hardware_topology(hardware)
|
||||
advisory = TOPOLOGY_ADVISORY.get(topology)
|
||||
|
||||
if args.format == "wizard":
|
||||
if topology in (TopologyClass.SINGLE_CARD, TopologyClass.HOMOGENEOUS):
|
||||
return 0
|
||||
detected = " + ".join(
|
||||
f"1x {hw.display_name} ({hw.vram_gb:g} GB, sm {hw.sm:g})"
|
||||
for _idx, hw in indexed_hardware
|
||||
)
|
||||
print(f"Detected: {detected}")
|
||||
print("")
|
||||
print(f"Topology: {topology.value}")
|
||||
if advisory:
|
||||
print(f" {advisory}")
|
||||
print("")
|
||||
print("Continue with the selected parallelism if that trade-off is acceptable.")
|
||||
return 0
|
||||
|
||||
print("Detected hardware:")
|
||||
for idx, hw in indexed_hardware:
|
||||
print(_hardware_line(idx, hw))
|
||||
print("")
|
||||
print(f"Topology class: {topology.value}")
|
||||
print("")
|
||||
for line in _standalone_recommendation(topology, len(hardware)):
|
||||
print(line)
|
||||
print("")
|
||||
if advisory:
|
||||
print("Advisory:")
|
||||
print(f" {advisory}")
|
||||
print("")
|
||||
print("For details, see docs/MULTI_CARD.md.")
|
||||
return 0
|
||||
|
||||
|
||||
def build_parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(description="Profile bridge for scripts/launch.sh")
|
||||
sub = parser.add_subparsers(dest="command", required=True)
|
||||
@@ -404,6 +517,11 @@ def build_parser() -> argparse.ArgumentParser:
|
||||
variant_pin.add_argument("--format", choices=("shell", "json", "value"), default="shell")
|
||||
variant_pin.set_defaults(func=command_resolve_variant_pin)
|
||||
|
||||
topology = sub.add_parser("topology")
|
||||
topology.add_argument("--gpu-spec", required=True)
|
||||
topology.add_argument("--format", choices=("standalone", "wizard"), default="standalone")
|
||||
topology.set_defaults(func=command_topology)
|
||||
|
||||
return parser
|
||||
|
||||
|
||||
|
||||
+52
-19
@@ -140,23 +140,34 @@ def parse_vllm_boot(boot_log: str) -> dict:
|
||||
def parse_bench(log: str) -> dict:
|
||||
"""Extract narrative + code TPS summaries from bench.sh output."""
|
||||
out: dict[str, Any] = {}
|
||||
# narrative summary block
|
||||
for kind in ("narrative", "code"):
|
||||
m = re.search(
|
||||
for kind in ("narrative", "code", "prompt-processing"):
|
||||
block = re.search(
|
||||
rf"=== summary \[{kind}\] \(n=\d+\) ===\s*\n"
|
||||
rf"\s+wall_TPS\s+mean=\s*([\d.]+)\s+std=\s*([\d.]+)\s+CV=\s*([\d.]+)%.*\n"
|
||||
rf"\s+decode_TPS\s+mean=\s*([\d.]+)\s+std=\s*([\d.]+)\s+CV=\s*([\d.]+)%.*\n"
|
||||
rf"\s+TTFT\s+mean=\s*([\d.]+)ms",
|
||||
rf"(?P<body>.*?)(?=\n==========|\n=== GPU state ===|\n=== Last|\Z)",
|
||||
log,
|
||||
re.DOTALL,
|
||||
)
|
||||
if not block:
|
||||
continue
|
||||
body = block.group("body")
|
||||
parsed: dict[str, Any] = {}
|
||||
m = re.search(r"\s+wall_TPS\s+mean=\s*([\d.]+)\s+std=\s*([\d.]+)\s+CV=\s*([\d.]+)%", body)
|
||||
if m:
|
||||
out[kind] = {
|
||||
"wall_tps_mean": float(m.group(1)),
|
||||
"wall_tps_cv": float(m.group(3)),
|
||||
"decode_tps_mean": float(m.group(4)),
|
||||
"decode_tps_cv": float(m.group(6)),
|
||||
"ttft_ms_mean": float(m.group(7)),
|
||||
}
|
||||
parsed["wall_tps_mean"] = float(m.group(1))
|
||||
parsed["wall_tps_cv"] = float(m.group(3))
|
||||
m = re.search(r"\s+decode_TPS\s+mean=\s*([\d.]+)\s+std=\s*([\d.]+)\s+CV=\s*([\d.]+)%", body)
|
||||
if m:
|
||||
parsed["decode_tps_mean"] = float(m.group(1))
|
||||
parsed["decode_tps_cv"] = float(m.group(3))
|
||||
m = re.search(r"\s+TTFT\s+mean=\s*([\d.]+)ms", body)
|
||||
if m:
|
||||
parsed["ttft_ms_mean"] = float(m.group(1))
|
||||
m = re.search(r"\s+PP tok/s\s+mean=\s*([\d.]+)\s+std=\s*([\d.]+)\s+CV=\s*([\d.]+)%", body)
|
||||
if m:
|
||||
parsed["pp_tps_mean"] = float(m.group(1))
|
||||
parsed["pp_tps_cv"] = float(m.group(3))
|
||||
if parsed:
|
||||
out[kind.replace("-", "_")] = parsed
|
||||
# GPU state at end
|
||||
gpu_block = re.search(r"=== GPU state ===\s*\n((?:\d.+\n){1,4})", log)
|
||||
if gpu_block:
|
||||
@@ -396,15 +407,22 @@ def render(report: dict) -> str:
|
||||
lines.append("## Performance — `bench.sh`")
|
||||
lines.append("")
|
||||
if bench.get("narrative") or bench.get("code"):
|
||||
lines.append("| Bench | wall TPS | decode TPS | TTFT | CV (wall/decode) |")
|
||||
lines.append("|---|---:|---:|---:|---:|")
|
||||
lines.append("| Bench | wall TPS | decode TPS | PP tok/s | TTFT | CV (wall/decode) |")
|
||||
lines.append("|---|---:|---:|---:|---:|---:|")
|
||||
for kind in ("narrative", "code"):
|
||||
b = bench.get(kind)
|
||||
if b:
|
||||
pp = f"{b['pp_tps_mean']:.0f}" if b.get("pp_tps_mean") is not None else "n/a"
|
||||
lines.append(
|
||||
f"| {kind} | {b['wall_tps_mean']:.2f} | **{b['decode_tps_mean']:.2f}** | "
|
||||
f"| {kind} | {b['wall_tps_mean']:.2f} | **{b['decode_tps_mean']:.2f}** | {pp} | "
|
||||
f"{b['ttft_ms_mean']:.0f} ms | {b['wall_tps_cv']:.1f}% / {b['decode_tps_cv']:.1f}% |"
|
||||
)
|
||||
pp_fallback = bench.get("prompt_processing")
|
||||
if pp_fallback and pp_fallback.get("pp_tps_mean") is not None:
|
||||
lines.append(
|
||||
f"| prompt-processing fallback | — | — | **{pp_fallback['pp_tps_mean']:.0f}** | "
|
||||
f"{pp_fallback.get('ttft_ms_mean', 0):.0f} ms | — |"
|
||||
)
|
||||
if bench.get("mtp"):
|
||||
m = bench["mtp"]
|
||||
lines.append("")
|
||||
@@ -600,9 +618,14 @@ def render_discuss(report: dict) -> str:
|
||||
if bench.get("narrative"):
|
||||
b_n, b_c = bench["narrative"], bench.get("code", {})
|
||||
lines.append("**TPS:**")
|
||||
lines.append(f"- Narrative: {b_n['decode_tps_mean']:.1f} TPS decode, {b_n['ttft_ms_mean']:.0f} ms TTFT (CV {b_n['decode_tps_cv']:.1f}%)")
|
||||
pp_n = f", PP {b_n['pp_tps_mean']:.0f} tok/s" if b_n.get("pp_tps_mean") is not None else ""
|
||||
lines.append(f"- Narrative: {b_n['decode_tps_mean']:.1f} TPS decode, {b_n['ttft_ms_mean']:.0f} ms TTFT{pp_n} (CV {b_n['decode_tps_cv']:.1f}%)")
|
||||
if b_c:
|
||||
lines.append(f"- Code: {b_c['decode_tps_mean']:.1f} TPS decode, {b_c['ttft_ms_mean']:.0f} ms TTFT (CV {b_c['decode_tps_cv']:.1f}%)")
|
||||
pp_c = f", PP {b_c['pp_tps_mean']:.0f} tok/s" if b_c.get("pp_tps_mean") is not None else ""
|
||||
lines.append(f"- Code: {b_c['decode_tps_mean']:.1f} TPS decode, {b_c['ttft_ms_mean']:.0f} ms TTFT{pp_c} (CV {b_c['decode_tps_cv']:.1f}%)")
|
||||
if bench.get("prompt_processing", {}).get("pp_tps_mean") is not None:
|
||||
pp = bench["prompt_processing"]["pp_tps_mean"]
|
||||
lines.append(f"- Prompt-processing fallback: {pp:.0f} tok/s")
|
||||
lines.append("")
|
||||
boot = report.get("vllm_boot", {})
|
||||
if boot.get("kv_cache_tokens"):
|
||||
@@ -641,11 +664,21 @@ def compute_tldr(report: dict) -> list[str]:
|
||||
bullets = []
|
||||
bench = report.get("bench", {})
|
||||
if bench.get("narrative") and bench.get("code"):
|
||||
pp_vals = [
|
||||
b.get("pp_tps_mean")
|
||||
for b in (bench.get("narrative", {}), bench.get("code", {}))
|
||||
if b.get("pp_tps_mean") is not None
|
||||
]
|
||||
pp_text = f", PP {sum(pp_vals) / len(pp_vals):.0f} tok/s" if pp_vals else ""
|
||||
bullets.append(
|
||||
f"TPS narrative **{bench['narrative']['decode_tps_mean']:.1f}** / "
|
||||
f"code **{bench['code']['decode_tps_mean']:.1f}** "
|
||||
f"(TTFT {bench['narrative']['ttft_ms_mean']:.0f}/{bench['code']['ttft_ms_mean']:.0f} ms)."
|
||||
f"(TTFT {bench['narrative']['ttft_ms_mean']:.0f}/{bench['code']['ttft_ms_mean']:.0f} ms{pp_text})."
|
||||
)
|
||||
elif bench.get("prompt_processing"):
|
||||
pp = bench["prompt_processing"].get("pp_tps_mean")
|
||||
if pp is not None:
|
||||
bullets.append(f"Prompt processing fallback: **{pp:.0f} tok/s**.")
|
||||
boot = report.get("vllm_boot", {})
|
||||
if boot.get("kv_cache_tokens") and boot.get("max_concurrency"):
|
||||
bullets.append(
|
||||
|
||||
@@ -49,6 +49,77 @@ assert r.recommended_kv_format == "turboquant_3bit_nc"
|
||||
assert r.diagnostics["constraints_skipped"] == ["C12"]
|
||||
PY
|
||||
|
||||
run_test "topology: single card classified" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"]])
|
||||
assert r == TopologyClass.SINGLE_CARD
|
||||
PY
|
||||
|
||||
run_test "topology: 2x3090 classified homogeneous" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"], p.hardware["rtx-3090"]])
|
||||
assert r == TopologyClass.HOMOGENEOUS
|
||||
PY
|
||||
|
||||
run_test "topology: 3090+4090 classified compute-mismatched" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"], p.hardware["rtx-4090"]])
|
||||
assert r == TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED
|
||||
PY
|
||||
|
||||
run_test "topology: 3090+3060 classified VRAM-mismatched" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"], p.hardware["rtx-3060-12gb"]])
|
||||
assert r == TopologyClass.VRAM_MISMATCHED
|
||||
PY
|
||||
|
||||
run_test "topology: VRAM cluster wins over mixed compute" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"], p.hardware["rtx-3060-12gb"], p.hardware["rtx-4090"]])
|
||||
assert r == TopologyClass.VRAM_MISMATCHED
|
||||
PY
|
||||
|
||||
run_test "C16 topology advisory emits note for compute mismatch" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, fits, TopologyClass
|
||||
p = load_profiles()
|
||||
r = fits(
|
||||
hardware=[p.hardware["rtx-3090"], p.hardware["rtx-4090"]],
|
||||
model=p.models["qwen3.6-27b"],
|
||||
workload=p.workloads["long-ctx-single"],
|
||||
engine=p.engines["vllm-nightly-mtp"],
|
||||
drafter=p.drafters["qwen-mtp-builtin"],
|
||||
tp=2,
|
||||
pp=1,
|
||||
project_vram=False,
|
||||
)
|
||||
assert r.topology_class == TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED
|
||||
assert "C16" in r.diagnostics["constraints_passed"]
|
||||
assert any("C16" in n and "vram_matched_compute_mismatched" in n for n in r.notes), r.notes
|
||||
PY
|
||||
|
||||
run_test "C16 topology advisory is silent for homogeneous GPUs" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, fits, TopologyClass
|
||||
p = load_profiles()
|
||||
r = fits(
|
||||
hardware=[p.hardware["rtx-3090"], p.hardware["rtx-3090"]],
|
||||
model=p.models["qwen3.6-27b"],
|
||||
workload=p.workloads["long-ctx-single"],
|
||||
engine=p.engines["vllm-nightly-mtp"],
|
||||
drafter=p.drafters["qwen-mtp-builtin"],
|
||||
tp=2,
|
||||
pp=1,
|
||||
project_vram=False,
|
||||
)
|
||||
assert r.topology_class == TopologyClass.HOMOGENEOUS
|
||||
assert "C16" in r.diagnostics["constraints_passed"]
|
||||
assert not any("C16" in n for n in r.notes), r.notes
|
||||
PY
|
||||
|
||||
run_test "C1 card count: world size mismatch rejected" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, fits
|
||||
p = load_profiles()
|
||||
@@ -236,7 +307,7 @@ from scripts.lib.profiles.compat import load_profiles, fits
|
||||
p = load_profiles()
|
||||
r = fits([p.hardware["rtx-3090"]], p.models["qwen3.6-27b"], p.workloads["long-ctx-single"], p.engines["vllm-nightly-mtp"], tp=1, project_vram=False)
|
||||
d = r.diagnostics
|
||||
assert d["constraints_evaluated"] == [f"C{i}" for i in range(1, 16)]
|
||||
assert d["constraints_evaluated"] == [f"C{i}" for i in range(1, 17)]
|
||||
assert "constraints_passed" in d and "constraints_failed" in d and "constraints_skipped" in d
|
||||
assert isinstance(d["elapsed_ms"], float)
|
||||
PY
|
||||
|
||||
@@ -143,6 +143,7 @@ out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:
|
||||
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
|
||||
--no-preflight --no-verify --model qwen3.6-27b --gpus 0,1 --no-projection 2>&1)"
|
||||
assert_contains "$out" "[launch] Tensor parallel TP=2"
|
||||
assert_not_contains "$out" "Topology:"
|
||||
assert_contains "$out" "SWITCHED vllm/dual CUDA=0,1 NVD=0,1 TP=2 PP=1"
|
||||
selected_count="$(grep -c "\[launch\] selected variant:" <<< "$out" || true)"
|
||||
if [[ "$selected_count" != "1" ]]; then
|
||||
@@ -179,6 +180,24 @@ if out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6
|
||||
fi
|
||||
assert_contains "$out" "Gemma 4 31B does not fit on a single 24 GB card today"
|
||||
|
||||
out="$(CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:RTX_3090:24576:8.6' \
|
||||
bash "${ROOT_DIR}/scripts/launch.sh" --topology 2>&1)"
|
||||
assert_contains "$out" "Topology class: homogeneous"
|
||||
assert_not_contains "$out" "Compute mismatch detected"
|
||||
|
||||
out="$(CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:RTX_4090:24576:8.9' \
|
||||
bash "${ROOT_DIR}/scripts/launch.sh" --topology 2>&1)"
|
||||
assert_contains "$out" "Topology class: vram_matched_compute_mismatched"
|
||||
assert_contains "$out" "Compute mismatch detected"
|
||||
assert_contains "$out" "Estate planner"
|
||||
|
||||
out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:RTX_4090:24576:8.9' \
|
||||
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
|
||||
--no-preflight --no-verify --model qwen3.6-27b --gpus 0,1 --no-projection 2>&1)"
|
||||
assert_contains "$out" "Topology: vram_matched_compute_mismatched"
|
||||
assert_contains "$out" "Compute mismatch detected"
|
||||
assert_contains "$out" "SWITCHED vllm/dual CUDA=0,1 NVD=0,1 TP=2 PP=1"
|
||||
|
||||
if out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:RTX_3090:24576:8.6,2:RTX_3090:24576:8.6,3:RTX_3090:24576:8.6,4:RTX_3090:24576:8.6,5:RTX_3090:24576:8.6' \
|
||||
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
|
||||
--no-preflight --no-verify --model qwen3.6-27b --gpus 0,1,2,3,4,5 --tp 6 --no-projection 2>&1)"; then
|
||||
|
||||
@@ -50,12 +50,18 @@ for dir in "${fixtures[@]}"; do
|
||||
}
|
||||
assert_contains "$row" "Report: \`results/rebench/${dir##*/}/REPORT.md\`"
|
||||
if [[ "$section" == "Gemma 4 31B (community-experimental)" ]]; then
|
||||
assert_columns "$row" 10
|
||||
assert_columns "$row" 11
|
||||
else
|
||||
assert_columns "$row" 8
|
||||
assert_columns "$row" 9
|
||||
fi
|
||||
done
|
||||
|
||||
out="$(BENCH_MOCK=1 RUNS=1 WARMUPS=0 bash scripts/bench.sh)"
|
||||
assert_contains "$out" "PP tok/s"
|
||||
out="$(BENCH_MOCK=1 PP=1 RUNS=1 WARMUPS=0 bash scripts/bench.sh)"
|
||||
assert_contains "$out" "summary [prompt-processing]"
|
||||
assert_contains "$out" "PP tok/s"
|
||||
|
||||
tag="qwen-int8-pth-n4-2026-05-10"
|
||||
rm -f "results/rebench/${tag}/BENCHMARKS-row.md" \
|
||||
"results/rebench/${tag}/PR-body.md" \
|
||||
|
||||
Reference in New Issue
Block a user