Files
club-3090/scripts/lib/profiles/compose_registry.py
T
noonghunnaandClaude Opus 4.8 aff9890720 Wire gemma-4-12b into curated catalog (vLLM gemma4-unified, bf16 + MTP)
Registers the new gemma4_unified 12B model (vLLM PR #44429, merged 2026-06-03) as
catalog slugs vllm/gemma-12b (base) + vllm/gemma-12b-mtp (assistant drafter n=4),
so launch.sh/switch.sh resolve them by slug. Status: experimental.

- ModelProfile (family gemma4-unified; 48L/8 full+40 sliding, 16/8 GQA, head_dim
  256/512, SWA 1024, max_ctx 131072 — past that vLLM CUDA-OOBs, vllm#39914).
- Engine profile vllm-gemma4-unified pins the gemma4-unified PREVIEW image
  (NOT stable v0.22.0, which lacks the arch). Caveat: ephemeral tag — pin a digest
  before any Production promotion.
- Drafter gemma-12b-it-assistant (n=4, 0.85GB), bf16 weights, 2 registry entries,
  no DEFAULTS row (experimental). kv-calc made gemma-4-12b-aware via the shared
  Gemma dense path + measured calibration anchor (384,019 tok / 8.16 GiB @ MTP/TP2/131072).
- ADDING_MODELS.md: name the central registry (compose_registry.py SoT → both
  launchers) explicitly in the intro + path-3.

KNOWN: kv-calc under-predicts the live pool (~204K vs measured 384K tok, -47%) — the
shared Gemma dense formula over-prices gemma4_unified global-layer KV; --calibration
GB-verdict still PASSES (TIGHT). Recorded in the calibration YAML header; refine the
gemma4_unified global-KV model as a follow-up. Guard suite 41/41 green.

Co-Authored-By: Claude Opus 4.8 (1M context) <[email protected]>
2026-06-04 01:17:19 +00:00

900 lines
50 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Static compose-to-profile bridge for v0.7.0.
The registry intentionally mirrors the shipped compose files. It is not a
generator and it does not attempt to normalize away historical variants.
"""
# Slug lifecycle / availability statuses — the canonical health flag.
#
# These are the registry-side equivalent of the compose `Status:` header enum
# (see the repo CLAUDE.md "Status enum" table). The compose-header emoji maps
# to one of these words; the drift-guard test asserts the two never diverge.
#
# functional → launches normally (production) or with a one-line notice
# (caveats).
# (NA) → surfaced in --list but not reliable: launch warns and requires
# --force so a user can't *unknowingly* boot a broken slug.
STATUS_VALUES = (
"production", # ✅ Production — recommended, fully validated.
"caveats", # ⚠️ Production w/ caveats — works under documented limits.
"experimental", # 🧪 Experimental — under active validation; may not boot.
"preview", # 👁️ Preview — known quality issues; tracked, not for prod.
"upstream-gated", # ⏸️ Upstream-gated — blocked by external action (pin/PR/HW).
"deprecated", # 🗑️ Deprecated — kept for reference; flagged for removal.
)
# Statuses that launch without --force. Everything else is "(NA)".
FUNCTIONAL_STATUSES = frozenset({"production", "caveats"})
# Compose `Status:` header emoji → registry status word. The header may carry
# trailing prose after the canonical token (e.g. "✅ Production (NEW — ...)");
# matching is by the leading emoji, so prose is tolerated.
COMPOSE_STATUS_EMOJI = {
"✅": "production",
"⚠️": "caveats",
"🧪": "experimental",
"👁️": "preview",
"⏸️": "upstream-gated",
"🗑️": "deprecated",
}
def _entry(
*,
model,
weights_variant,
workload,
engine,
drafter,
kv_format,
tp,
max_ctx,
max_num_seqs,
mem_util,
compose_path,
default_port,
kvcalc_key=None,
requires_nvlink=False,
required_engine_features=None,
recommended_engine_features=None,
required_sm=None,
status="production",
status_note=None,
):
if status not in STATUS_VALUES:
raise ValueError(
f"{compose_path}: status={status!r} not in {STATUS_VALUES}"
)
entry = {
"model": model,
"weights_variant": weights_variant,
"workload": workload,
"engine": engine,
"drafter": drafter,
"kv_format": kv_format,
"tp": tp,
"pp": 1,
"max_ctx": max_ctx,
"max_num_seqs": max_num_seqs,
"mem_util": mem_util,
"compose_path": compose_path,
"requires_nvlink": requires_nvlink,
"required_engine_features": list(required_engine_features or []),
"default_port": default_port,
"gpu_assignment_mode": "contiguous",
"kvcalc_key": kvcalc_key,
"status": status,
"status_note": status_note,
}
if recommended_engine_features:
entry["recommended_engine_features"] = list(recommended_engine_features)
if required_sm is not None:
entry["required_sm"] = required_sm
return entry
def compose_header_status(text):
"""Map a compose file's profile-schema `Status:` header to a status word.
Reads ONLY the `Status:` line inside the leading `# Profile (at-a-glance):`
comment block (the structured schema), stopping at the `# ---` separator so
a free-form `# Status: ...` prose line further down can't be mistaken for it.
Returns the status word (one of STATUS_VALUES) or None if no canonical
emoji is found. Matching is by the leading enum emoji, so trailing prose
after the canonical token (e.g. "✅ Production (NEW — ...)") is tolerated.
"""
in_schema = False
for line in text.splitlines():
stripped = line.strip()
if stripped.startswith("# Profile (at-a-glance):"):
in_schema = True
continue
if not in_schema:
continue
# The schema block ends at the dashed separator line.
if stripped.startswith("# --") or stripped.startswith("#--"):
break
# Match "# Status: <emoji> ..." within the schema block.
body = stripped.lstrip("#").strip()
if body.startswith("Status:"):
value = body[len("Status:"):].strip()
for emoji, word in COMPOSE_STATUS_EMOJI.items():
if value.startswith(emoji):
return word
return None
return None
COMPOSE_REGISTRY = {
# Qwen 3.6 27B, vLLM single-card.
"vllm/default": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="tool-heavy",
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="turboquant_3bit_nc",
tp=1, max_ctx=48000, max_num_seqs=1, mem_util=0.92,
compose_path="models/qwen3.6-27b/vllm/compose/single/autoround-int4/tq3-mtp.yml",
default_port=8020, required_engine_features=["turboquant_3bit_nc"],
kvcalc_key="qwen3.6-27b:long-vision",
status="deprecated",
status_note="DEPRECATED 2026-05-31 — Genesis/nightly path on hold (Sandermage); stack moving to stable vLLM. The vLLM single-card default repointed to vllm/minimal (v0.22.0). Use vllm/minimal or beellama single-card.",
),
"vllm/long-text": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="turboquant_3bit_nc",
tp=1, max_ctx=180000, max_num_seqs=1, mem_util=0.93,
compose_path="models/qwen3.6-27b/vllm/compose/single/autoround-int4/long-text.yml",
default_port=8020, required_engine_features=["turboquant_3bit_nc"],
kvcalc_key="qwen3.6-27b:long-text",
status="deprecated",
status_note="DEPRECATED 2026-05-31 — Genesis/nightly path on hold (Sandermage) + Cliff 2b >50K; stack moving to stable vLLM. Use vllm/minimal (single) or beellama single-card.",
),
"vllm/long-text-no-mtp": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
engine="vllm-nightly-mtp", drafter=None, kv_format="turboquant_3bit_nc",
tp=1, max_ctx=200000, max_num_seqs=1, mem_util=0.95,
compose_path="models/qwen3.6-27b/vllm/compose/single/autoround-int4/long-text-no-mtp.yml",
default_port=8021, required_engine_features=["turboquant_3bit_nc"],
kvcalc_key="qwen3.6-27b:long-text-no-mtp",
status="deprecated",
status_note="DEPRECATED 2026-05-31 — Genesis/nightly path on hold (Sandermage) + Cliff 2b >50K; stack moving to stable vLLM. Use vllm/minimal (single) or beellama single-card.",
),
"vllm/long-vision": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="vision-coding",
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="turboquant_3bit_nc",
tp=1, max_ctx=145000, max_num_seqs=1, mem_util=0.95,
compose_path="models/qwen3.6-27b/vllm/compose/single/autoround-int4/long-vision.yml",
default_port=8020, required_engine_features=["turboquant_3bit_nc"],
kvcalc_key="qwen3.6-27b:long-vision",
status="deprecated",
status_note="DEPRECATED 2026-05-31 — Genesis/nightly path on hold (Sandermage) + Cliff 2b >50K; stack moving to stable vLLM. Use vllm/minimal (single) or beellama single-card.",
),
"vllm/bounded-thinking": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="tool-heavy",
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="turboquant_3bit_nc",
tp=1, max_ctx=180000, max_num_seqs=1, mem_util=0.95,
compose_path="models/qwen3.6-27b/vllm/compose/single/autoround-int4/bounded-thinking.yml",
default_port=8020, required_engine_features=["turboquant_3bit_nc"],
kvcalc_key="qwen3.6-27b:bounded-thinking",
status="deprecated",
status_note="DEPRECATED 2026-05-31 — Genesis/nightly path on hold (Sandermage); stack moving to stable vLLM. Use vllm/minimal (single) or beellama single-card.",
),
"vllm/tools-text": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="tool-heavy",
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
tp=1, max_ctx=75000, max_num_seqs=1, mem_util=0.97,
compose_path="models/qwen3.6-27b/vllm/compose/single/autoround-int4/tools-text.yml",
default_port=8020,
kvcalc_key="qwen3.6-27b:tools-text",
status="deprecated",
status_note="DEPRECATED 2026-05-31 — Genesis/nightly path on hold (Sandermage); stack moving to stable vLLM. Use vllm/minimal (single) or beellama single-card.",
),
"vllm/minimal": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="fast-chat",
engine="vllm-nightly-clean", drafter=None, kv_format="fp8_e5m2",
tp=1, max_ctx=32768, max_num_seqs=1, mem_util=0.92,
compose_path="models/qwen3.6-27b/vllm/compose/single/autoround-int4/minimal.yml",
default_port=8020,
kvcalc_key="qwen3.6-27b:minimal",
),
# Qwen 3.6 27B, vLLM dual/multi-card.
"vllm/dual": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/fp8-mtp.yml",
default_port=8010, recommended_engine_features=["marlin_pad_sub_tile_n"],
kvcalc_key="qwen3.6-27b:dual",
),
"vllm/dual-turbo": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="multi-stream-tenant",
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="turboquant_3bit_nc",
tp=2, max_ctx=262144, max_num_seqs=4, mem_util=0.85,
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/turbo.yml",
default_port=8011, required_engine_features=["turboquant_3bit_nc"],
kvcalc_key="qwen3.6-27b:dual-turbo",
recommended_engine_features=["marlin_pad_sub_tile_n"],
status="deprecated",
status_note="DEPRECATED 2026-05-31 — Genesis/nightly path on hold (Sandermage); stack moving to stable vLLM. Use vllm/dual (v0.22.0) for dual-card.",
),
"vllm/dual-dflash": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="vision-coding",
engine="vllm-nightly-dflash", drafter="zlab-qwen-dflash", kv_format="fp16",
tp=2, max_ctx=185000, max_num_seqs=1, mem_util=0.95,
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/dflash.yml",
default_port=8012, required_engine_features=["marlin_pad_sub_tile_n"],
kvcalc_key="qwen3.6-27b:dual-dflash",
status="deprecated",
status_note="Pruned 2026-05-31: superseded by vllm/dual (fp8, 262K, vision, MTP, stock v0.22.0). DFlash traded ctx/concurrency (185K, 1 stream) for code TPS; recover from git if demand returns.",
),
"vllm/dual-dflash-noviz": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
engine="vllm-nightly-dflash", drafter="zlab-qwen-dflash", kv_format="fp16",
tp=2, max_ctx=200000, max_num_seqs=1, mem_util=0.95,
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/dflash-noviz.yml",
default_port=8013, required_engine_features=["marlin_pad_sub_tile_n"],
kvcalc_key="qwen3.6-27b:dual-dflash-noviz",
status="deprecated",
status_note="Pruned 2026-05-31: the lone no-vision dual A/B; dropped per the single-vision-config policy (vllm/dual covers vision + 262K + 2 streams). Recover from git if a text-only path is needed.",
),
"vllm/dual-bf16": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="bf16",
tp=2, max_ctx=200000, max_num_seqs=1, mem_util=0.92,
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/bf16.yml",
default_port=8012,
kvcalc_key="qwen3.6-27b:dual-bf16",
status="deprecated",
status_note="Pruned 2026-05-31: was a matched-config A/B vs Gemma bf16.yml, never validated on Qwen3-Next DeltaNet. Superseded by vllm/dual (fp8, 262K).",
),
"vllm/dual-int8": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
engine="vllm-nightly-full", drafter="qwen-mtp-builtin", kv_format="int8_per_token_head",
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/int8.yml",
default_port=8011, required_engine_features=["int8_per_token_head"],
kvcalc_key="qwen3.6-27b:dual-int8",
status="deprecated",
status_note="Pruned 2026-05-31: was a matched-config A/B vs Gemma int8.yml; never validated on Qwen DeltaNet, and fp8 is native on Qwen so int8 PTH buys nothing. Superseded by vllm/dual.",
),
"vllm/dual-tq3-mtp": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="multi-stream-tenant",
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="turboquant_3bit_nc",
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/tq3-mtp.yml",
default_port=8013, required_engine_features=["turboquant_3bit_nc"],
kvcalc_key="qwen3.6-27b:dual-tq3-mtp",
status="deprecated",
status_note="Tombstoned 2026-05-11 — Genesis-free TQ3+MTP needs 4 of 5 upstream fixes not yet landed. Use dual-tq3-mtp-genesis (gated) or dual-turbo.",
),
"vllm/dual-tq3-mtp-genesis": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="multi-stream-tenant",
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="turboquant_3bit_nc",
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.85,
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/tq3-mtp-genesis.yml",
default_port=8015, required_engine_features=["turboquant_3bit_nc"],
kvcalc_key="qwen3.6-27b:dual-tq3-mtp-genesis",
status="deprecated",
status_note="DEPRECATED 2026-05-31 — Genesis on hold pending Sandermage; stack moving to stable vLLM. Use vllm/dual (v0.22.0). (Was upstream-gated on the parked/drifted Genesis pin — see docs/UPSTREAM.md.)",
),
"vllm/dual-tq3-nomtp": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
engine="vllm-nightly-mtp", drafter=None, kv_format="turboquant_3bit_nc",
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/tq3-nomtp.yml",
default_port=8014, required_engine_features=["turboquant_3bit_nc"],
kvcalc_key="qwen3.6-27b:dual-tq3-nomtp",
status="deprecated",
status_note="Pruned 2026-05-31: superseded by vllm/dual (fp8, same 262K/2-stream, faster + MTP). TQ3 KV density not worth the decode cost here; recover from git if needed.",
),
"vllm/dual-carnice-bf16mtp": _entry(
model="qwen3.6-27b", weights_variant="carnice-bf16mtp", workload="long-ctx-single",
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
compose_path="models/qwen3.6-27b/vllm/compose/dual/carnice-bf16mtp/bf16-mtp.yml",
default_port=8070,
kvcalc_key="SKIP",
status="caveats",
status_note="Carnice fine-tune MTP AL=2.0 vs Lorbus 3.4-3.8 (working but suboptimal TPS).",
),
"vllm/dual-qwopus-bf16mtp": _entry(
model="qwen3.6-27b", weights_variant="qwopus-bf16mtp", workload="long-ctx-single",
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
compose_path="models/qwen3.6-27b/vllm/compose/dual/qwopus-bf16mtp/bf16-mtp.yml",
default_port=8071,
kvcalc_key="SKIP",
status="preview",
status_note="Qwopus fine-tune preview: line repetition + NIAH drop + silent-empty turn-5 in soak.",
),
"vllm/dual4": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="multi-stream-tenant",
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
tp=4, max_ctx=262144, max_num_seqs=4, mem_util=0.92,
compose_path="models/qwen3.6-27b/vllm/compose/multi4/autoround-int4/fp8-mtp.yml",
default_port=8015,
kvcalc_key="qwen3.6-27b:dual4",
),
"vllm/dual4-dflash": _entry(
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
engine="vllm-nightly-dflash", drafter="zlab-qwen-dflash", kv_format="fp16",
tp=4, max_ctx=262144, max_num_seqs=2, mem_util=0.95,
compose_path="models/qwen3.6-27b/vllm/compose/multi4/autoround-int4/dflash.yml",
default_port=8016, required_engine_features=["marlin_pad_sub_tile_n"],
kvcalc_key="qwen3.6-27b:dual4-dflash",
),
# Qwen 3.6 27B, llama.cpp single-card.
# `llamacpp/default` is an alias for `llamacpp/mtp` (collapsed 2026-05-22):
# the old Q3_K_XL vanilla compose was retired and `default` now points at
# the MTP compose. max_ctx = the 200K max-safe default (262K boots but walls
# ~125K at fill — see docs/CLIFFS.md; runtime CTX_SIZE default is 200000).
"llamacpp/default": _entry(
model="qwen3.6-27b", weights_variant="unsloth-q4km", workload="fast-chat",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
tp=1, max_ctx=200000, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/llama-cpp/compose/single/unsloth-q4km/mtp.yml",
default_port=8020,
kvcalc_key="SKIP",
),
"llamacpp/mtp": _entry(
model="qwen3.6-27b", weights_variant="unsloth-q4km", workload="fast-chat",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
tp=1, max_ctx=200000, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/llama-cpp/compose/single/unsloth-q4km/mtp.yml",
default_port=8020,
kvcalc_key="SKIP",
),
"llamacpp/bounded-thinking": _entry(
model="qwen3.6-27b", weights_variant="unsloth-q4km", workload="tool-heavy",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
tp=1, max_ctx=200000, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/llama-cpp/compose/single/unsloth-q4km/bounded-thinking.yml",
default_port=8020,
kvcalc_key="SKIP",
status="experimental",
status_note="New structured-CoT port; live grammar + MTP validation pending.",
),
"llamacpp/mtp-vision": _entry(
model="qwen3.6-27b", weights_variant="unsloth-q4km", workload="vision-coding",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
# 150K @ 1M-px (IMAGE_MAX_TOKENS=1024) — re-tuned 2026-05-25 (PR #227); was a
# stale 49152. Full-res 4M-px OOMs at fill, so 1M-px is the safe default.
tp=1, max_ctx=150000, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/llama-cpp/compose/single/unsloth-q4km/mtp-vision.yml",
default_port=8020,
kvcalc_key="SKIP",
),
# ik_llama.cpp — IQ4_KS (ubergarm). Same engine family as llamacpp, but the
# IQK quant is ~0.5-0.8 GB leaner on weights → best fit for VRAM-tight
# single-card (sub-24 GB, shared GPU, WSL display overhead). Its own image
# (ikawrakow/ik-llama-cpp), so unaffected by mainline llama.cpp drift.
"ik-llama/iq4ks-mtp": _entry(
model="qwen3.6-27b", weights_variant="ubergarm-iq4ks", workload="fast-chat",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
tp=1, max_ctx=200000, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/ik-llama/compose/single/ubergarm-iq4ks/mtp.yml",
default_port=8020,
kvcalc_key="SKIP",
),
"ik-llama/iq4ks-mtp-vision": _entry(
model="qwen3.6-27b", weights_variant="ubergarm-iq4ks", workload="vision-coding",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
tp=1, max_ctx=163840, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/ik-llama/compose/single/ubergarm-iq4ks/mtp-vision.yml",
default_port=8020,
kvcalc_key="SKIP",
),
"ik-llama/iq4ks-two-stage": _entry(
model="qwen3.6-27b", weights_variant="ubergarm-iq4ks", workload="fast-chat",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
tp=1, max_ctx=200000, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/ik-llama/compose/single/ubergarm-iq4ks/two-stage.yml",
default_port=8020,
kvcalc_key="SKIP",
),
# Qwen3.6-27B beellama.cpp DFlash — single-card DEFAULT (DFlash spec-dec,
# Q5_K_S target + Anbeeld DFlash-IQ4_XS draft, q5_0(K)/q4_1(V) KV). beellama
# is a llama.cpp-family engine (kvcalc SKIP, like ik-llama). Promoted to the
# single-GPU default 2026-05-30: code-throughput leader (~100 TPS) + slight
# 8-pack quality edge (107 vs ik 99, think-off) + output-lossless spec-dec.
# Served via our UNOFFICIAL multi-arch image (sm_86/89/120 = 3090/4090/5090);
# sm_89/sm_120 are compiled but unvalidated on our 3090-only rig — see Caveats
# in the compose. kv_format reflects K-side precision (V is q4_1).
"beellama/dflash": _entry(
model="qwen3.6-27b", weights_variant="beellama-q5ks-dflash", workload="fast-chat",
engine="beellama-local", drafter="anbeeld-qwen-dflash", kv_format="q5_0",
tp=1, max_ctx=102400, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/beellama/compose/single/beellama-q5ks-dflash/dflash.yml",
default_port=8060,
kvcalc_key="SKIP",
status="caveats",
status_note="Single-GPU default. Launchers inject Anbeeld's official beellama.cpp server-cuda-v0.3.0 image (sm_86/89 = 3090/4090); sm_89 compiled-not-validated on club-3090's 3090-only rig. 5090/sm_120: prefix BEELLAMA_IMAGE=ghcr.io/noonghunna/beellama-cpp:multiarch-v0.3.0-efe856397 (sm_120 compiled-not-validated). Usable ctx ceiling 160K (200K OOMs on prefill); ships 102K. DFlash prose is net-positive on tok/s (+27% vs no-spec, re-tested 2026-06-03); the earlier 'prose-DFlash regression' is RETRACTED — it was an AR over-read + wrong baseline (docs/UPSTREAM.md).",
),
# Qwen3.6-27B PRISM-PRO-DQ (Ex0bit dynamic-quant GGUF) — community-experimental, ik-llama.
"ik-llama/prism-pro-dq-mtp": _entry(
model="qwen3.6-27b", weights_variant="ex0bit-prism-pro-dq", workload="fast-chat",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
tp=1, max_ctx=122880, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/ik-llama/compose/single/ex0bit-prism-pro-dq/mtp.yml",
default_port=8020,
kvcalc_key="SKIP",
status="experimental",
status_note="PRISM-PRO-DQ community dynamic-quant GGUF — eval-only, not yet validated.",
),
"ik-llama/prism-pro-dq-long": _entry(
model="qwen3.6-27b", weights_variant="ex0bit-prism-pro-dq", workload="long-ctx-single",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
tp=1, max_ctx=180000, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/ik-llama/compose/single/ex0bit-prism-pro-dq/long.yml",
default_port=8052,
kvcalc_key="SKIP",
status="experimental",
status_note="PRISM-PRO-DQ community dynamic-quant GGUF — eval-only, not yet validated.",
),
"ik-llama/prism-pro-dq-two-stage": _entry(
model="qwen3.6-27b", weights_variant="ex0bit-prism-pro-dq", workload="tool-heavy",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
tp=1, max_ctx=200000, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/ik-llama/compose/single/ex0bit-prism-pro-dq/two-stage.yml",
default_port=8020,
kvcalc_key="SKIP",
status="experimental",
status_note="PRISM-PRO-DQ community dynamic-quant GGUF — eval-only, not yet validated.",
),
"ik-llama/prism-pro-dq-dual": _entry(
model="qwen3.6-27b", weights_variant="ex0bit-prism-pro-dq", workload="tool-heavy",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
tp=2, max_ctx=196608, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/ik-llama/compose/dual/ex0bit-prism-pro-dq/mtp.yml",
default_port=8053,
kvcalc_key="SKIP",
status="experimental",
status_note="PRISM-PRO-DQ community dynamic-quant GGUF — eval-only, not yet validated.",
),
"ik-llama/prism-pro-dq-dual-vision": _entry(
model="qwen3.6-27b", weights_variant="ex0bit-prism-pro-dq", workload="vision-coding",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q8_0",
tp=2, max_ctx=262144, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/ik-llama/compose/dual/ex0bit-prism-pro-dq/mtp-vision.yml",
default_port=8010,
kvcalc_key="SKIP",
status="experimental",
status_note="PRISM-PRO-DQ community dynamic-quant GGUF — eval-only, not yet validated.",
),
# Qwen3.6-35B-A3B APEX-MTP (mudler MoE GGUF — Compact + Quality) — community-experimental, ik-llama.
"ik-llama/apex-mtp-compact": _entry(
model="qwen3.6-35b-a3b", weights_variant="mudler-apex-compact", workload="fast-chat",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
tp=1, max_ctx=163840, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-35b-a3b/ik-llama/compose/single/mudler-apex-compact/mtp.yml",
default_port=8054,
kvcalc_key="SKIP",
status="experimental",
status_note="APEX-MTP community MoE GGUF — eval-only bring-up lane, not yet validated.",
),
"ik-llama/byteshape-iq4xs-mtp": _entry(
model="qwen3.6-35b-a3b", weights_variant="byteshape-iq4xs", workload="fast-chat",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
tp=1, max_ctx=262144, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-35b-a3b/ik-llama/compose/single/byteshape-iq4xs/mtp.yml",
default_port=8058,
kvcalc_key="SKIP",
status="caveats",
status_note="byteshape IQ4_XS 4.19bpw MoE GGUF (embedded MTP head) — community intake from PR #293 (@Rhonstin). Single-card 35B-A3B, q4_0 KV + --fit → 262K. First-party validated 2026-06-02 on 1× 3090: verify-full all-pass, verify-stress 8/8 (NIAH→240K, no Cliff), bench n=5 (narrative 113/116 · code 129/137 wall/decode TPS, CV<2.3%), 8-pack 110/150 (≈ author's 111/150), soak-continuous PASS (0 err, 0 VRAM growth, 0/25 silent-empty). Caveat: single-rig; agent packs modest (hermes 55%, cli 42%) as typical for the class. Intake fixes vs #293: image cu13, port 8058.",
),
"ik-llama/apex-mtp-compact-long": _entry(
model="qwen3.6-35b-a3b", weights_variant="mudler-apex-compact", workload="long-ctx-single",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q8_0",
tp=1, max_ctx=196608, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-35b-a3b/ik-llama/compose/single/mudler-apex-compact/long.yml",
default_port=8056,
kvcalc_key="SKIP",
status="experimental",
status_note="APEX-MTP community MoE GGUF — eval-only bring-up lane, not yet validated.",
),
# @laurimyllari's --fit + asymmetric q8_0(K)/q5_0(V) KV config from
# discussion #241, retuned + measured on 1× 3090. +7% narr / +4% code
# over the q4/q4 mtp.yml sibling on the same APEX I-Compact GGUF.
# kv_format "q8_0" reflects K-side precision; V is q5_0 (see compose).
"ik-llama/apex-fit-q8q5": _entry(
model="qwen3.6-35b-a3b", weights_variant="mudler-apex-compact", workload="fast-chat",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q8_0",
tp=1, max_ctx=196608, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-35b-a3b/ik-llama/compose/single/mudler-apex-compact/fit-mtp.yml",
default_port=8057,
kvcalc_key="SKIP",
),
"ik-llama/apex-mtp-quality-dual": _entry(
model="qwen3.6-35b-a3b", weights_variant="mudler-apex-quality", workload="long-ctx-single",
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q8_0",
tp=2, max_ctx=196608, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-35b-a3b/ik-llama/compose/dual/mudler-apex-quality/mtp.yml",
default_port=8055,
kvcalc_key="SKIP",
status="experimental",
status_note="APEX-MTP community MoE GGUF — eval-only bring-up lane, not yet validated.",
),
# Gemma 4 31B, vLLM. Lean v0.21.0 set: bf16 default, int8 long-context, single-card fp8 risk path.
"vllm/gemma-mtp-tp1": _entry(
model="gemma-4-31b", weights_variant="autoround-int4", workload="fast-chat",
engine="vllm-gemma-stable", drafter="gemma-it-assistant", kv_format="fp8_e4m3",
tp=1, max_ctx=8192, max_num_seqs=256, mem_util=0.95,
compose_path="models/gemma-4-31b/vllm/compose/single/autoround-int4/fp8-mtp.yml",
default_port=8031, required_sm=9.0,
kvcalc_key="gemma-4-31b:gemma-single",
status="deprecated",
status_note="Dead on Ampere: no fp8 KV path for Gemma 4 on sm_86 (attention asserts kv ∈ {fp8,fp8_e4m3,nvfp4} — rejects fp8_e5m2; fp8/fp8_e4m3 need the fp8e4nv kernel sm_86 lacks; nvfp4 Blackwell-only). Live-confirmed stock v0.22.0 2026-05-31. Single-card → beellama/gemma-dflash; dual → vllm/gemma-bf16-mtp. See compose Caveats.",
),
"vllm/gemma-bf16-mtp": _entry(
model="gemma-4-31b", weights_variant="autoround-int4", workload="fast-chat",
engine="vllm-gemma-stable", drafter="gemma-it-assistant", kv_format="bf16",
tp=2, max_ctx=131072, max_num_seqs=4, mem_util=0.95,
compose_path="models/gemma-4-31b/vllm/compose/dual/autoround-int4/bf16-mtp.yml",
default_port=8030,
kvcalc_key="gemma-4-31b:gemma-dual",
),
"vllm/gemma-int8-mtp": _entry(
model="gemma-4-31b", weights_variant="autoround-int4", workload="multi-stream-tenant",
engine="vllm-gemma-stable", drafter="gemma-it-assistant", kv_format="int8_per_token_head",
tp=2, max_ctx=262144, max_num_seqs=4, mem_util=0.95,
compose_path="models/gemma-4-31b/vllm/compose/dual/autoround-int4/int8.yml",
default_port=8032, required_engine_features=["int8_per_token_head"],
kvcalc_key="gemma-4-31b:gemma-dual-int8",
),
# Gemma-4-12B (gemma4_unified arch — vLLM PR #44429, merged 2026-06-03),
# dual-3090 bf16 on the ephemeral vllm/vllm-openai:gemma4-unified preview
# image (== today's vLLM main; Gemma-4 fixes baked in, nothing vendored).
# bf16 weights (~24 GB) don't fit one 24 GB card with KV → TP=2 mandatory.
# Both EXPERIMENTAL: arch-preview image, sm_86 gemma4_unified support
# unverified, ephemeral tag (pin a digest before promotion). Max ctx 131072
# is the HARD ceiling (card claims 256K but vLLM OOB-crashes past the
# trained max, vllm#39914). kvcalc routes through the shared Gemma dense
# path (gemma4_unified TEXT backbone == gemma4-swa-dense KV family).
"vllm/gemma-12b": _entry(
model="gemma-4-12b", weights_variant="bf16", workload="fast-chat",
engine="vllm-gemma4-unified", drafter=None, kv_format="bf16",
tp=2, max_ctx=131072, max_num_seqs=4, mem_util=0.90,
compose_path="models/gemma-4-12b/vllm/compose/dual/bf16/base.yml",
default_port=8035,
kvcalc_key="gemma-4-12b:gemma-dual",
status="experimental",
status_note="Gemma-4-12B (gemma4_unified, vLLM PR #44429) dual-3090 bf16, no drafter. Ephemeral gemma4-unified arch-preview image; sm_86 support UNVERIFIED. Max ctx 131072 = hard ceiling (256K card-claim OOB-crashes vLLM past the trained max, vllm#39914). Pin a digest before any Production promotion.",
),
"vllm/gemma-12b-mtp": _entry(
model="gemma-4-12b", weights_variant="bf16", workload="fast-chat",
engine="vllm-gemma4-unified", drafter="gemma-12b-it-assistant", kv_format="bf16",
tp=2, max_ctx=131072, max_num_seqs=4, mem_util=0.90,
compose_path="models/gemma-4-12b/vllm/compose/dual/bf16/mtp.yml",
default_port=8036,
kvcalc_key="gemma-4-12b:gemma-dual",
status="experimental",
status_note="Gemma-4-12B (gemma4_unified, vLLM PR #44429) dual-3090 bf16 + assistant spec-dec (n=4). Ephemeral gemma4-unified arch-preview image; sm_86 support UNVERIFIED. Max ctx 131072 = hard ceiling (256K card-claim OOB-crashes vLLM past the trained max, vllm#39914). Pin a digest before any Production promotion.",
),
# Gemma-4-31B beellama.cpp DFlash — single-card DEFAULT (Q4_K_S target +
# Anbeeld DFlash-IQ4_XS draft, q5_0(K)/q4_1(V) KV). The ONLY viable fast
# single-card Gemma-4 path: does SWA windowed KV (big ctx) AND Gemma-4
# spec-dec, where vLLM is FA-walled (head_dim=512), ik-llama walls ~24K, and
# stock llama.cpp is ~12 TPS (no FA_ALL_QUANTS). Promoted to single-GPU
# default 2026-05-30 (no functional default existed before). llama.cpp-family
# → kvcalc SKIP. Re-point to the no-fork mainline path when llama.cpp#23398
# (Gemma-4 MTP) merges — see docs/UPSTREAM.md.
"beellama/gemma-dflash": _entry(
model="gemma-4-31b", weights_variant="beellama-q4ks-dflash", workload="fast-chat",
engine="beellama-local", drafter="anbeeld-gemma-dflash", kv_format="q5_0",
tp=1, max_ctx=128000, max_num_seqs=1, mem_util=None,
compose_path="models/gemma-4-31b/beellama/compose/single/beellama-q4ks-dflash/dflash.yml",
default_port=8061,
kvcalc_key="SKIP",
status="caveats",
status_note="Single-GPU default — the only viable fast single-card Gemma-4 path on Ampere. Launchers inject Anbeeld's official beellama.cpp server-cuda-v0.3.0 image (sm_86/89 = 3090/4090); sm_89 compiled-not-validated on club-3090's 3090-only rig. 5090/sm_120: prefix BEELLAMA_IMAGE=ghcr.io/noonghunna/beellama-cpp:multiarch-v0.3.0-efe856397 (sm_120 compiled-not-validated). DFlash prose is net-positive on tok/s (+28–31% vs no-spec, re-tested 2026-06-03; earlier 'prose regression' RETRACTED — AR over-read + wrong baseline); re-point to mainline llama.cpp#23398 Gemma-4 MTP when it merges — docs/UPSTREAM.md.",
),
# Dual-card beellama Gemma-4 (layer-split, 262K) — PARKED upstream-gated 2026-05-31.
# Boots + recalls 262K fine, but DFlash spec-dec is broken on multi-GPU in our pinned
# build (07ac3ce): drafter decode fails, accept 0.357, ~24/38 TPS; --device-draft crashes.
# Fixes live on Anbeeld's v0.3.0 dev branch (414 commits ahead) but no tagged release yet.
# Re-test (DFlash-fix AND --spec-type mtp) when a beellama release lands. docs/UPSTREAM.md.
"beellama/gemma-dflash-dual": _entry(
model="gemma-4-31b", weights_variant="beellama-q4ks-dflash", workload="fast-chat",
engine="beellama-local", drafter="anbeeld-gemma-dflash", kv_format="q5_0",
tp=2, max_ctx=262144, max_num_seqs=1, mem_util=None,
compose_path="models/gemma-4-31b/beellama/compose/dual/beellama-q4ks-dflash/dflash.yml",
default_port=8062,
kvcalc_key="SKIP",
status="experimental",
status_note="Dual-card beellama Gemma-4 (layer-split, 262K) on v0.3.0 — RELEASED experimental for community v0.3.0 testing (Anbeeld's request, club-3090#288). Multi-GPU DFlash FIXED on v0.3.0 (GPU cross-ring; validated sm_86 2026-06-01, FA_ALL_QUANTS=1; image injected from beellama-local install.spec = Anbeeld's official server-cuda-v0.3.0 commit tag). Earlier 'v0.3.0-wide DFlash-on-PROSE regression' RETRACTED (2026-06-03) — did NOT reproduce on qwen single+dual or gemma single (DFlash prose net-positive +27–58% vs no-spec); was an AR over-read + wrong baseline. This gemma-dual not separately re-benched. Promote experimental→caveats when Anbeeld tags a STABLE release. docs/UPSTREAM.md.",
),
# ------------------------------------------------------------------
# beellama v0.3.0 Q8_K_XL dual-card composes — RELEASED experimental for
# community v0.3.0 testing (Anbeeld's request, club-3090#288). Image
# injected centrally from engines/beellama-local.yml install.spec
# (Anbeeld's official server-cuda-v0.3.0 commit tag). Validated 2× 3090
# sm_86 2026-06-01. kvcalc_key=SKIP (llama.cpp family — no vLLM kv-calc).
# ------------------------------------------------------------------
"beellama/qwen-mtp-dual": _entry(
model="qwen3.6-27b", weights_variant="beellama-q8kxl-mtp", workload="fast-chat",
engine="beellama-local", drafter="unsloth-mtp-gguf", kv_format="q5_0",
tp=2, max_ctx=65536, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/beellama/compose/dual/beellama-q8kxl-mtp/mtp.yml",
default_port=8064,
kvcalc_key="SKIP",
status="experimental",
status_note="Dual-card beellama Qwen3.6-27B Q8_K_XL + embedded MTP head (--spec-type draft-mtp, unsloth-mtp-gguf drafter). v0.3.0 sm_86 2026-06-01: boots + coherent, MTP active (code accept ~0.90, ~58 TPS decode). Ships 65536 safe first-boot ctx; validated robust to ~160K (262K impossible — DeltaNet recurrent draft state hard-pins to one card). High-fidelity Q8 sibling of vllm/dual fp8-mtp. Promote experimental→caveats on a STABLE Anbeeld tag.",
),
"beellama/qwen-dflash-dual": _entry(
model="qwen3.6-27b", weights_variant="beellama-q8kxl-dflash", workload="fast-chat",
engine="beellama-local", drafter="anbeeld-qwen-dflash", kv_format="q5_0",
tp=2, max_ctx=262144, max_num_seqs=1, mem_util=None,
compose_path="models/qwen3.6-27b/beellama/compose/dual/beellama-q8kxl-dflash/dflash.yml",
default_port=8065,
kvcalc_key="SKIP",
status="experimental",
status_note="Dual-card beellama Qwen3.6-27B Q8_K_XL + DFlash (Anbeeld DFlash-IQ4_XS draft, --spec-type dflash). v0.3.0 sm_86 2026-06-01: boots + coherent at full 262K (fixed draft footprint; tensor-split 0.575,0.425 → ~21.2 GB/card). DFlash prose net-positive on tok/s (+52% vs no-spec @262K, re-tested 2026-06-03; earlier 'prose regression' RETRACTED — AR over-read + wrong baseline). Tool-grammar-neutral spec-dec for Qwen agents (club-3090#237). Promote on a STABLE tag.",
),
"beellama/gemma-q8-dflash-dual": _entry(
model="gemma-4-31b", weights_variant="beellama-q8kxl-dflash", workload="fast-chat",
engine="beellama-local", drafter="anbeeld-gemma-dflash", kv_format="q5_0",
tp=2, max_ctx=196608, max_num_seqs=1, mem_util=None,
compose_path="models/gemma-4-31b/beellama/compose/dual/beellama-q8kxl-dflash/dflash.yml",
default_port=8066,
kvcalc_key="SKIP",
status="experimental",
status_note="Dual-card beellama Gemma-4-31B Q8_K_XL + DFlash (Anbeeld DFlash-IQ4_XS draft). v0.3.0 sm_86 2026-06-01: 192K balanced ceiling (tensor-split 0.55,0.45 → ~21.4/21.9 GB; 262K OOMs — Gemma full-attn layers grow KV). High-fidelity Q8 sibling of beellama/gemma-dflash-dual (q4ks). Earlier 'v0.3.0-wide DFlash-on-PROSE regression' RETRACTED (2026-06-03) — didn't reproduce on qwen single+dual or gemma single; was an AR over-read + wrong baseline. This gemma-Q8-dual not separately re-benched. Promote on a STABLE tag.",
),
# v0.7.3 MoE onboarding — Gemma 4 26B-A4B + Qwen 3.6 35B-A3B.
# Both target the unconstrained-nightly engine (vllm-nightly-clean) which
# rides nightly-bf610c2f (2026-05-15, post-PR-#42521). Gemma is the
# shippable path; Qwen 35B-A3B is preview-only until Genesis v7.73.x
# re-anchors on a post-#42521 nightly.
"vllm/gemma-a4b-single": _entry(
model="gemma-4-26b-a4b", weights_variant="autoround-int4-mixed", workload="fast-chat",
engine="vllm-nightly-clean", drafter=None, kv_format="bf16",
tp=1, max_ctx=8192, max_num_seqs=256, mem_util=0.92,
compose_path="models/gemma-4-26b-a4b/vllm/compose/single/autoround-int4-mixed/bf16.yml",
default_port=8040,
kvcalc_key="gemma-4-26b-a4b:gemma-a4b-single",
status="experimental",
status_note="v0.7.3 MoE onboarding — first-boot smoke, validation pending.",
),
"vllm/gemma-a4b": _entry(
model="gemma-4-26b-a4b", weights_variant="autoround-int4-mixed", workload="fast-chat",
engine="vllm-nightly-clean", drafter=None, kv_format="bf16",
tp=2, max_ctx=32768, max_num_seqs=256, mem_util=0.92,
compose_path="models/gemma-4-26b-a4b/vllm/compose/dual/autoround-int4-mixed/bf16.yml",
default_port=8041,
kvcalc_key="gemma-4-26b-a4b:gemma-a4b",
status="experimental",
status_note="v0.7.3 MoE onboarding — primary bench target, validation pending.",
),
"vllm/gemma-a4b-awq": _entry(
model="gemma-4-26b-a4b", weights_variant="awq", workload="fast-chat",
engine="vllm-nightly-clean", drafter=None, kv_format="bf16",
tp=2, max_ctx=32768, max_num_seqs=256, mem_util=0.92,
compose_path="models/gemma-4-26b-a4b/vllm/compose/dual/awq/bf16.yml",
default_port=8042,
kvcalc_key="gemma-4-26b-a4b:gemma-a4b-awq",
status="experimental",
status_note="v0.7.3 MoE onboarding — AWQ path with PR #40886 overlay, validation pending.",
),
"vllm/gemma-a4b-awq-mtp": _entry(
model="gemma-4-26b-a4b", weights_variant="awq", workload="fast-chat",
engine="vllm-nightly-clean", drafter="gemma-26b-it-assistant", kv_format="bf16",
tp=2, max_ctx=32768, max_num_seqs=256, mem_util=0.92,
compose_path="models/gemma-4-26b-a4b/vllm/compose/dual/awq/mtp.yml",
default_port=8043,
kvcalc_key="gemma-4-26b-a4b:gemma-a4b-awq-mtp",
status="experimental",
status_note="v0.7.3 MoE onboarding — AWQ + MTP, validation pending.",
),
"vllm/qwen-a3b-preview-single": _entry(
model="qwen3.6-35b-a3b", weights_variant="autoround-int4", workload="fast-chat",
engine="vllm-nightly-clean", drafter=None, kv_format="fp8_e5m2",
tp=1, max_ctx=8192, max_num_seqs=1, mem_util=0.92,
compose_path="models/qwen3.6-35b-a3b/vllm/compose/single/autoround-int4/preview.yml",
default_port=8050,
kvcalc_key="qwen3.6-35b-a3b:qwen-a3b-preview-single",
status="preview",
status_note="MoE onboarding smoke — Cliff 2 mitigations unavailable without Genesis. Do NOT use for long-ctx.",
),
"vllm/qwen-35b-a3b-dual": _entry(
model="qwen3.6-35b-a3b", weights_variant="autoround-int4", workload="fast-chat",
engine="vllm-nightly-clean", drafter=None, kv_format="fp8_e5m2",
tp=2, max_ctx=262144, max_num_seqs=1, mem_util=0.92,
compose_path="models/qwen3.6-35b-a3b/vllm/compose/dual/autoround-int4/fp8.yml",
default_port=8051,
kvcalc_key="qwen3.6-35b-a3b:qwen-35b-a3b-dual",
),
}
DEFAULTS = {
("qwen3.6-27b", "vllm", "single"): "vllm/minimal",
("qwen3.6-27b", "vllm", "dual"): "vllm/dual",
("qwen3.6-27b", "vllm", "multi4"): "vllm/dual4",
("qwen3.6-27b", "llamacpp", "single"): "llamacpp/default",
("qwen3.6-27b", "ik-llama", "single"): "ik-llama/iq4ks-mtp",
("qwen3.6-27b", "beellama", "single"): "beellama/dflash",
# No vLLM single-card Gemma default: fp8 KV is hardware-impossible on Ampere
# sm_86 (vllm/gemma-mtp-tp1 deprecated 2026-05-31) and no bf16 single compose
# ships. Single-card Gemma → beellama/gemma-dflash (the curated walk picks it).
("gemma-4-31b", "beellama", "single"): "beellama/gemma-dflash",
# Dual default is gemma-int8-mtp: full 262K + vision + 4 streams (the full-context
# priority). It rides v0.21.0 + the vendored #40391 per-head-KV overlay (the one
# gemma config that can't follow stable). gemma-bf16-mtp stays as the stable v0.22.0
# no-overlay 32K fallback — kept, not deprecated, just no longer the default.
("gemma-4-31b", "vllm", "dual"): "vllm/gemma-int8-mtp",
("gemma-4-26b-a4b", "vllm", "single"): "vllm/gemma-a4b-single",
("gemma-4-26b-a4b", "vllm", "dual"): "vllm/gemma-a4b",
("qwen3.6-35b-a3b", "vllm", "single"): "vllm/qwen-a3b-preview-single",
("qwen3.6-35b-a3b", "vllm", "dual"): "vllm/qwen-35b-a3b-dual",
}
# --- PR-B: model-default resolver knobs (maintainer-owned, design §13.3) ----
#
# Two curated tables drive the `<model>/default` resolver. They are maintainer
# knobs — edited by PR, never auto-grown. See docs/model-default-resolver
# design + the repo CLAUDE.md "Default rule" note.
# A SHORT opt-in shortlist of models eligible to be the *bare-launch* default
# (`launch.sh` with no model + no pin → first INSTALLED model on this list →
# its `<model>/default`). This is NOT an exhaustive ranking of the catalog:
# - Models absent from it are fully runnable by name (`--model X` /
# `X/default`); they are simply never the auto-default.
# - New models are NOT auto-added. Adding a model touches nothing here;
# promote one explicitly only when desired.
# - Order within the (short) list = the tiebreak for "first installed".
RECOMMENDED_DEFAULT_MODELS = ["qwen3.6-27b", "gemma-4-31b"]
# Which engine wins, per detected topology, when resolving `<model>/default`
# with no user pin. The resolver walks this list in order and picks the FIRST
# engine that has a functional DEFAULTS[(model, engine, topology)] entry (i.e.
# whose status is NOT in the (NA) set). This is the whole recommendation
# policy expressed as data — reorder a row to change a recommendation, no code
# change, any topology.
#
# Engine identifiers are the slug-prefix form used in DEFAULTS keys:
# vllm · ik-llama · llamacpp (+ beellama, aspirational — see below).
# `beellama` is ranked but has ZERO registry entries today (blocked on an
# upstream Docker image — see docs/UPSTREAM.md). The resolver skips it
# naturally (no DEFAULTS hit) → no behavior change until it is onboarded, at
# which point it AUTO-PROMOTES to the single-GPU default with no resolver edit.
ENGINE_PREFERENCE = {
"single": ["beellama", "ik-llama", "llamacpp", "vllm"],
"dual": ["vllm", "ik-llama", "llamacpp", "beellama"],
"multi": ["vllm", "ik-llama", "llamacpp", "beellama"],
}
def _topology_family(topology):
"""Map a concrete topology to its ENGINE_PREFERENCE family.
Concrete topologies are `single` · `dual` · `multi4` · `multiN`; the
preference table keys on the family `single` · `dual` · `multi`.
"""
if topology == "single":
return "single"
if topology == "dual":
return "dual"
if topology.startswith("multi"):
return "multi"
return topology
def _nearest_lower_topology(topology):
"""Degradation order (design §6): notice + nearest-lower topology.
multiN → dual → single → None. Returns the next topology to try, or None
when there is nowhere lower to fall.
"""
if topology.startswith("multi"):
return "dual"
if topology == "dual":
return "single"
return None
def engine_set():
"""The closed set of engine namespace-prefixes (DEFAULTS keys + ranked).
`X/default` dispatch (design §13.1): `X ∈ engine_set` → engine
recommendation; else `X ∈ model_set` → model default; else error.
Engines and model-ids are disjoint by construction.
"""
engines = set()
for _model, engine, _topology in DEFAULTS:
engines.add(engine)
for ranked in ENGINE_PREFERENCE.values():
engines.update(ranked)
return engines
def model_set():
"""The set of model-ids that appear in DEFAULTS (the runnable catalog)."""
return {model for (model, _engine, _topology) in DEFAULTS}
def _functional_default(model, engine, topology):
"""A DEFAULTS slug for (model, engine, topology) whose status is functional.
Returns the slug only when an entry exists AND its registry status is NOT
in the (NA) set (experimental/preview/upstream-gated/deprecated) — a
broken/preview config must never become someone's auto-default (§12.5).
Returns None otherwise.
"""
slug = DEFAULTS.get((model, engine, topology))
if not slug:
return None
entry = COMPOSE_REGISTRY.get(slug)
if entry is None:
return None
if entry.get("status", "production") not in FUNCTIONAL_STATUSES:
return None
return slug
def curated_default_target(model, topology):
"""Curated fallback (§4): walk ENGINE_PREFERENCE[family], first functional
DEFAULTS slug wins. Returns the slug, or None if no functional curated
default exists for (model, topology).
"""
family = _topology_family(topology)
for engine in ENGINE_PREFERENCE.get(family, []):
slug = _functional_default(model, engine, topology)
if slug:
return slug
return None
def community_default_target(model, topology, hw_class=None): # noqa: ARG001
"""Community-ranked best config — the FUTURE middle precedence rung (§13.4).
Contract: returns a ranked slug when the submissions/ranking app exists;
returns None today (always skipped). The resolver inserts a non-None result
BETWEEN the user pin and the curated fallback. v1 ships this stub returning
None so the ladder rung is real, not aspirational; a test asserts it is
skipped.
"""
return None
def model_default_pin_key(model):
"""The .env key for a per-model user pin (design §13.2).
`CLUB3090_DEFAULT_<MODELID uppercased, non-alnum→_>`, e.g.
qwen3.6-27b → CLUB3090_DEFAULT_QWEN3_6_27B.
"""
suffix = "".join(c if c.isalnum() else "_" for c in model).upper()
return f"CLUB3090_DEFAULT_{suffix}"
def model_of_slug(slug):
"""The model-id a slug belongs to, or None if the slug is unknown."""
entry = COMPOSE_REGISTRY.get(slug)
return entry.get("model") if entry else None
def slug_topology(slug):
"""The topology family a slug serves, derived from its compose_path.
compose_path is `models/<model>/<engine>/compose/<topology>/<quant>/...`.
Returns `single`/`dual`/`multi` (the ENGINE_PREFERENCE family) or None.
"""
entry = COMPOSE_REGISTRY.get(slug)
if not entry:
return None
cp = entry.get("compose_path", "")
if "/compose/" not in cp:
return None
after = cp.split("/compose/", 1)[1]
topo = after.split("/", 1)[0]
return _topology_family(topo)