Registers the new gemma4_unified 12B model (vLLM PR #44429, merged 2026-06-03) as catalog slugs vllm/gemma-12b (base) + vllm/gemma-12b-mtp (assistant drafter n=4), so launch.sh/switch.sh resolve them by slug. Status: experimental. - ModelProfile (family gemma4-unified; 48L/8 full+40 sliding, 16/8 GQA, head_dim 256/512, SWA 1024, max_ctx 131072 — past that vLLM CUDA-OOBs, vllm#39914). - Engine profile vllm-gemma4-unified pins the gemma4-unified PREVIEW image (NOT stable v0.22.0, which lacks the arch). Caveat: ephemeral tag — pin a digest before any Production promotion. - Drafter gemma-12b-it-assistant (n=4, 0.85GB), bf16 weights, 2 registry entries, no DEFAULTS row (experimental). kv-calc made gemma-4-12b-aware via the shared Gemma dense path + measured calibration anchor (384,019 tok / 8.16 GiB @ MTP/TP2/131072). - ADDING_MODELS.md: name the central registry (compose_registry.py SoT → both launchers) explicitly in the intro + path-3. KNOWN: kv-calc under-predicts the live pool (~204K vs measured 384K tok, -47%) — the shared Gemma dense formula over-prices gemma4_unified global-layer KV; --calibration GB-verdict still PASSES (TIGHT). Recorded in the calibration YAML header; refine the gemma4_unified global-KV model as a follow-up. Guard suite 41/41 green. Co-Authored-By: Claude Opus 4.8 (1M context) <[email protected]>
900 lines
50 KiB
Python
900 lines
50 KiB
Python
"""Static compose-to-profile bridge for v0.7.0.
|
||
|
||
The registry intentionally mirrors the shipped compose files. It is not a
|
||
generator and it does not attempt to normalize away historical variants.
|
||
"""
|
||
|
||
# Slug lifecycle / availability statuses — the canonical health flag.
|
||
#
|
||
# These are the registry-side equivalent of the compose `Status:` header enum
|
||
# (see the repo CLAUDE.md "Status enum" table). The compose-header emoji maps
|
||
# to one of these words; the drift-guard test asserts the two never diverge.
|
||
#
|
||
# functional → launches normally (production) or with a one-line notice
|
||
# (caveats).
|
||
# (NA) → surfaced in --list but not reliable: launch warns and requires
|
||
# --force so a user can't *unknowingly* boot a broken slug.
|
||
STATUS_VALUES = (
|
||
"production", # ✅ Production — recommended, fully validated.
|
||
"caveats", # ⚠️ Production w/ caveats — works under documented limits.
|
||
"experimental", # 🧪 Experimental — under active validation; may not boot.
|
||
"preview", # 👁️ Preview — known quality issues; tracked, not for prod.
|
||
"upstream-gated", # ⏸️ Upstream-gated — blocked by external action (pin/PR/HW).
|
||
"deprecated", # 🗑️ Deprecated — kept for reference; flagged for removal.
|
||
)
|
||
|
||
# Statuses that launch without --force. Everything else is "(NA)".
|
||
FUNCTIONAL_STATUSES = frozenset({"production", "caveats"})
|
||
|
||
# Compose `Status:` header emoji → registry status word. The header may carry
|
||
# trailing prose after the canonical token (e.g. "✅ Production (NEW — ...)");
|
||
# matching is by the leading emoji, so prose is tolerated.
|
||
COMPOSE_STATUS_EMOJI = {
|
||
"✅": "production",
|
||
"⚠️": "caveats",
|
||
"🧪": "experimental",
|
||
"👁️": "preview",
|
||
"⏸️": "upstream-gated",
|
||
"🗑️": "deprecated",
|
||
}
|
||
|
||
|
||
def _entry(
|
||
*,
|
||
model,
|
||
weights_variant,
|
||
workload,
|
||
engine,
|
||
drafter,
|
||
kv_format,
|
||
tp,
|
||
max_ctx,
|
||
max_num_seqs,
|
||
mem_util,
|
||
compose_path,
|
||
default_port,
|
||
kvcalc_key=None,
|
||
requires_nvlink=False,
|
||
required_engine_features=None,
|
||
recommended_engine_features=None,
|
||
required_sm=None,
|
||
status="production",
|
||
status_note=None,
|
||
):
|
||
if status not in STATUS_VALUES:
|
||
raise ValueError(
|
||
f"{compose_path}: status={status!r} not in {STATUS_VALUES}"
|
||
)
|
||
entry = {
|
||
"model": model,
|
||
"weights_variant": weights_variant,
|
||
"workload": workload,
|
||
"engine": engine,
|
||
"drafter": drafter,
|
||
"kv_format": kv_format,
|
||
"tp": tp,
|
||
"pp": 1,
|
||
"max_ctx": max_ctx,
|
||
"max_num_seqs": max_num_seqs,
|
||
"mem_util": mem_util,
|
||
"compose_path": compose_path,
|
||
"requires_nvlink": requires_nvlink,
|
||
"required_engine_features": list(required_engine_features or []),
|
||
"default_port": default_port,
|
||
"gpu_assignment_mode": "contiguous",
|
||
"kvcalc_key": kvcalc_key,
|
||
"status": status,
|
||
"status_note": status_note,
|
||
}
|
||
if recommended_engine_features:
|
||
entry["recommended_engine_features"] = list(recommended_engine_features)
|
||
if required_sm is not None:
|
||
entry["required_sm"] = required_sm
|
||
return entry
|
||
|
||
|
||
def compose_header_status(text):
|
||
"""Map a compose file's profile-schema `Status:` header to a status word.
|
||
|
||
Reads ONLY the `Status:` line inside the leading `# Profile (at-a-glance):`
|
||
comment block (the structured schema), stopping at the `# ---` separator so
|
||
a free-form `# Status: ...` prose line further down can't be mistaken for it.
|
||
Returns the status word (one of STATUS_VALUES) or None if no canonical
|
||
emoji is found. Matching is by the leading enum emoji, so trailing prose
|
||
after the canonical token (e.g. "✅ Production (NEW — ...)") is tolerated.
|
||
"""
|
||
in_schema = False
|
||
for line in text.splitlines():
|
||
stripped = line.strip()
|
||
if stripped.startswith("# Profile (at-a-glance):"):
|
||
in_schema = True
|
||
continue
|
||
if not in_schema:
|
||
continue
|
||
# The schema block ends at the dashed separator line.
|
||
if stripped.startswith("# --") or stripped.startswith("#--"):
|
||
break
|
||
# Match "# Status: <emoji> ..." within the schema block.
|
||
body = stripped.lstrip("#").strip()
|
||
if body.startswith("Status:"):
|
||
value = body[len("Status:"):].strip()
|
||
for emoji, word in COMPOSE_STATUS_EMOJI.items():
|
||
if value.startswith(emoji):
|
||
return word
|
||
return None
|
||
return None
|
||
|
||
|
||
COMPOSE_REGISTRY = {
|
||
# Qwen 3.6 27B, vLLM single-card.
|
||
"vllm/default": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="tool-heavy",
|
||
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="turboquant_3bit_nc",
|
||
tp=1, max_ctx=48000, max_num_seqs=1, mem_util=0.92,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/single/autoround-int4/tq3-mtp.yml",
|
||
default_port=8020, required_engine_features=["turboquant_3bit_nc"],
|
||
kvcalc_key="qwen3.6-27b:long-vision",
|
||
status="deprecated",
|
||
status_note="DEPRECATED 2026-05-31 — Genesis/nightly path on hold (Sandermage); stack moving to stable vLLM. The vLLM single-card default repointed to vllm/minimal (v0.22.0). Use vllm/minimal or beellama single-card.",
|
||
),
|
||
"vllm/long-text": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
|
||
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="turboquant_3bit_nc",
|
||
tp=1, max_ctx=180000, max_num_seqs=1, mem_util=0.93,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/single/autoround-int4/long-text.yml",
|
||
default_port=8020, required_engine_features=["turboquant_3bit_nc"],
|
||
kvcalc_key="qwen3.6-27b:long-text",
|
||
status="deprecated",
|
||
status_note="DEPRECATED 2026-05-31 — Genesis/nightly path on hold (Sandermage) + Cliff 2b >50K; stack moving to stable vLLM. Use vllm/minimal (single) or beellama single-card.",
|
||
),
|
||
"vllm/long-text-no-mtp": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
|
||
engine="vllm-nightly-mtp", drafter=None, kv_format="turboquant_3bit_nc",
|
||
tp=1, max_ctx=200000, max_num_seqs=1, mem_util=0.95,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/single/autoround-int4/long-text-no-mtp.yml",
|
||
default_port=8021, required_engine_features=["turboquant_3bit_nc"],
|
||
kvcalc_key="qwen3.6-27b:long-text-no-mtp",
|
||
status="deprecated",
|
||
status_note="DEPRECATED 2026-05-31 — Genesis/nightly path on hold (Sandermage) + Cliff 2b >50K; stack moving to stable vLLM. Use vllm/minimal (single) or beellama single-card.",
|
||
),
|
||
"vllm/long-vision": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="vision-coding",
|
||
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="turboquant_3bit_nc",
|
||
tp=1, max_ctx=145000, max_num_seqs=1, mem_util=0.95,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/single/autoround-int4/long-vision.yml",
|
||
default_port=8020, required_engine_features=["turboquant_3bit_nc"],
|
||
kvcalc_key="qwen3.6-27b:long-vision",
|
||
status="deprecated",
|
||
status_note="DEPRECATED 2026-05-31 — Genesis/nightly path on hold (Sandermage) + Cliff 2b >50K; stack moving to stable vLLM. Use vllm/minimal (single) or beellama single-card.",
|
||
),
|
||
"vllm/bounded-thinking": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="tool-heavy",
|
||
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="turboquant_3bit_nc",
|
||
tp=1, max_ctx=180000, max_num_seqs=1, mem_util=0.95,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/single/autoround-int4/bounded-thinking.yml",
|
||
default_port=8020, required_engine_features=["turboquant_3bit_nc"],
|
||
kvcalc_key="qwen3.6-27b:bounded-thinking",
|
||
status="deprecated",
|
||
status_note="DEPRECATED 2026-05-31 — Genesis/nightly path on hold (Sandermage); stack moving to stable vLLM. Use vllm/minimal (single) or beellama single-card.",
|
||
),
|
||
"vllm/tools-text": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="tool-heavy",
|
||
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||
tp=1, max_ctx=75000, max_num_seqs=1, mem_util=0.97,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/single/autoround-int4/tools-text.yml",
|
||
default_port=8020,
|
||
kvcalc_key="qwen3.6-27b:tools-text",
|
||
status="deprecated",
|
||
status_note="DEPRECATED 2026-05-31 — Genesis/nightly path on hold (Sandermage); stack moving to stable vLLM. Use vllm/minimal (single) or beellama single-card.",
|
||
),
|
||
"vllm/minimal": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="fast-chat",
|
||
engine="vllm-nightly-clean", drafter=None, kv_format="fp8_e5m2",
|
||
tp=1, max_ctx=32768, max_num_seqs=1, mem_util=0.92,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/single/autoround-int4/minimal.yml",
|
||
default_port=8020,
|
||
kvcalc_key="qwen3.6-27b:minimal",
|
||
),
|
||
|
||
# Qwen 3.6 27B, vLLM dual/multi-card.
|
||
"vllm/dual": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
|
||
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/fp8-mtp.yml",
|
||
default_port=8010, recommended_engine_features=["marlin_pad_sub_tile_n"],
|
||
kvcalc_key="qwen3.6-27b:dual",
|
||
),
|
||
"vllm/dual-turbo": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="multi-stream-tenant",
|
||
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="turboquant_3bit_nc",
|
||
tp=2, max_ctx=262144, max_num_seqs=4, mem_util=0.85,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/turbo.yml",
|
||
default_port=8011, required_engine_features=["turboquant_3bit_nc"],
|
||
kvcalc_key="qwen3.6-27b:dual-turbo",
|
||
recommended_engine_features=["marlin_pad_sub_tile_n"],
|
||
status="deprecated",
|
||
status_note="DEPRECATED 2026-05-31 — Genesis/nightly path on hold (Sandermage); stack moving to stable vLLM. Use vllm/dual (v0.22.0) for dual-card.",
|
||
),
|
||
"vllm/dual-dflash": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="vision-coding",
|
||
engine="vllm-nightly-dflash", drafter="zlab-qwen-dflash", kv_format="fp16",
|
||
tp=2, max_ctx=185000, max_num_seqs=1, mem_util=0.95,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/dflash.yml",
|
||
default_port=8012, required_engine_features=["marlin_pad_sub_tile_n"],
|
||
kvcalc_key="qwen3.6-27b:dual-dflash",
|
||
status="deprecated",
|
||
status_note="Pruned 2026-05-31: superseded by vllm/dual (fp8, 262K, vision, MTP, stock v0.22.0). DFlash traded ctx/concurrency (185K, 1 stream) for code TPS; recover from git if demand returns.",
|
||
),
|
||
"vllm/dual-dflash-noviz": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
|
||
engine="vllm-nightly-dflash", drafter="zlab-qwen-dflash", kv_format="fp16",
|
||
tp=2, max_ctx=200000, max_num_seqs=1, mem_util=0.95,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/dflash-noviz.yml",
|
||
default_port=8013, required_engine_features=["marlin_pad_sub_tile_n"],
|
||
kvcalc_key="qwen3.6-27b:dual-dflash-noviz",
|
||
status="deprecated",
|
||
status_note="Pruned 2026-05-31: the lone no-vision dual A/B; dropped per the single-vision-config policy (vllm/dual covers vision + 262K + 2 streams). Recover from git if a text-only path is needed.",
|
||
),
|
||
"vllm/dual-bf16": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
|
||
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="bf16",
|
||
tp=2, max_ctx=200000, max_num_seqs=1, mem_util=0.92,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/bf16.yml",
|
||
default_port=8012,
|
||
kvcalc_key="qwen3.6-27b:dual-bf16",
|
||
status="deprecated",
|
||
status_note="Pruned 2026-05-31: was a matched-config A/B vs Gemma bf16.yml, never validated on Qwen3-Next DeltaNet. Superseded by vllm/dual (fp8, 262K).",
|
||
),
|
||
"vllm/dual-int8": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
|
||
engine="vllm-nightly-full", drafter="qwen-mtp-builtin", kv_format="int8_per_token_head",
|
||
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/int8.yml",
|
||
default_port=8011, required_engine_features=["int8_per_token_head"],
|
||
kvcalc_key="qwen3.6-27b:dual-int8",
|
||
status="deprecated",
|
||
status_note="Pruned 2026-05-31: was a matched-config A/B vs Gemma int8.yml; never validated on Qwen DeltaNet, and fp8 is native on Qwen so int8 PTH buys nothing. Superseded by vllm/dual.",
|
||
),
|
||
"vllm/dual-tq3-mtp": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="multi-stream-tenant",
|
||
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="turboquant_3bit_nc",
|
||
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/tq3-mtp.yml",
|
||
default_port=8013, required_engine_features=["turboquant_3bit_nc"],
|
||
kvcalc_key="qwen3.6-27b:dual-tq3-mtp",
|
||
status="deprecated",
|
||
status_note="Tombstoned 2026-05-11 — Genesis-free TQ3+MTP needs 4 of 5 upstream fixes not yet landed. Use dual-tq3-mtp-genesis (gated) or dual-turbo.",
|
||
),
|
||
"vllm/dual-tq3-mtp-genesis": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="multi-stream-tenant",
|
||
engine="vllm-nightly-mtp", drafter="qwen-mtp-builtin", kv_format="turboquant_3bit_nc",
|
||
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.85,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/tq3-mtp-genesis.yml",
|
||
default_port=8015, required_engine_features=["turboquant_3bit_nc"],
|
||
kvcalc_key="qwen3.6-27b:dual-tq3-mtp-genesis",
|
||
status="deprecated",
|
||
status_note="DEPRECATED 2026-05-31 — Genesis on hold pending Sandermage; stack moving to stable vLLM. Use vllm/dual (v0.22.0). (Was upstream-gated on the parked/drifted Genesis pin — see docs/UPSTREAM.md.)",
|
||
),
|
||
"vllm/dual-tq3-nomtp": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
|
||
engine="vllm-nightly-mtp", drafter=None, kv_format="turboquant_3bit_nc",
|
||
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/dual/autoround-int4/tq3-nomtp.yml",
|
||
default_port=8014, required_engine_features=["turboquant_3bit_nc"],
|
||
kvcalc_key="qwen3.6-27b:dual-tq3-nomtp",
|
||
status="deprecated",
|
||
status_note="Pruned 2026-05-31: superseded by vllm/dual (fp8, same 262K/2-stream, faster + MTP). TQ3 KV density not worth the decode cost here; recover from git if needed.",
|
||
),
|
||
"vllm/dual-carnice-bf16mtp": _entry(
|
||
model="qwen3.6-27b", weights_variant="carnice-bf16mtp", workload="long-ctx-single",
|
||
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/dual/carnice-bf16mtp/bf16-mtp.yml",
|
||
default_port=8070,
|
||
kvcalc_key="SKIP",
|
||
status="caveats",
|
||
status_note="Carnice fine-tune MTP AL=2.0 vs Lorbus 3.4-3.8 (working but suboptimal TPS).",
|
||
),
|
||
"vllm/dual-qwopus-bf16mtp": _entry(
|
||
model="qwen3.6-27b", weights_variant="qwopus-bf16mtp", workload="long-ctx-single",
|
||
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||
tp=2, max_ctx=262144, max_num_seqs=2, mem_util=0.92,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/dual/qwopus-bf16mtp/bf16-mtp.yml",
|
||
default_port=8071,
|
||
kvcalc_key="SKIP",
|
||
status="preview",
|
||
status_note="Qwopus fine-tune preview: line repetition + NIAH drop + silent-empty turn-5 in soak.",
|
||
),
|
||
"vllm/dual4": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="multi-stream-tenant",
|
||
engine="vllm-nightly-clean", drafter="qwen-mtp-builtin", kv_format="fp8_e5m2",
|
||
tp=4, max_ctx=262144, max_num_seqs=4, mem_util=0.92,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/multi4/autoround-int4/fp8-mtp.yml",
|
||
default_port=8015,
|
||
kvcalc_key="qwen3.6-27b:dual4",
|
||
),
|
||
"vllm/dual4-dflash": _entry(
|
||
model="qwen3.6-27b", weights_variant="autoround-int4", workload="long-ctx-single",
|
||
engine="vllm-nightly-dflash", drafter="zlab-qwen-dflash", kv_format="fp16",
|
||
tp=4, max_ctx=262144, max_num_seqs=2, mem_util=0.95,
|
||
compose_path="models/qwen3.6-27b/vllm/compose/multi4/autoround-int4/dflash.yml",
|
||
default_port=8016, required_engine_features=["marlin_pad_sub_tile_n"],
|
||
kvcalc_key="qwen3.6-27b:dual4-dflash",
|
||
),
|
||
|
||
# Qwen 3.6 27B, llama.cpp single-card.
|
||
# `llamacpp/default` is an alias for `llamacpp/mtp` (collapsed 2026-05-22):
|
||
# the old Q3_K_XL vanilla compose was retired and `default` now points at
|
||
# the MTP compose. max_ctx = the 200K max-safe default (262K boots but walls
|
||
# ~125K at fill — see docs/CLIFFS.md; runtime CTX_SIZE default is 200000).
|
||
"llamacpp/default": _entry(
|
||
model="qwen3.6-27b", weights_variant="unsloth-q4km", workload="fast-chat",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
|
||
tp=1, max_ctx=200000, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/llama-cpp/compose/single/unsloth-q4km/mtp.yml",
|
||
default_port=8020,
|
||
kvcalc_key="SKIP",
|
||
),
|
||
"llamacpp/mtp": _entry(
|
||
model="qwen3.6-27b", weights_variant="unsloth-q4km", workload="fast-chat",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
|
||
tp=1, max_ctx=200000, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/llama-cpp/compose/single/unsloth-q4km/mtp.yml",
|
||
default_port=8020,
|
||
kvcalc_key="SKIP",
|
||
),
|
||
"llamacpp/bounded-thinking": _entry(
|
||
model="qwen3.6-27b", weights_variant="unsloth-q4km", workload="tool-heavy",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
|
||
tp=1, max_ctx=200000, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/llama-cpp/compose/single/unsloth-q4km/bounded-thinking.yml",
|
||
default_port=8020,
|
||
kvcalc_key="SKIP",
|
||
status="experimental",
|
||
status_note="New structured-CoT port; live grammar + MTP validation pending.",
|
||
),
|
||
"llamacpp/mtp-vision": _entry(
|
||
model="qwen3.6-27b", weights_variant="unsloth-q4km", workload="vision-coding",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
|
||
# 150K @ 1M-px (IMAGE_MAX_TOKENS=1024) — re-tuned 2026-05-25 (PR #227); was a
|
||
# stale 49152. Full-res 4M-px OOMs at fill, so 1M-px is the safe default.
|
||
tp=1, max_ctx=150000, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/llama-cpp/compose/single/unsloth-q4km/mtp-vision.yml",
|
||
default_port=8020,
|
||
kvcalc_key="SKIP",
|
||
),
|
||
|
||
# ik_llama.cpp — IQ4_KS (ubergarm). Same engine family as llamacpp, but the
|
||
# IQK quant is ~0.5-0.8 GB leaner on weights → best fit for VRAM-tight
|
||
# single-card (sub-24 GB, shared GPU, WSL display overhead). Its own image
|
||
# (ikawrakow/ik-llama-cpp), so unaffected by mainline llama.cpp drift.
|
||
"ik-llama/iq4ks-mtp": _entry(
|
||
model="qwen3.6-27b", weights_variant="ubergarm-iq4ks", workload="fast-chat",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
|
||
tp=1, max_ctx=200000, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/ik-llama/compose/single/ubergarm-iq4ks/mtp.yml",
|
||
default_port=8020,
|
||
kvcalc_key="SKIP",
|
||
),
|
||
"ik-llama/iq4ks-mtp-vision": _entry(
|
||
model="qwen3.6-27b", weights_variant="ubergarm-iq4ks", workload="vision-coding",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
|
||
tp=1, max_ctx=163840, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/ik-llama/compose/single/ubergarm-iq4ks/mtp-vision.yml",
|
||
default_port=8020,
|
||
kvcalc_key="SKIP",
|
||
),
|
||
"ik-llama/iq4ks-two-stage": _entry(
|
||
model="qwen3.6-27b", weights_variant="ubergarm-iq4ks", workload="fast-chat",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
|
||
tp=1, max_ctx=200000, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/ik-llama/compose/single/ubergarm-iq4ks/two-stage.yml",
|
||
default_port=8020,
|
||
kvcalc_key="SKIP",
|
||
),
|
||
|
||
# Qwen3.6-27B beellama.cpp DFlash — single-card DEFAULT (DFlash spec-dec,
|
||
# Q5_K_S target + Anbeeld DFlash-IQ4_XS draft, q5_0(K)/q4_1(V) KV). beellama
|
||
# is a llama.cpp-family engine (kvcalc SKIP, like ik-llama). Promoted to the
|
||
# single-GPU default 2026-05-30: code-throughput leader (~100 TPS) + slight
|
||
# 8-pack quality edge (107 vs ik 99, think-off) + output-lossless spec-dec.
|
||
# Served via our UNOFFICIAL multi-arch image (sm_86/89/120 = 3090/4090/5090);
|
||
# sm_89/sm_120 are compiled but unvalidated on our 3090-only rig — see Caveats
|
||
# in the compose. kv_format reflects K-side precision (V is q4_1).
|
||
"beellama/dflash": _entry(
|
||
model="qwen3.6-27b", weights_variant="beellama-q5ks-dflash", workload="fast-chat",
|
||
engine="beellama-local", drafter="anbeeld-qwen-dflash", kv_format="q5_0",
|
||
tp=1, max_ctx=102400, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/beellama/compose/single/beellama-q5ks-dflash/dflash.yml",
|
||
default_port=8060,
|
||
kvcalc_key="SKIP",
|
||
status="caveats",
|
||
status_note="Single-GPU default. Launchers inject Anbeeld's official beellama.cpp server-cuda-v0.3.0 image (sm_86/89 = 3090/4090); sm_89 compiled-not-validated on club-3090's 3090-only rig. 5090/sm_120: prefix BEELLAMA_IMAGE=ghcr.io/noonghunna/beellama-cpp:multiarch-v0.3.0-efe856397 (sm_120 compiled-not-validated). Usable ctx ceiling 160K (200K OOMs on prefill); ships 102K. DFlash prose is net-positive on tok/s (+27% vs no-spec, re-tested 2026-06-03); the earlier 'prose-DFlash regression' is RETRACTED — it was an AR over-read + wrong baseline (docs/UPSTREAM.md).",
|
||
),
|
||
|
||
# Qwen3.6-27B PRISM-PRO-DQ (Ex0bit dynamic-quant GGUF) — community-experimental, ik-llama.
|
||
"ik-llama/prism-pro-dq-mtp": _entry(
|
||
model="qwen3.6-27b", weights_variant="ex0bit-prism-pro-dq", workload="fast-chat",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
|
||
tp=1, max_ctx=122880, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/ik-llama/compose/single/ex0bit-prism-pro-dq/mtp.yml",
|
||
default_port=8020,
|
||
kvcalc_key="SKIP",
|
||
status="experimental",
|
||
status_note="PRISM-PRO-DQ community dynamic-quant GGUF — eval-only, not yet validated.",
|
||
),
|
||
"ik-llama/prism-pro-dq-long": _entry(
|
||
model="qwen3.6-27b", weights_variant="ex0bit-prism-pro-dq", workload="long-ctx-single",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
|
||
tp=1, max_ctx=180000, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/ik-llama/compose/single/ex0bit-prism-pro-dq/long.yml",
|
||
default_port=8052,
|
||
kvcalc_key="SKIP",
|
||
status="experimental",
|
||
status_note="PRISM-PRO-DQ community dynamic-quant GGUF — eval-only, not yet validated.",
|
||
),
|
||
"ik-llama/prism-pro-dq-two-stage": _entry(
|
||
model="qwen3.6-27b", weights_variant="ex0bit-prism-pro-dq", workload="tool-heavy",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
|
||
tp=1, max_ctx=200000, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/ik-llama/compose/single/ex0bit-prism-pro-dq/two-stage.yml",
|
||
default_port=8020,
|
||
kvcalc_key="SKIP",
|
||
status="experimental",
|
||
status_note="PRISM-PRO-DQ community dynamic-quant GGUF — eval-only, not yet validated.",
|
||
),
|
||
"ik-llama/prism-pro-dq-dual": _entry(
|
||
model="qwen3.6-27b", weights_variant="ex0bit-prism-pro-dq", workload="tool-heavy",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
|
||
tp=2, max_ctx=196608, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/ik-llama/compose/dual/ex0bit-prism-pro-dq/mtp.yml",
|
||
default_port=8053,
|
||
kvcalc_key="SKIP",
|
||
status="experimental",
|
||
status_note="PRISM-PRO-DQ community dynamic-quant GGUF — eval-only, not yet validated.",
|
||
),
|
||
"ik-llama/prism-pro-dq-dual-vision": _entry(
|
||
model="qwen3.6-27b", weights_variant="ex0bit-prism-pro-dq", workload="vision-coding",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q8_0",
|
||
tp=2, max_ctx=262144, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/ik-llama/compose/dual/ex0bit-prism-pro-dq/mtp-vision.yml",
|
||
default_port=8010,
|
||
kvcalc_key="SKIP",
|
||
status="experimental",
|
||
status_note="PRISM-PRO-DQ community dynamic-quant GGUF — eval-only, not yet validated.",
|
||
),
|
||
# Qwen3.6-35B-A3B APEX-MTP (mudler MoE GGUF — Compact + Quality) — community-experimental, ik-llama.
|
||
"ik-llama/apex-mtp-compact": _entry(
|
||
model="qwen3.6-35b-a3b", weights_variant="mudler-apex-compact", workload="fast-chat",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
|
||
tp=1, max_ctx=163840, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-35b-a3b/ik-llama/compose/single/mudler-apex-compact/mtp.yml",
|
||
default_port=8054,
|
||
kvcalc_key="SKIP",
|
||
status="experimental",
|
||
status_note="APEX-MTP community MoE GGUF — eval-only bring-up lane, not yet validated.",
|
||
),
|
||
"ik-llama/byteshape-iq4xs-mtp": _entry(
|
||
model="qwen3.6-35b-a3b", weights_variant="byteshape-iq4xs", workload="fast-chat",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q4_0",
|
||
tp=1, max_ctx=262144, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-35b-a3b/ik-llama/compose/single/byteshape-iq4xs/mtp.yml",
|
||
default_port=8058,
|
||
kvcalc_key="SKIP",
|
||
status="caveats",
|
||
status_note="byteshape IQ4_XS 4.19bpw MoE GGUF (embedded MTP head) — community intake from PR #293 (@Rhonstin). Single-card 35B-A3B, q4_0 KV + --fit → 262K. First-party validated 2026-06-02 on 1× 3090: verify-full all-pass, verify-stress 8/8 (NIAH→240K, no Cliff), bench n=5 (narrative 113/116 · code 129/137 wall/decode TPS, CV<2.3%), 8-pack 110/150 (≈ author's 111/150), soak-continuous PASS (0 err, 0 VRAM growth, 0/25 silent-empty). Caveat: single-rig; agent packs modest (hermes 55%, cli 42%) as typical for the class. Intake fixes vs #293: image cu13, port 8058.",
|
||
),
|
||
"ik-llama/apex-mtp-compact-long": _entry(
|
||
model="qwen3.6-35b-a3b", weights_variant="mudler-apex-compact", workload="long-ctx-single",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q8_0",
|
||
tp=1, max_ctx=196608, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-35b-a3b/ik-llama/compose/single/mudler-apex-compact/long.yml",
|
||
default_port=8056,
|
||
kvcalc_key="SKIP",
|
||
status="experimental",
|
||
status_note="APEX-MTP community MoE GGUF — eval-only bring-up lane, not yet validated.",
|
||
),
|
||
# @laurimyllari's --fit + asymmetric q8_0(K)/q5_0(V) KV config from
|
||
# discussion #241, retuned + measured on 1× 3090. +7% narr / +4% code
|
||
# over the q4/q4 mtp.yml sibling on the same APEX I-Compact GGUF.
|
||
# kv_format "q8_0" reflects K-side precision; V is q5_0 (see compose).
|
||
"ik-llama/apex-fit-q8q5": _entry(
|
||
model="qwen3.6-35b-a3b", weights_variant="mudler-apex-compact", workload="fast-chat",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q8_0",
|
||
tp=1, max_ctx=196608, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-35b-a3b/ik-llama/compose/single/mudler-apex-compact/fit-mtp.yml",
|
||
default_port=8057,
|
||
kvcalc_key="SKIP",
|
||
),
|
||
"ik-llama/apex-mtp-quality-dual": _entry(
|
||
model="qwen3.6-35b-a3b", weights_variant="mudler-apex-quality", workload="long-ctx-single",
|
||
engine="llama-cpp-local", drafter="qwen-mtp-builtin", kv_format="q8_0",
|
||
tp=2, max_ctx=196608, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-35b-a3b/ik-llama/compose/dual/mudler-apex-quality/mtp.yml",
|
||
default_port=8055,
|
||
kvcalc_key="SKIP",
|
||
status="experimental",
|
||
status_note="APEX-MTP community MoE GGUF — eval-only bring-up lane, not yet validated.",
|
||
),
|
||
|
||
# Gemma 4 31B, vLLM. Lean v0.21.0 set: bf16 default, int8 long-context, single-card fp8 risk path.
|
||
"vllm/gemma-mtp-tp1": _entry(
|
||
model="gemma-4-31b", weights_variant="autoround-int4", workload="fast-chat",
|
||
engine="vllm-gemma-stable", drafter="gemma-it-assistant", kv_format="fp8_e4m3",
|
||
tp=1, max_ctx=8192, max_num_seqs=256, mem_util=0.95,
|
||
compose_path="models/gemma-4-31b/vllm/compose/single/autoround-int4/fp8-mtp.yml",
|
||
default_port=8031, required_sm=9.0,
|
||
kvcalc_key="gemma-4-31b:gemma-single",
|
||
status="deprecated",
|
||
status_note="Dead on Ampere: no fp8 KV path for Gemma 4 on sm_86 (attention asserts kv ∈ {fp8,fp8_e4m3,nvfp4} — rejects fp8_e5m2; fp8/fp8_e4m3 need the fp8e4nv kernel sm_86 lacks; nvfp4 Blackwell-only). Live-confirmed stock v0.22.0 2026-05-31. Single-card → beellama/gemma-dflash; dual → vllm/gemma-bf16-mtp. See compose Caveats.",
|
||
),
|
||
"vllm/gemma-bf16-mtp": _entry(
|
||
model="gemma-4-31b", weights_variant="autoround-int4", workload="fast-chat",
|
||
engine="vllm-gemma-stable", drafter="gemma-it-assistant", kv_format="bf16",
|
||
tp=2, max_ctx=131072, max_num_seqs=4, mem_util=0.95,
|
||
compose_path="models/gemma-4-31b/vllm/compose/dual/autoround-int4/bf16-mtp.yml",
|
||
default_port=8030,
|
||
kvcalc_key="gemma-4-31b:gemma-dual",
|
||
),
|
||
"vllm/gemma-int8-mtp": _entry(
|
||
model="gemma-4-31b", weights_variant="autoround-int4", workload="multi-stream-tenant",
|
||
engine="vllm-gemma-stable", drafter="gemma-it-assistant", kv_format="int8_per_token_head",
|
||
tp=2, max_ctx=262144, max_num_seqs=4, mem_util=0.95,
|
||
compose_path="models/gemma-4-31b/vllm/compose/dual/autoround-int4/int8.yml",
|
||
default_port=8032, required_engine_features=["int8_per_token_head"],
|
||
kvcalc_key="gemma-4-31b:gemma-dual-int8",
|
||
),
|
||
|
||
# Gemma-4-12B (gemma4_unified arch — vLLM PR #44429, merged 2026-06-03),
|
||
# dual-3090 bf16 on the ephemeral vllm/vllm-openai:gemma4-unified preview
|
||
# image (== today's vLLM main; Gemma-4 fixes baked in, nothing vendored).
|
||
# bf16 weights (~24 GB) don't fit one 24 GB card with KV → TP=2 mandatory.
|
||
# Both EXPERIMENTAL: arch-preview image, sm_86 gemma4_unified support
|
||
# unverified, ephemeral tag (pin a digest before promotion). Max ctx 131072
|
||
# is the HARD ceiling (card claims 256K but vLLM OOB-crashes past the
|
||
# trained max, vllm#39914). kvcalc routes through the shared Gemma dense
|
||
# path (gemma4_unified TEXT backbone == gemma4-swa-dense KV family).
|
||
"vllm/gemma-12b": _entry(
|
||
model="gemma-4-12b", weights_variant="bf16", workload="fast-chat",
|
||
engine="vllm-gemma4-unified", drafter=None, kv_format="bf16",
|
||
tp=2, max_ctx=131072, max_num_seqs=4, mem_util=0.90,
|
||
compose_path="models/gemma-4-12b/vllm/compose/dual/bf16/base.yml",
|
||
default_port=8035,
|
||
kvcalc_key="gemma-4-12b:gemma-dual",
|
||
status="experimental",
|
||
status_note="Gemma-4-12B (gemma4_unified, vLLM PR #44429) dual-3090 bf16, no drafter. Ephemeral gemma4-unified arch-preview image; sm_86 support UNVERIFIED. Max ctx 131072 = hard ceiling (256K card-claim OOB-crashes vLLM past the trained max, vllm#39914). Pin a digest before any Production promotion.",
|
||
),
|
||
"vllm/gemma-12b-mtp": _entry(
|
||
model="gemma-4-12b", weights_variant="bf16", workload="fast-chat",
|
||
engine="vllm-gemma4-unified", drafter="gemma-12b-it-assistant", kv_format="bf16",
|
||
tp=2, max_ctx=131072, max_num_seqs=4, mem_util=0.90,
|
||
compose_path="models/gemma-4-12b/vllm/compose/dual/bf16/mtp.yml",
|
||
default_port=8036,
|
||
kvcalc_key="gemma-4-12b:gemma-dual",
|
||
status="experimental",
|
||
status_note="Gemma-4-12B (gemma4_unified, vLLM PR #44429) dual-3090 bf16 + assistant spec-dec (n=4). Ephemeral gemma4-unified arch-preview image; sm_86 support UNVERIFIED. Max ctx 131072 = hard ceiling (256K card-claim OOB-crashes vLLM past the trained max, vllm#39914). Pin a digest before any Production promotion.",
|
||
),
|
||
|
||
# Gemma-4-31B beellama.cpp DFlash — single-card DEFAULT (Q4_K_S target +
|
||
# Anbeeld DFlash-IQ4_XS draft, q5_0(K)/q4_1(V) KV). The ONLY viable fast
|
||
# single-card Gemma-4 path: does SWA windowed KV (big ctx) AND Gemma-4
|
||
# spec-dec, where vLLM is FA-walled (head_dim=512), ik-llama walls ~24K, and
|
||
# stock llama.cpp is ~12 TPS (no FA_ALL_QUANTS). Promoted to single-GPU
|
||
# default 2026-05-30 (no functional default existed before). llama.cpp-family
|
||
# → kvcalc SKIP. Re-point to the no-fork mainline path when llama.cpp#23398
|
||
# (Gemma-4 MTP) merges — see docs/UPSTREAM.md.
|
||
"beellama/gemma-dflash": _entry(
|
||
model="gemma-4-31b", weights_variant="beellama-q4ks-dflash", workload="fast-chat",
|
||
engine="beellama-local", drafter="anbeeld-gemma-dflash", kv_format="q5_0",
|
||
tp=1, max_ctx=128000, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/gemma-4-31b/beellama/compose/single/beellama-q4ks-dflash/dflash.yml",
|
||
default_port=8061,
|
||
kvcalc_key="SKIP",
|
||
status="caveats",
|
||
status_note="Single-GPU default — the only viable fast single-card Gemma-4 path on Ampere. Launchers inject Anbeeld's official beellama.cpp server-cuda-v0.3.0 image (sm_86/89 = 3090/4090); sm_89 compiled-not-validated on club-3090's 3090-only rig. 5090/sm_120: prefix BEELLAMA_IMAGE=ghcr.io/noonghunna/beellama-cpp:multiarch-v0.3.0-efe856397 (sm_120 compiled-not-validated). DFlash prose is net-positive on tok/s (+28–31% vs no-spec, re-tested 2026-06-03; earlier 'prose regression' RETRACTED — AR over-read + wrong baseline); re-point to mainline llama.cpp#23398 Gemma-4 MTP when it merges — docs/UPSTREAM.md.",
|
||
),
|
||
# Dual-card beellama Gemma-4 (layer-split, 262K) — PARKED upstream-gated 2026-05-31.
|
||
# Boots + recalls 262K fine, but DFlash spec-dec is broken on multi-GPU in our pinned
|
||
# build (07ac3ce): drafter decode fails, accept 0.357, ~24/38 TPS; --device-draft crashes.
|
||
# Fixes live on Anbeeld's v0.3.0 dev branch (414 commits ahead) but no tagged release yet.
|
||
# Re-test (DFlash-fix AND --spec-type mtp) when a beellama release lands. docs/UPSTREAM.md.
|
||
"beellama/gemma-dflash-dual": _entry(
|
||
model="gemma-4-31b", weights_variant="beellama-q4ks-dflash", workload="fast-chat",
|
||
engine="beellama-local", drafter="anbeeld-gemma-dflash", kv_format="q5_0",
|
||
tp=2, max_ctx=262144, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/gemma-4-31b/beellama/compose/dual/beellama-q4ks-dflash/dflash.yml",
|
||
default_port=8062,
|
||
kvcalc_key="SKIP",
|
||
status="experimental",
|
||
status_note="Dual-card beellama Gemma-4 (layer-split, 262K) on v0.3.0 — RELEASED experimental for community v0.3.0 testing (Anbeeld's request, club-3090#288). Multi-GPU DFlash FIXED on v0.3.0 (GPU cross-ring; validated sm_86 2026-06-01, FA_ALL_QUANTS=1; image injected from beellama-local install.spec = Anbeeld's official server-cuda-v0.3.0 commit tag). Earlier 'v0.3.0-wide DFlash-on-PROSE regression' RETRACTED (2026-06-03) — did NOT reproduce on qwen single+dual or gemma single (DFlash prose net-positive +27–58% vs no-spec); was an AR over-read + wrong baseline. This gemma-dual not separately re-benched. Promote experimental→caveats when Anbeeld tags a STABLE release. docs/UPSTREAM.md.",
|
||
),
|
||
|
||
# ------------------------------------------------------------------
|
||
# beellama v0.3.0 Q8_K_XL dual-card composes — RELEASED experimental for
|
||
# community v0.3.0 testing (Anbeeld's request, club-3090#288). Image
|
||
# injected centrally from engines/beellama-local.yml install.spec
|
||
# (Anbeeld's official server-cuda-v0.3.0 commit tag). Validated 2× 3090
|
||
# sm_86 2026-06-01. kvcalc_key=SKIP (llama.cpp family — no vLLM kv-calc).
|
||
# ------------------------------------------------------------------
|
||
"beellama/qwen-mtp-dual": _entry(
|
||
model="qwen3.6-27b", weights_variant="beellama-q8kxl-mtp", workload="fast-chat",
|
||
engine="beellama-local", drafter="unsloth-mtp-gguf", kv_format="q5_0",
|
||
tp=2, max_ctx=65536, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/beellama/compose/dual/beellama-q8kxl-mtp/mtp.yml",
|
||
default_port=8064,
|
||
kvcalc_key="SKIP",
|
||
status="experimental",
|
||
status_note="Dual-card beellama Qwen3.6-27B Q8_K_XL + embedded MTP head (--spec-type draft-mtp, unsloth-mtp-gguf drafter). v0.3.0 sm_86 2026-06-01: boots + coherent, MTP active (code accept ~0.90, ~58 TPS decode). Ships 65536 safe first-boot ctx; validated robust to ~160K (262K impossible — DeltaNet recurrent draft state hard-pins to one card). High-fidelity Q8 sibling of vllm/dual fp8-mtp. Promote experimental→caveats on a STABLE Anbeeld tag.",
|
||
),
|
||
"beellama/qwen-dflash-dual": _entry(
|
||
model="qwen3.6-27b", weights_variant="beellama-q8kxl-dflash", workload="fast-chat",
|
||
engine="beellama-local", drafter="anbeeld-qwen-dflash", kv_format="q5_0",
|
||
tp=2, max_ctx=262144, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/qwen3.6-27b/beellama/compose/dual/beellama-q8kxl-dflash/dflash.yml",
|
||
default_port=8065,
|
||
kvcalc_key="SKIP",
|
||
status="experimental",
|
||
status_note="Dual-card beellama Qwen3.6-27B Q8_K_XL + DFlash (Anbeeld DFlash-IQ4_XS draft, --spec-type dflash). v0.3.0 sm_86 2026-06-01: boots + coherent at full 262K (fixed draft footprint; tensor-split 0.575,0.425 → ~21.2 GB/card). DFlash prose net-positive on tok/s (+52% vs no-spec @262K, re-tested 2026-06-03; earlier 'prose regression' RETRACTED — AR over-read + wrong baseline). Tool-grammar-neutral spec-dec for Qwen agents (club-3090#237). Promote on a STABLE tag.",
|
||
),
|
||
"beellama/gemma-q8-dflash-dual": _entry(
|
||
model="gemma-4-31b", weights_variant="beellama-q8kxl-dflash", workload="fast-chat",
|
||
engine="beellama-local", drafter="anbeeld-gemma-dflash", kv_format="q5_0",
|
||
tp=2, max_ctx=196608, max_num_seqs=1, mem_util=None,
|
||
compose_path="models/gemma-4-31b/beellama/compose/dual/beellama-q8kxl-dflash/dflash.yml",
|
||
default_port=8066,
|
||
kvcalc_key="SKIP",
|
||
status="experimental",
|
||
status_note="Dual-card beellama Gemma-4-31B Q8_K_XL + DFlash (Anbeeld DFlash-IQ4_XS draft). v0.3.0 sm_86 2026-06-01: 192K balanced ceiling (tensor-split 0.55,0.45 → ~21.4/21.9 GB; 262K OOMs — Gemma full-attn layers grow KV). High-fidelity Q8 sibling of beellama/gemma-dflash-dual (q4ks). Earlier 'v0.3.0-wide DFlash-on-PROSE regression' RETRACTED (2026-06-03) — didn't reproduce on qwen single+dual or gemma single; was an AR over-read + wrong baseline. This gemma-Q8-dual not separately re-benched. Promote on a STABLE tag.",
|
||
),
|
||
|
||
# v0.7.3 MoE onboarding — Gemma 4 26B-A4B + Qwen 3.6 35B-A3B.
|
||
# Both target the unconstrained-nightly engine (vllm-nightly-clean) which
|
||
# rides nightly-bf610c2f (2026-05-15, post-PR-#42521). Gemma is the
|
||
# shippable path; Qwen 35B-A3B is preview-only until Genesis v7.73.x
|
||
# re-anchors on a post-#42521 nightly.
|
||
"vllm/gemma-a4b-single": _entry(
|
||
model="gemma-4-26b-a4b", weights_variant="autoround-int4-mixed", workload="fast-chat",
|
||
engine="vllm-nightly-clean", drafter=None, kv_format="bf16",
|
||
tp=1, max_ctx=8192, max_num_seqs=256, mem_util=0.92,
|
||
compose_path="models/gemma-4-26b-a4b/vllm/compose/single/autoround-int4-mixed/bf16.yml",
|
||
default_port=8040,
|
||
kvcalc_key="gemma-4-26b-a4b:gemma-a4b-single",
|
||
status="experimental",
|
||
status_note="v0.7.3 MoE onboarding — first-boot smoke, validation pending.",
|
||
),
|
||
"vllm/gemma-a4b": _entry(
|
||
model="gemma-4-26b-a4b", weights_variant="autoround-int4-mixed", workload="fast-chat",
|
||
engine="vllm-nightly-clean", drafter=None, kv_format="bf16",
|
||
tp=2, max_ctx=32768, max_num_seqs=256, mem_util=0.92,
|
||
compose_path="models/gemma-4-26b-a4b/vllm/compose/dual/autoround-int4-mixed/bf16.yml",
|
||
default_port=8041,
|
||
kvcalc_key="gemma-4-26b-a4b:gemma-a4b",
|
||
status="experimental",
|
||
status_note="v0.7.3 MoE onboarding — primary bench target, validation pending.",
|
||
),
|
||
"vllm/gemma-a4b-awq": _entry(
|
||
model="gemma-4-26b-a4b", weights_variant="awq", workload="fast-chat",
|
||
engine="vllm-nightly-clean", drafter=None, kv_format="bf16",
|
||
tp=2, max_ctx=32768, max_num_seqs=256, mem_util=0.92,
|
||
compose_path="models/gemma-4-26b-a4b/vllm/compose/dual/awq/bf16.yml",
|
||
default_port=8042,
|
||
kvcalc_key="gemma-4-26b-a4b:gemma-a4b-awq",
|
||
status="experimental",
|
||
status_note="v0.7.3 MoE onboarding — AWQ path with PR #40886 overlay, validation pending.",
|
||
),
|
||
"vllm/gemma-a4b-awq-mtp": _entry(
|
||
model="gemma-4-26b-a4b", weights_variant="awq", workload="fast-chat",
|
||
engine="vllm-nightly-clean", drafter="gemma-26b-it-assistant", kv_format="bf16",
|
||
tp=2, max_ctx=32768, max_num_seqs=256, mem_util=0.92,
|
||
compose_path="models/gemma-4-26b-a4b/vllm/compose/dual/awq/mtp.yml",
|
||
default_port=8043,
|
||
kvcalc_key="gemma-4-26b-a4b:gemma-a4b-awq-mtp",
|
||
status="experimental",
|
||
status_note="v0.7.3 MoE onboarding — AWQ + MTP, validation pending.",
|
||
),
|
||
"vllm/qwen-a3b-preview-single": _entry(
|
||
model="qwen3.6-35b-a3b", weights_variant="autoround-int4", workload="fast-chat",
|
||
engine="vllm-nightly-clean", drafter=None, kv_format="fp8_e5m2",
|
||
tp=1, max_ctx=8192, max_num_seqs=1, mem_util=0.92,
|
||
compose_path="models/qwen3.6-35b-a3b/vllm/compose/single/autoround-int4/preview.yml",
|
||
default_port=8050,
|
||
kvcalc_key="qwen3.6-35b-a3b:qwen-a3b-preview-single",
|
||
status="preview",
|
||
status_note="MoE onboarding smoke — Cliff 2 mitigations unavailable without Genesis. Do NOT use for long-ctx.",
|
||
),
|
||
"vllm/qwen-35b-a3b-dual": _entry(
|
||
model="qwen3.6-35b-a3b", weights_variant="autoround-int4", workload="fast-chat",
|
||
engine="vllm-nightly-clean", drafter=None, kv_format="fp8_e5m2",
|
||
tp=2, max_ctx=262144, max_num_seqs=1, mem_util=0.92,
|
||
compose_path="models/qwen3.6-35b-a3b/vllm/compose/dual/autoround-int4/fp8.yml",
|
||
default_port=8051,
|
||
kvcalc_key="qwen3.6-35b-a3b:qwen-35b-a3b-dual",
|
||
),
|
||
}
|
||
|
||
|
||
|
||
DEFAULTS = {
|
||
("qwen3.6-27b", "vllm", "single"): "vllm/minimal",
|
||
("qwen3.6-27b", "vllm", "dual"): "vllm/dual",
|
||
("qwen3.6-27b", "vllm", "multi4"): "vllm/dual4",
|
||
("qwen3.6-27b", "llamacpp", "single"): "llamacpp/default",
|
||
("qwen3.6-27b", "ik-llama", "single"): "ik-llama/iq4ks-mtp",
|
||
("qwen3.6-27b", "beellama", "single"): "beellama/dflash",
|
||
# No vLLM single-card Gemma default: fp8 KV is hardware-impossible on Ampere
|
||
# sm_86 (vllm/gemma-mtp-tp1 deprecated 2026-05-31) and no bf16 single compose
|
||
# ships. Single-card Gemma → beellama/gemma-dflash (the curated walk picks it).
|
||
("gemma-4-31b", "beellama", "single"): "beellama/gemma-dflash",
|
||
# Dual default is gemma-int8-mtp: full 262K + vision + 4 streams (the full-context
|
||
# priority). It rides v0.21.0 + the vendored #40391 per-head-KV overlay (the one
|
||
# gemma config that can't follow stable). gemma-bf16-mtp stays as the stable v0.22.0
|
||
# no-overlay 32K fallback — kept, not deprecated, just no longer the default.
|
||
("gemma-4-31b", "vllm", "dual"): "vllm/gemma-int8-mtp",
|
||
("gemma-4-26b-a4b", "vllm", "single"): "vllm/gemma-a4b-single",
|
||
("gemma-4-26b-a4b", "vllm", "dual"): "vllm/gemma-a4b",
|
||
("qwen3.6-35b-a3b", "vllm", "single"): "vllm/qwen-a3b-preview-single",
|
||
("qwen3.6-35b-a3b", "vllm", "dual"): "vllm/qwen-35b-a3b-dual",
|
||
}
|
||
|
||
|
||
# --- PR-B: model-default resolver knobs (maintainer-owned, design §13.3) ----
|
||
#
|
||
# Two curated tables drive the `<model>/default` resolver. They are maintainer
|
||
# knobs — edited by PR, never auto-grown. See docs/model-default-resolver
|
||
# design + the repo CLAUDE.md "Default rule" note.
|
||
|
||
# A SHORT opt-in shortlist of models eligible to be the *bare-launch* default
|
||
# (`launch.sh` with no model + no pin → first INSTALLED model on this list →
|
||
# its `<model>/default`). This is NOT an exhaustive ranking of the catalog:
|
||
# - Models absent from it are fully runnable by name (`--model X` /
|
||
# `X/default`); they are simply never the auto-default.
|
||
# - New models are NOT auto-added. Adding a model touches nothing here;
|
||
# promote one explicitly only when desired.
|
||
# - Order within the (short) list = the tiebreak for "first installed".
|
||
RECOMMENDED_DEFAULT_MODELS = ["qwen3.6-27b", "gemma-4-31b"]
|
||
|
||
# Which engine wins, per detected topology, when resolving `<model>/default`
|
||
# with no user pin. The resolver walks this list in order and picks the FIRST
|
||
# engine that has a functional DEFAULTS[(model, engine, topology)] entry (i.e.
|
||
# whose status is NOT in the (NA) set). This is the whole recommendation
|
||
# policy expressed as data — reorder a row to change a recommendation, no code
|
||
# change, any topology.
|
||
#
|
||
# Engine identifiers are the slug-prefix form used in DEFAULTS keys:
|
||
# vllm · ik-llama · llamacpp (+ beellama, aspirational — see below).
|
||
# `beellama` is ranked but has ZERO registry entries today (blocked on an
|
||
# upstream Docker image — see docs/UPSTREAM.md). The resolver skips it
|
||
# naturally (no DEFAULTS hit) → no behavior change until it is onboarded, at
|
||
# which point it AUTO-PROMOTES to the single-GPU default with no resolver edit.
|
||
ENGINE_PREFERENCE = {
|
||
"single": ["beellama", "ik-llama", "llamacpp", "vllm"],
|
||
"dual": ["vllm", "ik-llama", "llamacpp", "beellama"],
|
||
"multi": ["vllm", "ik-llama", "llamacpp", "beellama"],
|
||
}
|
||
|
||
|
||
def _topology_family(topology):
|
||
"""Map a concrete topology to its ENGINE_PREFERENCE family.
|
||
|
||
Concrete topologies are `single` · `dual` · `multi4` · `multiN`; the
|
||
preference table keys on the family `single` · `dual` · `multi`.
|
||
"""
|
||
if topology == "single":
|
||
return "single"
|
||
if topology == "dual":
|
||
return "dual"
|
||
if topology.startswith("multi"):
|
||
return "multi"
|
||
return topology
|
||
|
||
|
||
def _nearest_lower_topology(topology):
|
||
"""Degradation order (design §6): notice + nearest-lower topology.
|
||
|
||
multiN → dual → single → None. Returns the next topology to try, or None
|
||
when there is nowhere lower to fall.
|
||
"""
|
||
if topology.startswith("multi"):
|
||
return "dual"
|
||
if topology == "dual":
|
||
return "single"
|
||
return None
|
||
|
||
|
||
def engine_set():
|
||
"""The closed set of engine namespace-prefixes (DEFAULTS keys + ranked).
|
||
|
||
`X/default` dispatch (design §13.1): `X ∈ engine_set` → engine
|
||
recommendation; else `X ∈ model_set` → model default; else error.
|
||
Engines and model-ids are disjoint by construction.
|
||
"""
|
||
engines = set()
|
||
for _model, engine, _topology in DEFAULTS:
|
||
engines.add(engine)
|
||
for ranked in ENGINE_PREFERENCE.values():
|
||
engines.update(ranked)
|
||
return engines
|
||
|
||
|
||
def model_set():
|
||
"""The set of model-ids that appear in DEFAULTS (the runnable catalog)."""
|
||
return {model for (model, _engine, _topology) in DEFAULTS}
|
||
|
||
|
||
def _functional_default(model, engine, topology):
|
||
"""A DEFAULTS slug for (model, engine, topology) whose status is functional.
|
||
|
||
Returns the slug only when an entry exists AND its registry status is NOT
|
||
in the (NA) set (experimental/preview/upstream-gated/deprecated) — a
|
||
broken/preview config must never become someone's auto-default (§12.5).
|
||
Returns None otherwise.
|
||
"""
|
||
slug = DEFAULTS.get((model, engine, topology))
|
||
if not slug:
|
||
return None
|
||
entry = COMPOSE_REGISTRY.get(slug)
|
||
if entry is None:
|
||
return None
|
||
if entry.get("status", "production") not in FUNCTIONAL_STATUSES:
|
||
return None
|
||
return slug
|
||
|
||
|
||
def curated_default_target(model, topology):
|
||
"""Curated fallback (§4): walk ENGINE_PREFERENCE[family], first functional
|
||
DEFAULTS slug wins. Returns the slug, or None if no functional curated
|
||
default exists for (model, topology).
|
||
"""
|
||
family = _topology_family(topology)
|
||
for engine in ENGINE_PREFERENCE.get(family, []):
|
||
slug = _functional_default(model, engine, topology)
|
||
if slug:
|
||
return slug
|
||
return None
|
||
|
||
|
||
def community_default_target(model, topology, hw_class=None): # noqa: ARG001
|
||
"""Community-ranked best config — the FUTURE middle precedence rung (§13.4).
|
||
|
||
Contract: returns a ranked slug when the submissions/ranking app exists;
|
||
returns None today (always skipped). The resolver inserts a non-None result
|
||
BETWEEN the user pin and the curated fallback. v1 ships this stub returning
|
||
None so the ladder rung is real, not aspirational; a test asserts it is
|
||
skipped.
|
||
"""
|
||
return None
|
||
|
||
|
||
def model_default_pin_key(model):
|
||
"""The .env key for a per-model user pin (design §13.2).
|
||
|
||
`CLUB3090_DEFAULT_<MODELID uppercased, non-alnum→_>`, e.g.
|
||
qwen3.6-27b → CLUB3090_DEFAULT_QWEN3_6_27B.
|
||
"""
|
||
suffix = "".join(c if c.isalnum() else "_" for c in model).upper()
|
||
return f"CLUB3090_DEFAULT_{suffix}"
|
||
|
||
|
||
def model_of_slug(slug):
|
||
"""The model-id a slug belongs to, or None if the slug is unknown."""
|
||
entry = COMPOSE_REGISTRY.get(slug)
|
||
return entry.get("model") if entry else None
|
||
|
||
|
||
def slug_topology(slug):
|
||
"""The topology family a slug serves, derived from its compose_path.
|
||
|
||
compose_path is `models/<model>/<engine>/compose/<topology>/<quant>/...`.
|
||
Returns `single`/`dual`/`multi` (the ENGINE_PREFERENCE family) or None.
|
||
"""
|
||
entry = COMPOSE_REGISTRY.get(slug)
|
||
if not entry:
|
||
return None
|
||
cp = entry.get("compose_path", "")
|
||
if "/compose/" not in cp:
|
||
return None
|
||
after = cp.split("/compose/", 1)[1]
|
||
topo = after.split("/", 1)[0]
|
||
return _topology_family(topo)
|