@lolren reported 25 TPS on dual-dflash.yml vs 80+ on dual.yml — root cause was that scripts/setup.sh doesn't download the DFlash draft model (z-lab/Qwen3.6-27B-DFlash), and the compose path /root/.cache/huggingface/qwen3.6-27b-dflash silently fell back to baseline bf16 decode when missing. There was no docs page that told users they needed to grab it separately, and UPSTREAM.md's "watch list" framing for the z-lab draft was muddled with the shipping vLLM compose. Three closes: 1. scripts/setup.sh — new WITH_DFLASH_DRAFT=1 env var. When set, fetches z-lab/Qwen3.6-27B-DFlash to <MODEL_DIR>/qwen3.6-27b-dflash/ after the main model. Bumps disk preflight from 25 to 28 GB. Documents that the draft is still under training (per UPSTREAM.md re-test trigger). 2. docker-compose.dual-dflash.yml + dual-dflash-noviz.yml headers — added explicit "Prerequisite" block with both `WITH_DFLASH_DRAFT=1` and manual `hf download` instructions, plus the under-training caveat. 3. docs/UPSTREAM.md + docs/DUAL_CARD.md — reconciled the inconsistency. UPSTREAM.md now explicitly distinguishes "Luce-Org/lucebox-hub (single-card llama.cpp fork — not shipping)" from "vLLM dual-dflash compose (shipping with same draft, different engine)". DUAL_CARD.md's peak code TPS section now flags the prereq + recommends dual.yml (FP8 + MTP) for autonomous coding agents until z-lab tags training-complete. Co-Authored-By: Claude Opus 4.7 (1M context) <[email protected]>
301 lines
13 KiB
Bash
Executable File
301 lines
13 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
#
|
|
# Model-aware one-shot setup for club-3090.
|
|
#
|
|
# bash scripts/setup.sh <model-name>
|
|
#
|
|
# Currently supported:
|
|
# qwen3.6-27b → Lorbus/Qwen3.6-27B-int4-AutoRound + Genesis patches
|
|
#
|
|
# What it does (per supported model):
|
|
# - clones Sandermage/genesis-vllm-patches into models/<model>/vllm/patches/genesis
|
|
# (vLLM-only; skip with SKIP_GENESIS=1 if you only need llama.cpp / SGLang)
|
|
# - downloads model weights into $MODEL_DIR with SHA256 verification
|
|
# against HF x-linked-etag
|
|
#
|
|
# Env vars (optional):
|
|
# MODEL_DIR Where to place model weights. Default: <repo>/models-cache
|
|
# HF_TOKEN HF token (public models, usually unnecessary)
|
|
# SKIP_MODEL Set to 1 to skip the model download step
|
|
# SKIP_GENESIS Set to 1 to skip cloning Genesis patches
|
|
# WITH_DFLASH_DRAFT Set to 1 to ALSO download z-lab/Qwen3.6-27B-DFlash
|
|
# (~1.75 GB; required ONLY for dual-dflash.yml /
|
|
# dual-dflash-noviz.yml composes). Default: 0.
|
|
# Note: draft model is still under training as of
|
|
# 2026-04-26; bench numbers in DUAL_CARD.md were
|
|
# measured against that snapshot. AL improvements
|
|
# expected when z-lab tags training-complete.
|
|
# PREFLIGHT_DISK_GB Required free space at MODEL_DIR (default: 25, or
|
|
# 28 if WITH_DFLASH_DRAFT=1)
|
|
#
|
|
# Idempotent: safe to re-run — skips steps already done.
|
|
|
|
set -euo pipefail
|
|
|
|
# ---------- Model dispatch ----------
|
|
MODEL_NAME="${1:-}"
|
|
if [[ -z "${MODEL_NAME}" ]]; then
|
|
echo "Usage: $0 <model-name>"
|
|
echo ""
|
|
echo "Supported model names:"
|
|
echo " qwen3.6-27b"
|
|
exit 1
|
|
fi
|
|
|
|
case "${MODEL_NAME}" in
|
|
qwen3.6-27b)
|
|
MODEL_REPO="Lorbus/Qwen3.6-27B-int4-AutoRound"
|
|
MODEL_SUBDIR="qwen3.6-27b-autoround-int4"
|
|
NEEDS_GENESIS=1
|
|
;;
|
|
*)
|
|
echo "ERROR: unsupported model '${MODEL_NAME}'."
|
|
echo "Supported: qwen3.6-27b"
|
|
echo "(To add a new model, extend the case dispatch in scripts/setup.sh)"
|
|
exit 1
|
|
;;
|
|
esac
|
|
|
|
ROOT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)"
|
|
MODEL_DIR="${MODEL_DIR:-${ROOT_DIR}/models-cache}"
|
|
GENESIS_DIR="${ROOT_DIR}/models/${MODEL_NAME}/vllm/patches/genesis"
|
|
|
|
cd "${ROOT_DIR}"
|
|
|
|
# ---------- Pre-flight checks ----------
|
|
# Catches the common "first-run failures": missing docker, no GPU visible,
|
|
# disk too small for the ~14 GB AutoRound int4 download. Fails fast with
|
|
# actionable hints rather than mid-download or first-boot crash.
|
|
# shellcheck source=preflight.sh
|
|
source "${ROOT_DIR}/scripts/preflight.sh"
|
|
|
|
# Required disk: model is ~14 GB on disk; 25 GB gives buffer for download
|
|
# temp files + safetensors + tokenizer/config. Add ~3 GB if also pulling
|
|
# the DFlash draft (~1.75 GB packed + buffer for download tempfiles).
|
|
if [[ "${WITH_DFLASH_DRAFT:-0}" == "1" ]]; then
|
|
PREFLIGHT_DISK_GB="${PREFLIGHT_DISK_GB:-28}"
|
|
else
|
|
PREFLIGHT_DISK_GB="${PREFLIGHT_DISK_GB:-25}"
|
|
fi
|
|
|
|
echo "[preflight] checking environment..."
|
|
preflight_docker || exit 1
|
|
preflight_gpu 1 || exit 1
|
|
preflight_disk "${MODEL_DIR}" "${PREFLIGHT_DISK_GB}" || exit 1
|
|
echo "[preflight] ok."
|
|
echo ""
|
|
|
|
# ---------- Tool checks ----------
|
|
need() {
|
|
command -v "$1" >/dev/null 2>&1 || {
|
|
echo "ERROR: required tool '$1' not found in PATH." >&2
|
|
exit 1
|
|
}
|
|
}
|
|
need git
|
|
need curl
|
|
need sha256sum
|
|
|
|
echo "Setup root: ${ROOT_DIR}"
|
|
echo "Model dir: ${MODEL_DIR}"
|
|
|
|
# ---------- Genesis patches ----------
|
|
# We track Sandermage's tree at HEAD and rely on tagged commits / SHA pinning
|
|
# in the compose files for reproducibility. The repo layout changed substantially
|
|
# between v7.13 (monolithic patch_genesis_unified.py shim) and v7.14 (modular
|
|
# vllm/_genesis package + per-patch env opts). Newer composes mount the package;
|
|
# the legacy compose still references the v7.13 shim.
|
|
# Pin Genesis to the exact commit our published numbers were measured against.
|
|
# Currently pointing at v7.66 dev tip (commit fc89395, 2026-05-02 AM). Bumped
|
|
# from v7.64 (64dd18b) for the v7.65 patch set:
|
|
# - P38B / P15B — close the Cliff 1 mech B cascade (issues #14 + #15) via
|
|
# compile-safe in-source hook + FA varlen workspace clamp.
|
|
# - PN25 — Inductor-safe silu_and_mul opaque op (replaces our local
|
|
# patch_pn12_compile_safe_custom_op.py — now removed).
|
|
# - PN26b — Genesis-original sparse-V Triton kernel for SM86 (Ampere
|
|
# consumer). First sparse-V kernel in any public tree for SM86. Default
|
|
# ON in v0.20+ composes (BLOCK_KV=8 num_warps=4 threshold=0.01 per
|
|
# Sandermage's 27B-specific tuning).
|
|
# - PN28 — merge_attn_states NaN guard backport (vllm#39148).
|
|
# - Cliff 8 hardening (partial_apply_warnings counter in boot summary).
|
|
# Pinned to dev SHA fc89395 because v7.66 is feature-complete on dev but not
|
|
# yet tagged; SHA pin is immutable.
|
|
# Bumping GENESIS_PIN requires re-running verify-full.sh against your composes
|
|
# to confirm the new commit works on your config.
|
|
GENESIS_PIN="${GENESIS_PIN:-fc89395}"
|
|
|
|
if [[ "${SKIP_GENESIS:-0}" != "1" ]]; then
|
|
if [[ -d "${GENESIS_DIR}/.git" ]]; then
|
|
echo "[genesis] Already cloned at ${GENESIS_DIR} — fetching + checking out ${GENESIS_PIN} ..."
|
|
(cd "${GENESIS_DIR}" && git fetch origin && git checkout "${GENESIS_PIN}" 2>&1 | tail -3)
|
|
else
|
|
echo "[genesis] Cloning Sandermage/genesis-vllm-patches at ${GENESIS_PIN} ..."
|
|
# Full clone (commit SHAs aren't reachable via --branch + --depth 1).
|
|
git clone https://github.com/Sandermage/genesis-vllm-patches.git "${GENESIS_DIR}"
|
|
(cd "${GENESIS_DIR}" && git checkout "${GENESIS_PIN}")
|
|
fi
|
|
|
|
# v7.14+ layout sanity check
|
|
if [[ ! -d "${GENESIS_DIR}/vllm/_genesis" ]]; then
|
|
echo "ERROR: genesis tree at ${GENESIS_PIN} missing vllm/_genesis package." >&2
|
|
echo " Re-run with GENESIS_PIN=<other-ref> to try a different version." >&2
|
|
exit 1
|
|
fi
|
|
echo "[genesis] Pinned to ${GENESIS_PIN} ($(cd "${GENESIS_DIR}" && git rev-parse --short HEAD))"
|
|
|
|
# PN25 worker-spawn registration fix — local backport.
|
|
#
|
|
# Sandermage shipped his own fix in d92bcb3 (hasattr global-registry guard
|
|
# in `_register_op_once`), but cross-rig validation on our TP=1 single-card
|
|
# showed it doesn't work — `torch.ops.genesis.silu_and_mul_pooled` doesn't
|
|
# exist in spawned workers on TP=1 (whereas it does on his TP=2 PROD).
|
|
# Reported back as comment on Sandermage/genesis-vllm-patches#16.
|
|
#
|
|
# Our local v3 patch takes a different approach: register at activation.py
|
|
# import time as a module-level cached global, BEFORE any dynamo trace runs.
|
|
# Survives worker spawn correctly on TP=1.
|
|
#
|
|
# Idempotent. Safe to re-run.
|
|
if [[ -f "${ROOT_DIR}/models/qwen3.6-27b/vllm/patches/patch_pn25_genesis_register_fix.py" ]]; then
|
|
(cd "${ROOT_DIR}" && python3 models/qwen3.6-27b/vllm/patches/patch_pn25_genesis_register_fix.py) || {
|
|
echo "[genesis] WARN: PN25 register fix did not apply cleanly. PN25 may not work in workers." >&2
|
|
}
|
|
fi
|
|
|
|
# PN30 DS conv-state layout fix — local correction for Genesis issue #17.
|
|
#
|
|
# Sander's PN30 avoided vLLM's DS+spec-decode NotImplementedError by
|
|
# compacting state[src_block, :, offset:] and raw-memcpying it into the
|
|
# destination block. That corrupts DS row strides. Our sidecar patches PN30
|
|
# so collect_mamba_copy_meta builds a full destination-shaped temp block,
|
|
# copies the source tail into the dst prefix, then reuses PN30's temp-list
|
|
# lifetime handling.
|
|
if [[ -f "${ROOT_DIR}/models/qwen3.6-27b/vllm/patches/patch_pn30_dst_shaped_temp_fix.py" ]]; then
|
|
(cd "${ROOT_DIR}" && python3 models/qwen3.6-27b/vllm/patches/patch_pn30_dst_shaped_temp_fix.py) || {
|
|
echo "[genesis] WARN: PN30 dst-shaped temp fix did not apply cleanly. Keep PN30 disabled or use SD layout." >&2
|
|
}
|
|
fi
|
|
else
|
|
echo "[genesis] SKIP_GENESIS=1 — not cloning."
|
|
fi
|
|
|
|
# ---------- Model download ----------
|
|
if [[ "${SKIP_MODEL:-0}" == "1" ]]; then
|
|
echo "[model] SKIP_MODEL=1 — not downloading."
|
|
exit 0
|
|
fi
|
|
|
|
mkdir -p "${MODEL_DIR}/${MODEL_SUBDIR}"
|
|
|
|
# Prefer `hf` CLI if available (faster with hf_transfer); fall back to curl.
|
|
download_via_hf() {
|
|
echo "[model] Using 'hf download' (hf_transfer if available) ..."
|
|
HF_HUB_ENABLE_HF_TRANSFER=1 HF_HUB_DISABLE_XET=1 \
|
|
hf download "${MODEL_REPO}" --local-dir "${MODEL_DIR}/${MODEL_SUBDIR}"
|
|
}
|
|
|
|
if command -v hf >/dev/null 2>&1; then
|
|
download_via_hf
|
|
elif command -v huggingface-cli >/dev/null 2>&1; then
|
|
echo "[model] Using 'huggingface-cli download' ..."
|
|
HF_HUB_ENABLE_HF_TRANSFER=1 HF_HUB_DISABLE_XET=1 \
|
|
huggingface-cli download "${MODEL_REPO}" --local-dir "${MODEL_DIR}/${MODEL_SUBDIR}"
|
|
else
|
|
echo "ERROR: neither 'hf' nor 'huggingface-cli' found. Install with:" >&2
|
|
echo " pip install 'huggingface-hub[hf_transfer]'" >&2
|
|
echo "or:" >&2
|
|
echo " uv tool install --with hf_transfer huggingface-hub" >&2
|
|
exit 1
|
|
fi
|
|
|
|
# ---------- SHA verification ----------
|
|
echo "[verify] Checking SHA256 of every *.safetensors against HF x-linked-etag ..."
|
|
cd "${MODEL_DIR}/${MODEL_SUBDIR}"
|
|
|
|
fail=0
|
|
count=0
|
|
for f in *.safetensors; do
|
|
[[ -f "$f" ]] || continue
|
|
count=$((count + 1))
|
|
expected="$(curl -sfI "https://huggingface.co/${MODEL_REPO}/resolve/main/$f" \
|
|
| grep -i '^x-linked-etag:' | tr -d '"\r' | awk '{print $NF}' || true)"
|
|
actual="$(sha256sum "$f" | awk '{print $1}')"
|
|
if [[ -z "$expected" ]]; then
|
|
printf " %-50s SKIP (no etag)\n" "$f"
|
|
elif [[ "$expected" == "$actual" ]]; then
|
|
printf " %-50s OK\n" "$f"
|
|
else
|
|
printf " %-50s FAIL exp=%.12s act=%.12s\n" "$f" "$expected" "$actual"
|
|
fail=$((fail + 1))
|
|
fi
|
|
done
|
|
cd "${ROOT_DIR}"
|
|
|
|
if [[ "$fail" != "0" ]]; then
|
|
echo "[verify] ${fail} shard(s) failed SHA check." >&2
|
|
echo " Delete ${MODEL_DIR}/${MODEL_SUBDIR} and re-run setup.sh." >&2
|
|
exit 1
|
|
fi
|
|
|
|
if [[ "$count" == "0" ]]; then
|
|
echo "[verify] No .safetensors found in ${MODEL_DIR}/${MODEL_SUBDIR} — download may have failed." >&2
|
|
exit 1
|
|
fi
|
|
|
|
echo ""
|
|
echo "[done] ${count} shards SHA-verified."
|
|
[[ -d "${GENESIS_DIR}/.git" ]] && echo " Genesis pinned at ${GENESIS_PIN} ($(cd "${GENESIS_DIR}" && git rev-parse --short HEAD))."
|
|
echo ""
|
|
|
|
# ---------- Optional DFlash draft model ----------
|
|
# Required ONLY for `docker-compose.dual-dflash.yml` / `dual-dflash-noviz.yml`.
|
|
# vLLM `method:"dflash"` spec-decode loads this as the draft. The compose
|
|
# expects it at <MODEL_DIR>/qwen3.6-27b-dflash/ (~1.75 GB / card after load).
|
|
#
|
|
# Caveat: as of 2026-04-26, z-lab/Qwen3.6-27B-DFlash is still under training.
|
|
# Published bench in docs/DUAL_CARD.md (82 narr / 125 code TPS on dual-3090)
|
|
# was measured against the 2026-04-26 snapshot at peak code-prompt conditions.
|
|
# Real agent traffic (mixed code + narrative + tool schemas) will see lower
|
|
# AL until z-lab tags training-complete. See docs/UPSTREAM.md for the watch
|
|
# entry and re-test trigger.
|
|
DFLASH_REPO="z-lab/Qwen3.6-27B-DFlash"
|
|
DFLASH_SUBDIR="qwen3.6-27b-dflash"
|
|
if [[ "${WITH_DFLASH_DRAFT:-0}" == "1" ]] && [[ "${SKIP_MODEL:-0}" != "1" ]]; then
|
|
echo "[dflash] WITH_DFLASH_DRAFT=1 — downloading ${DFLASH_REPO} ..."
|
|
mkdir -p "${MODEL_DIR}/${DFLASH_SUBDIR}"
|
|
if command -v hf >/dev/null 2>&1; then
|
|
HF_HUB_ENABLE_HF_TRANSFER=1 HF_HUB_DISABLE_XET=1 \
|
|
hf download "${DFLASH_REPO}" --local-dir "${MODEL_DIR}/${DFLASH_SUBDIR}"
|
|
elif command -v huggingface-cli >/dev/null 2>&1; then
|
|
HF_HUB_ENABLE_HF_TRANSFER=1 HF_HUB_DISABLE_XET=1 \
|
|
huggingface-cli download "${DFLASH_REPO}" --local-dir "${MODEL_DIR}/${DFLASH_SUBDIR}"
|
|
else
|
|
echo "[dflash] ERROR: neither 'hf' nor 'huggingface-cli' available — cannot download DFlash draft." >&2
|
|
exit 1
|
|
fi
|
|
echo "[dflash] Downloaded ${DFLASH_REPO} to ${MODEL_DIR}/${DFLASH_SUBDIR}"
|
|
elif [[ -d "${MODEL_DIR}/${DFLASH_SUBDIR}" ]]; then
|
|
echo "[dflash] ${MODEL_DIR}/${DFLASH_SUBDIR} already exists — using existing draft."
|
|
else
|
|
echo "[dflash] Skipping DFlash draft model. Set WITH_DFLASH_DRAFT=1 to fetch"
|
|
echo " ${DFLASH_REPO} (~1.75 GB; required only for dual-dflash composes)."
|
|
fi
|
|
echo ""
|
|
|
|
echo "Next — single-card vLLM (default):"
|
|
echo " cd models/${MODEL_NAME}/vllm/compose && docker compose up -d"
|
|
echo " docker logs -f vllm-qwen36-27b"
|
|
echo ""
|
|
echo "For dual-card composes, you ALSO need the Marlin pad fork mounted at"
|
|
echo "/opt/ai/vllm-src/ (vLLM PR #40361 — open upstream, drops out when it lands):"
|
|
echo " sudo mkdir -p /opt/ai && sudo chown \$USER /opt/ai"
|
|
echo " git clone -b marlin-pad-sub-tile-n https://github.com/noonghunna/vllm.git /opt/ai/vllm-src"
|
|
echo ""
|
|
echo "Then:"
|
|
echo " cd models/${MODEL_NAME}/vllm/compose && docker compose -f docker-compose.dual.yml up -d"
|
|
echo ""
|
|
echo "Sanity test (after 'Application startup complete'):"
|
|
echo " curl -sf http://localhost:8020/v1/chat/completions \\"
|
|
echo " -H 'Content-Type: application/json' \\"
|
|
echo " -d '{\"model\":\"qwen3.6-27b-autoround\",\"messages\":[{\"role\":\"user\",\"content\":\"Capital of France?\"}],\"max_tokens\":200}'"
|