Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d116ba9ba3 | ||
|
|
6775d10919 |
@@ -16,6 +16,27 @@ history; SemVer takes over from `v0.3.0` onward.
|
||||
|
||||
---
|
||||
|
||||
## v0.7.1 — 2026-05-15
|
||||
|
||||
|
||||
### ✨ Features
|
||||
|
||||
- feat(bench): surface prompt processing throughput ([2a148d7](https://github.com/noonghunna/club-3090/commit/2a148d702b9415129d4c4ec9d3e7d30765927aa4))
|
||||
- feat(llamacpp): expose batch tuning knobs ([02249ab](https://github.com/noonghunna/club-3090/commit/02249ab1939f354ac062d343efefe32677203174))
|
||||
|
||||
|
||||
### 🐛 Bug fixes
|
||||
|
||||
- fix(ci): simplify vllm image workflow, drop smoke-gate (#135) ([ce2617e](https://github.com/noonghunna/club-3090/commit/ce2617e0bc0f56d42caf64e96847d966380be80c))
|
||||
|
||||
|
||||
### 📝 Documentation
|
||||
|
||||
- docs(upstream): PR #42102 closed-as-slop; local overlay permanent ([57eb269](https://github.com/noonghunna/club-3090/commit/57eb269cd70935fc3069b85e46ead8f0f0af13dc))
|
||||
|
||||
|
||||
|
||||
[Pin: `git checkout v0.7.1`] · [Full diff](https://github.com/noonghunna/club-3090/compare/v0.7.0...v0.7.1)
|
||||
## v0.7.0 — 2026-05-14
|
||||
|
||||
|
||||
|
||||
@@ -42,6 +42,57 @@ isn't your topology.
|
||||
|
||||
---
|
||||
|
||||
## Topology classification
|
||||
|
||||
The launcher classifies your selected hardware and emits strategy guidance
|
||||
when the cards are not matched. You can run the classifier without booting
|
||||
anything:
|
||||
|
||||
```bash
|
||||
bash scripts/launch.sh --topology
|
||||
```
|
||||
|
||||
Use `--gpus 0,1` or `--cards 2` with `--topology` if you only want advice
|
||||
for a subset.
|
||||
|
||||
| Class | What it means | Example | Recommended |
|
||||
|---|---|---|---|
|
||||
| `single_card` | 1 GPU detected | 1x RTX 3090 | Use the largest single-card compose that fits (`vllm/default`, `vllm/long-text`, `llamacpp/default`). |
|
||||
| `homogeneous` | All cards have matched VRAM and matched SM | 2x RTX 3090 | TP=N is the optimal default; use the shipped `vllm/dual*` or `vllm/dual4*` composes. |
|
||||
| `vram_matched_compute_mismatched` | Same VRAM, different compute tier | RTX 3090 + RTX 4090 | TP=N works correctly, but faster cards wait at NCCL allreduce. Estate planner is better for multi-model workloads. |
|
||||
| `vram_mismatched` | Different VRAM sizes | RTX 3060 12 GB + RTX 3090 24 GB | Prefer llama.cpp `--tensor-split`, manual PP=N experiments, or estate planner. Avoid TP=N across the full mismatched set. |
|
||||
| `heterogeneous_mixed` | Multiple VRAM and compute tiers | RTX 3060 + RTX 3090 + RTX 4090 | Manual selection. Run one model on the largest matched subset or use estate planner for separate endpoints. |
|
||||
|
||||
### Why TP=N is poor on VRAM-mismatched cards
|
||||
|
||||
Tensor parallelism splits weights evenly across cards. If one card has 24 GB
|
||||
and another has 12 GB, TP=2 still puts roughly half the model on each card.
|
||||
The smaller card becomes the hard ceiling for weights, KV cache, activations,
|
||||
and fragmentation. For Qwen 3.6 27B INT4, that usually leaves too little KV
|
||||
headroom to be useful.
|
||||
|
||||
For mismatched VRAM, the practical paths are:
|
||||
|
||||
- llama.cpp `--tensor-split` for weighted layer placement.
|
||||
- PP=N as a manual vLLM flag flip (`--pipeline-parallel-size N`) when you are
|
||||
deliberately experimenting. club-3090 does not ship a PP compose today.
|
||||
- Estate planner: `bash scripts/launch.sh --estate` runs different models on
|
||||
different card subsets without forcing one model across uneven VRAM.
|
||||
|
||||
### When compute-mismatched TP is fine
|
||||
|
||||
Matched VRAM with different SM, such as RTX 3090 + RTX 4090, is a different
|
||||
trade-off. TP=2 works because both cards have enough memory for the same model
|
||||
shard and KV budget. The cost is throughput: the faster card waits at NCCL
|
||||
allreduce barriers, so effective pair speed caps near the slower card. You
|
||||
preserve per-card VRAM capacity, but waste some compute on the faster card.
|
||||
|
||||
That is acceptable for one-model serving. If your goal is maximum aggregate
|
||||
throughput from two different cards, estate planner usually wins because each
|
||||
card runs its own model at full speed.
|
||||
|
||||
---
|
||||
|
||||
## Valid TP values for Qwen3.6-27B
|
||||
|
||||
vLLM's tensor parallelism splits attention heads across cards. The TP
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
# bash scripts/launch.sh --estate-file <path> # boot an existing estate plan
|
||||
# bash scripts/launch.sh --validate-estate <path> # validate estate.yml, no boot
|
||||
# bash scripts/launch.sh --down-estate <path> # stop estate instances
|
||||
# bash scripts/launch.sh --topology # print GPU topology advisory, no boot
|
||||
# bash scripts/launch.sh --model qwen3.6-27b --gpus 0,1
|
||||
# bash scripts/launch.sh --engine vllm --cards 1 # deprecated; prefer --gpus
|
||||
# bash scripts/launch.sh --workload long-ctx-single # profile-aware filter
|
||||
@@ -66,6 +67,7 @@ ESTATE_FILE=""
|
||||
VALIDATE_ESTATE=""
|
||||
DOWN_ESTATE=""
|
||||
ONLY_NAMES=""
|
||||
TOPOLOGY_ONLY=0
|
||||
CARDS=""
|
||||
VARIANT=""
|
||||
MODEL_NAME=""
|
||||
@@ -87,6 +89,7 @@ while [[ $# -gt 0 ]]; do
|
||||
--validate-estate) VALIDATE_ESTATE="$2"; shift 2 ;;
|
||||
--down-estate) DOWN_ESTATE="$2"; shift 2 ;;
|
||||
--only) ONLY_NAMES="$2"; shift 2 ;;
|
||||
--topology) TOPOLOGY_ONLY=1; SKIP_PREFLIGHT=1; shift ;;
|
||||
--engine) ENGINE="$2"; shift 2 ;;
|
||||
--workload) WORKLOAD_ID="$2"; shift 2 ;;
|
||||
--drafter) DRAFTER_ID="$2"; shift 2 ;;
|
||||
@@ -602,6 +605,77 @@ selected_gpu_profile_spec() {
|
||||
printf '%s' "$joined"
|
||||
}
|
||||
|
||||
select_topology_gpus() {
|
||||
GPU_LINES="$(compose_hw_detect_gpus 2>/dev/null || true)"
|
||||
[[ -n "$GPU_LINES" ]] || return 1
|
||||
CARD_INDICES=()
|
||||
CARD_NAMES=()
|
||||
CARD_MEM_MIB=()
|
||||
CARD_SM=()
|
||||
|
||||
if [[ -n "$GPU_ARG" && "$GPU_ARG" != "all" ]]; then
|
||||
IFS=',' read -ra _launch_topology_tokens <<< "$GPU_ARG"
|
||||
local idx
|
||||
for idx in "${_launch_topology_tokens[@]}"; do
|
||||
idx="$(_compose_meta_trim "$idx")"
|
||||
[[ -z "$idx" ]] && continue
|
||||
gpu_exists "$idx" || { echo "[launch] ERROR: requested GPU ${idx}, but it was not detected." >&2; exit 1; }
|
||||
append_selected_gpu "$idx"
|
||||
done
|
||||
elif [[ -n "$CARDS" ]]; then
|
||||
[[ "$CARDS" =~ ^[0-9]+$ && "$CARDS" -ge 1 ]] || { echo "[launch] ERROR: --cards expects a positive integer." >&2; exit 1; }
|
||||
local idx name mem_mib sm selected=0
|
||||
while IFS=$'\t' read -r idx name mem_mib sm; do
|
||||
[[ -z "$idx" ]] && continue
|
||||
append_selected_gpu "$idx"
|
||||
selected=$((selected + 1))
|
||||
(( selected >= CARDS )) && break
|
||||
done <<< "$GPU_LINES"
|
||||
(( selected == CARDS )) || { echo "[launch] ERROR: --cards ${CARDS} requested, but only ${selected} GPU(s) were detected." >&2; exit 1; }
|
||||
else
|
||||
local idx name mem_mib sm
|
||||
while IFS=$'\t' read -r idx name mem_mib sm; do
|
||||
[[ -z "$idx" ]] && continue
|
||||
append_selected_gpu "$idx"
|
||||
done <<< "$GPU_LINES"
|
||||
fi
|
||||
|
||||
[[ "${#CARD_INDICES[@]}" -gt 0 ]] || return 1
|
||||
SELECTED_GPU_CSV="$(IFS=','; echo "${CARD_INDICES[*]}")"
|
||||
summarize_selected_vram >/dev/null
|
||||
return 0
|
||||
}
|
||||
|
||||
print_topology_advisory() {
|
||||
local output
|
||||
output="$(python3 "$LAUNCH_PROFILE" topology --gpu-spec "$(selected_gpu_profile_spec)" --format wizard 2>&1)" || {
|
||||
echo "$output" >&2
|
||||
exit 2
|
||||
}
|
||||
if [[ -n "$output" ]]; then
|
||||
echo "$output" >&2
|
||||
fi
|
||||
}
|
||||
|
||||
print_topology_and_exit() {
|
||||
local output
|
||||
if ! select_topology_gpus; then
|
||||
echo "Detected hardware:"
|
||||
echo " no NVIDIA GPUs detected"
|
||||
echo ""
|
||||
echo "Topology class: unavailable"
|
||||
echo ""
|
||||
echo "For details, see docs/MULTI_CARD.md."
|
||||
exit 0
|
||||
fi
|
||||
output="$(python3 "$LAUNCH_PROFILE" topology --gpu-spec "$(selected_gpu_profile_spec)" --format standalone 2>&1)" || {
|
||||
echo "$output" >&2
|
||||
exit 0
|
||||
}
|
||||
echo "$output"
|
||||
exit 0
|
||||
}
|
||||
|
||||
launch_nvlink_active() {
|
||||
if [[ "${#CARD_INDICES[@]}" -ne 2 ]]; then
|
||||
printf '0'
|
||||
@@ -932,6 +1006,10 @@ if [[ "$ESTATE_MODE" -eq 1 || -n "$ESTATE_FILE" ]]; then
|
||||
exit $?
|
||||
fi
|
||||
|
||||
if [[ "$TOPOLOGY_ONLY" -eq 1 ]]; then
|
||||
print_topology_and_exit
|
||||
fi
|
||||
|
||||
# --- wizard ---
|
||||
if [[ -z "$VARIANT" ]]; then
|
||||
echo "" >&2
|
||||
@@ -939,6 +1017,7 @@ if [[ -z "$VARIANT" ]]; then
|
||||
echo "(Use --variant <name> next time to skip the wizard.)" >&2
|
||||
choose_model
|
||||
choose_gpus
|
||||
print_topology_advisory
|
||||
pick_parallelism
|
||||
if [[ "$MODEL_NAME" == "gemma-4-31b" && "${#CARD_INDICES[@]}" -eq 1 && "$MIN_VRAM_GB" -lt 32 ]]; then
|
||||
gemma_single_24gb_guidance
|
||||
|
||||
@@ -12,6 +12,7 @@ import os
|
||||
import subprocess
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
|
||||
@@ -26,7 +27,7 @@ from .compose_registry import COMPOSE_REGISTRY
|
||||
SUPPORTED_SCHEMA_VERSIONS = {1}
|
||||
PROFILE_ROOT = Path(__file__).resolve().parent
|
||||
REPO_ROOT = Path(__file__).resolve().parents[3]
|
||||
CONSTRAINT_IDS = [f"C{i}" for i in range(1, 16)]
|
||||
CONSTRAINT_IDS = [f"C{i}" for i in range(1, 17)]
|
||||
ESTATE_CONSTRAINT_IDS = [f"E{i}" for i in range(1, 5)]
|
||||
|
||||
|
||||
@@ -42,6 +43,37 @@ class CrossReferenceError(ProfileError):
|
||||
"""Raised when a profile references a missing profile id."""
|
||||
|
||||
|
||||
class TopologyClass(str, Enum):
|
||||
SINGLE_CARD = "single_card"
|
||||
HOMOGENEOUS = "homogeneous"
|
||||
VRAM_MATCHED_COMPUTE_MISMATCHED = "vram_matched_compute_mismatched"
|
||||
VRAM_MISMATCHED = "vram_mismatched"
|
||||
HETEROGENEOUS_MIXED = "heterogeneous_mixed"
|
||||
|
||||
|
||||
TOPOLOGY_ADVISORY = {
|
||||
TopologyClass.SINGLE_CARD: None,
|
||||
TopologyClass.HOMOGENEOUS: None,
|
||||
TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED: (
|
||||
"Compute mismatch detected (VRAM matched). TP=N works fine but the faster card "
|
||||
"waits at every NCCL allreduce — effective throughput caps at slower card's speed "
|
||||
"(~30% of faster card idle at allreduce). Full per-card VRAM capacity preserved. "
|
||||
"Alternative: estate planner (--estate) to run different models per card at full speed."
|
||||
),
|
||||
TopologyClass.VRAM_MISMATCHED: (
|
||||
"VRAM mismatch detected. TP=N would cap to smaller card's usable model size. "
|
||||
"Recommended paths: (a) llama.cpp `--tensor-split` for weighted layer split, "
|
||||
"(b) PP=N (manual flag flip — `--pipeline-parallel-size N` on a vllm/dual compose; "
|
||||
"no shipping PP compose), (c) estate planner (--estate) to run different models per card."
|
||||
),
|
||||
TopologyClass.HETEROGENEOUS_MIXED: (
|
||||
"Heterogeneous hardware detected (multiple VRAM and compute tiers). Manual selection "
|
||||
"recommended. Consider the estate planner (--estate) to put different models on "
|
||||
"different card subsets, or run a single model on the largest matched subset."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def _logger() -> logging.Logger:
|
||||
logger = logging.getLogger("compat")
|
||||
if not logger.handlers:
|
||||
@@ -92,6 +124,30 @@ class HardwareProfile:
|
||||
notes: Optional[str] = None
|
||||
|
||||
|
||||
def classify_hardware_topology(hardware: list[HardwareProfile]) -> TopologyClass:
|
||||
"""Classify selected GPUs for TP-vs-PP/estate advisory output."""
|
||||
if not hardware:
|
||||
raise ProfileError("classify_hardware_topology requires at least one HardwareProfile")
|
||||
if len(hardware) == 1:
|
||||
return TopologyClass.SINGLE_CARD
|
||||
|
||||
vrams = sorted(hw.vram_gb for hw in hardware)
|
||||
sms = {hw.sm for hw in hardware}
|
||||
|
||||
vram_clusters = 1
|
||||
for i in range(1, len(vrams)):
|
||||
if vrams[i] - vrams[i - 1] > 1.0:
|
||||
vram_clusters += 1
|
||||
|
||||
if vram_clusters == 1 and len(sms) == 1:
|
||||
return TopologyClass.HOMOGENEOUS
|
||||
if vram_clusters == 1 and len(sms) > 1:
|
||||
return TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED
|
||||
if vram_clusters > 1:
|
||||
return TopologyClass.VRAM_MISMATCHED
|
||||
return TopologyClass.HETEROGENEOUS_MIXED
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ModelProfile:
|
||||
schema_version: int
|
||||
@@ -203,6 +259,7 @@ class FitsResult:
|
||||
world_size: Optional[int] = None
|
||||
bottleneck_vram_gb: Optional[float] = None
|
||||
homogeneous: Optional[bool] = None
|
||||
topology_class: Optional[TopologyClass] = None
|
||||
kv_projection: Optional[dict[str, Any]] = None
|
||||
compose_name: Optional[str] = None
|
||||
weights_variant: Optional[str] = None
|
||||
@@ -686,6 +743,7 @@ def fits(
|
||||
effective_max_num_seqs = max_num_seqs if max_num_seqs is not None else int(workload.defaults.get("max_num_seqs", 1))
|
||||
effective_weights = resolve_weights_variant(model, engine, weights_variant)
|
||||
homogeneous = len({hw.id for hw in hardware}) <= 1
|
||||
topology_class = classify_hardware_topology(hardware) if hardware else None
|
||||
bottleneck = min((hw.vram_gb for hw in hardware), default=None)
|
||||
effective_cudagraph = _cudagraph_mode(hardware)
|
||||
|
||||
@@ -830,6 +888,14 @@ def fits(
|
||||
else:
|
||||
ok("C12")
|
||||
|
||||
if topology_class is None:
|
||||
skip("C16", "Topology advisory not run; no hardware profiles provided.")
|
||||
else:
|
||||
ok("C16")
|
||||
advisory = TOPOLOGY_ADVISORY.get(topology_class)
|
||||
if advisory:
|
||||
notes.append(f"C16 topology={topology_class.value}: {advisory}")
|
||||
|
||||
diagnostics = {
|
||||
"constraints_evaluated": list(CONSTRAINT_IDS),
|
||||
"constraints_passed": passed,
|
||||
@@ -850,6 +916,7 @@ def fits(
|
||||
world_size=world_size,
|
||||
bottleneck_vram_gb=bottleneck,
|
||||
homogeneous=homogeneous,
|
||||
topology_class=topology_class,
|
||||
kv_projection=kv_projection,
|
||||
weights_variant=effective_weights,
|
||||
diagnostics=diagnostics,
|
||||
|
||||
@@ -19,7 +19,16 @@ if str(REPO_ROOT) not in sys.path:
|
||||
|
||||
os.environ.setdefault("CLUB3090_LOG_LEVEL", "ERROR")
|
||||
|
||||
from scripts.lib.profiles.compat import FitsResult, ProfileError, fits, load_profiles, to_compose_name # noqa: E402
|
||||
from scripts.lib.profiles.compat import ( # noqa: E402
|
||||
TOPOLOGY_ADVISORY,
|
||||
FitsResult,
|
||||
ProfileError,
|
||||
TopologyClass,
|
||||
classify_hardware_topology,
|
||||
fits,
|
||||
load_profiles,
|
||||
to_compose_name,
|
||||
)
|
||||
from scripts.lib.profiles.compose_registry import COMPOSE_REGISTRY # noqa: E402
|
||||
|
||||
|
||||
@@ -103,6 +112,26 @@ def _parse_gpu_specs(value: str, profiles) -> list:
|
||||
return hardware
|
||||
|
||||
|
||||
def _parse_gpu_specs_with_indices(value: str, profiles) -> list[tuple[str, object]]:
|
||||
hardware = []
|
||||
for raw in value.split(";"):
|
||||
raw = raw.strip()
|
||||
if not raw:
|
||||
continue
|
||||
try:
|
||||
idx, name, mem_mib, sm = raw.split("|", 3)
|
||||
except ValueError as exc:
|
||||
raise LaunchCompatError(f"invalid --gpu-spec entry `{raw}`") from exc
|
||||
hardware_id = _hardware_id_from_gpu(name, int(mem_mib), float(sm))
|
||||
try:
|
||||
hardware.append((idx, profiles.hardware[hardware_id]))
|
||||
except KeyError as exc:
|
||||
raise LaunchCompatError(f"hardware profile `{hardware_id}` is not installed") from exc
|
||||
if not hardware:
|
||||
raise LaunchCompatError("no GPU specs were provided for topology classification")
|
||||
return hardware
|
||||
|
||||
|
||||
def _engine_family(engine_type: str) -> str:
|
||||
return "llamacpp" if engine_type == "llama.cpp" else engine_type
|
||||
|
||||
@@ -364,6 +393,90 @@ def command_resolve_variant_pin(args: argparse.Namespace) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
def _hardware_line(index: str, hardware) -> str:
|
||||
return f" GPU {index}: {hardware.display_name} ({hardware.vram_gb:g} GB, sm {hardware.sm:g})"
|
||||
|
||||
|
||||
def _standalone_recommendation(topology: TopologyClass, count: int) -> list[str]:
|
||||
if topology == TopologyClass.SINGLE_CARD:
|
||||
return [
|
||||
"Recommended:",
|
||||
" 1. Use the largest single-card compose your model fits.",
|
||||
" 2. Add another matched card for TP=2 when long-context concurrency matters.",
|
||||
]
|
||||
if topology == TopologyClass.HOMOGENEOUS:
|
||||
return [
|
||||
"Recommended:",
|
||||
f" 1. TP={count} is the default path for matched cards; use the shipped vllm/dual* or multi-card composes.",
|
||||
" 2. Estate planner remains useful when you want separate models/endpoints instead of one larger TP instance.",
|
||||
]
|
||||
if topology == TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED:
|
||||
return [
|
||||
"Recommended:",
|
||||
f" 1. TP={count} works as-is. Compute mismatch means the faster card waits at every NCCL allreduce; effective throughput caps at the slower card's speed (~30% of faster card idle). Full per-card VRAM capacity preserved.",
|
||||
" 2. Estate planner — `bash scripts/launch.sh --estate` runs different models per card, each at full speed.",
|
||||
"",
|
||||
"Not recommended:",
|
||||
" - PP=N: possible as a manual flag flip (`--pipeline-parallel-size N`) on a vllm/dual compose, but no PP compose ships today.",
|
||||
]
|
||||
if topology == TopologyClass.VRAM_MISMATCHED:
|
||||
return [
|
||||
"Recommended:",
|
||||
" 1. llama.cpp `--tensor-split` for weighted layer split on mismatched VRAM.",
|
||||
" 2. PP=N as a manual vLLM flag flip (`--pipeline-parallel-size N`) if you are deliberately experimenting.",
|
||||
" 3. Estate planner — run different models per card or use the largest matched subset.",
|
||||
"",
|
||||
"Not recommended:",
|
||||
" - TP=N on the full mismatched set: the smaller card caps usable model size and KV headroom.",
|
||||
]
|
||||
return [
|
||||
"Recommended:",
|
||||
" 1. Manual selection. Use the largest matched subset for one model.",
|
||||
" 2. Estate planner — put different models on different card subsets.",
|
||||
]
|
||||
|
||||
|
||||
def command_topology(args: argparse.Namespace) -> int:
|
||||
_quiet_compat_logger()
|
||||
profiles = load_profiles()
|
||||
indexed_hardware = _parse_gpu_specs_with_indices(args.gpu_spec, profiles)
|
||||
hardware = [item[1] for item in indexed_hardware]
|
||||
topology = classify_hardware_topology(hardware)
|
||||
advisory = TOPOLOGY_ADVISORY.get(topology)
|
||||
|
||||
if args.format == "wizard":
|
||||
if topology in (TopologyClass.SINGLE_CARD, TopologyClass.HOMOGENEOUS):
|
||||
return 0
|
||||
detected = " + ".join(
|
||||
f"1x {hw.display_name} ({hw.vram_gb:g} GB, sm {hw.sm:g})"
|
||||
for _idx, hw in indexed_hardware
|
||||
)
|
||||
print(f"Detected: {detected}")
|
||||
print("")
|
||||
print(f"Topology: {topology.value}")
|
||||
if advisory:
|
||||
print(f" {advisory}")
|
||||
print("")
|
||||
print("Continue with the selected parallelism if that trade-off is acceptable.")
|
||||
return 0
|
||||
|
||||
print("Detected hardware:")
|
||||
for idx, hw in indexed_hardware:
|
||||
print(_hardware_line(idx, hw))
|
||||
print("")
|
||||
print(f"Topology class: {topology.value}")
|
||||
print("")
|
||||
for line in _standalone_recommendation(topology, len(hardware)):
|
||||
print(line)
|
||||
print("")
|
||||
if advisory:
|
||||
print("Advisory:")
|
||||
print(f" {advisory}")
|
||||
print("")
|
||||
print("For details, see docs/MULTI_CARD.md.")
|
||||
return 0
|
||||
|
||||
|
||||
def build_parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(description="Profile bridge for scripts/launch.sh")
|
||||
sub = parser.add_subparsers(dest="command", required=True)
|
||||
@@ -404,6 +517,11 @@ def build_parser() -> argparse.ArgumentParser:
|
||||
variant_pin.add_argument("--format", choices=("shell", "json", "value"), default="shell")
|
||||
variant_pin.set_defaults(func=command_resolve_variant_pin)
|
||||
|
||||
topology = sub.add_parser("topology")
|
||||
topology.add_argument("--gpu-spec", required=True)
|
||||
topology.add_argument("--format", choices=("standalone", "wizard"), default="standalone")
|
||||
topology.set_defaults(func=command_topology)
|
||||
|
||||
return parser
|
||||
|
||||
|
||||
|
||||
@@ -49,6 +49,77 @@ assert r.recommended_kv_format == "turboquant_3bit_nc"
|
||||
assert r.diagnostics["constraints_skipped"] == ["C12"]
|
||||
PY
|
||||
|
||||
run_test "topology: single card classified" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"]])
|
||||
assert r == TopologyClass.SINGLE_CARD
|
||||
PY
|
||||
|
||||
run_test "topology: 2x3090 classified homogeneous" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"], p.hardware["rtx-3090"]])
|
||||
assert r == TopologyClass.HOMOGENEOUS
|
||||
PY
|
||||
|
||||
run_test "topology: 3090+4090 classified compute-mismatched" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"], p.hardware["rtx-4090"]])
|
||||
assert r == TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED
|
||||
PY
|
||||
|
||||
run_test "topology: 3090+3060 classified VRAM-mismatched" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"], p.hardware["rtx-3060-12gb"]])
|
||||
assert r == TopologyClass.VRAM_MISMATCHED
|
||||
PY
|
||||
|
||||
run_test "topology: VRAM cluster wins over mixed compute" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, classify_hardware_topology, TopologyClass
|
||||
p = load_profiles()
|
||||
r = classify_hardware_topology([p.hardware["rtx-3090"], p.hardware["rtx-3060-12gb"], p.hardware["rtx-4090"]])
|
||||
assert r == TopologyClass.VRAM_MISMATCHED
|
||||
PY
|
||||
|
||||
run_test "C16 topology advisory emits note for compute mismatch" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, fits, TopologyClass
|
||||
p = load_profiles()
|
||||
r = fits(
|
||||
hardware=[p.hardware["rtx-3090"], p.hardware["rtx-4090"]],
|
||||
model=p.models["qwen3.6-27b"],
|
||||
workload=p.workloads["long-ctx-single"],
|
||||
engine=p.engines["vllm-nightly-mtp"],
|
||||
drafter=p.drafters["qwen-mtp-builtin"],
|
||||
tp=2,
|
||||
pp=1,
|
||||
project_vram=False,
|
||||
)
|
||||
assert r.topology_class == TopologyClass.VRAM_MATCHED_COMPUTE_MISMATCHED
|
||||
assert "C16" in r.diagnostics["constraints_passed"]
|
||||
assert any("C16" in n and "vram_matched_compute_mismatched" in n for n in r.notes), r.notes
|
||||
PY
|
||||
|
||||
run_test "C16 topology advisory is silent for homogeneous GPUs" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, fits, TopologyClass
|
||||
p = load_profiles()
|
||||
r = fits(
|
||||
hardware=[p.hardware["rtx-3090"], p.hardware["rtx-3090"]],
|
||||
model=p.models["qwen3.6-27b"],
|
||||
workload=p.workloads["long-ctx-single"],
|
||||
engine=p.engines["vllm-nightly-mtp"],
|
||||
drafter=p.drafters["qwen-mtp-builtin"],
|
||||
tp=2,
|
||||
pp=1,
|
||||
project_vram=False,
|
||||
)
|
||||
assert r.topology_class == TopologyClass.HOMOGENEOUS
|
||||
assert "C16" in r.diagnostics["constraints_passed"]
|
||||
assert not any("C16" in n for n in r.notes), r.notes
|
||||
PY
|
||||
|
||||
run_test "C1 card count: world size mismatch rejected" <<'PY'
|
||||
from scripts.lib.profiles.compat import load_profiles, fits
|
||||
p = load_profiles()
|
||||
@@ -236,7 +307,7 @@ from scripts.lib.profiles.compat import load_profiles, fits
|
||||
p = load_profiles()
|
||||
r = fits([p.hardware["rtx-3090"]], p.models["qwen3.6-27b"], p.workloads["long-ctx-single"], p.engines["vllm-nightly-mtp"], tp=1, project_vram=False)
|
||||
d = r.diagnostics
|
||||
assert d["constraints_evaluated"] == [f"C{i}" for i in range(1, 16)]
|
||||
assert d["constraints_evaluated"] == [f"C{i}" for i in range(1, 17)]
|
||||
assert "constraints_passed" in d and "constraints_failed" in d and "constraints_skipped" in d
|
||||
assert isinstance(d["elapsed_ms"], float)
|
||||
PY
|
||||
|
||||
@@ -143,6 +143,7 @@ out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:
|
||||
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
|
||||
--no-preflight --no-verify --model qwen3.6-27b --gpus 0,1 --no-projection 2>&1)"
|
||||
assert_contains "$out" "[launch] Tensor parallel TP=2"
|
||||
assert_not_contains "$out" "Topology:"
|
||||
assert_contains "$out" "SWITCHED vllm/dual CUDA=0,1 NVD=0,1 TP=2 PP=1"
|
||||
selected_count="$(grep -c "\[launch\] selected variant:" <<< "$out" || true)"
|
||||
if [[ "$selected_count" != "1" ]]; then
|
||||
@@ -179,6 +180,24 @@ if out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6
|
||||
fi
|
||||
assert_contains "$out" "Gemma 4 31B does not fit on a single 24 GB card today"
|
||||
|
||||
out="$(CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:RTX_3090:24576:8.6' \
|
||||
bash "${ROOT_DIR}/scripts/launch.sh" --topology 2>&1)"
|
||||
assert_contains "$out" "Topology class: homogeneous"
|
||||
assert_not_contains "$out" "Compute mismatch detected"
|
||||
|
||||
out="$(CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:RTX_4090:24576:8.9' \
|
||||
bash "${ROOT_DIR}/scripts/launch.sh" --topology 2>&1)"
|
||||
assert_contains "$out" "Topology class: vram_matched_compute_mismatched"
|
||||
assert_contains "$out" "Compute mismatch detected"
|
||||
assert_contains "$out" "Estate planner"
|
||||
|
||||
out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:RTX_4090:24576:8.9' \
|
||||
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
|
||||
--no-preflight --no-verify --model qwen3.6-27b --gpus 0,1 --no-projection 2>&1)"
|
||||
assert_contains "$out" "Topology: vram_matched_compute_mismatched"
|
||||
assert_contains "$out" "Compute mismatch detected"
|
||||
assert_contains "$out" "SWITCHED vllm/dual CUDA=0,1 NVD=0,1 TP=2 PP=1"
|
||||
|
||||
if out="$(MODEL_DIR="${TMP_DIR}/models" CLUB3090_FAKE_GPUS='0:RTX_3090:24576:8.6,1:RTX_3090:24576:8.6,2:RTX_3090:24576:8.6,3:RTX_3090:24576:8.6,4:RTX_3090:24576:8.6,5:RTX_3090:24576:8.6' \
|
||||
SWITCH="${TMP_DIR}/switch-mock" bash "${ROOT_DIR}/scripts/launch.sh" \
|
||||
--no-preflight --no-verify --model qwen3.6-27b --gpus 0,1,2,3,4,5 --tp 6 --no-projection 2>&1)"; then
|
||||
|
||||
Reference in New Issue
Block a user