@lamentofhighborne (#85) submitted the first 1× 3090 cross-rig data on a llama.cpp HOST build (no Docker container) and had to write local verify-full-mtp.sh + verify-stress-mtp.sh adaptations because our shipped scripts assumed vLLM compose stack throughout. Two scripts fixed in this commit: ## verify-full.sh Add `detect_engine()` helper at startup. Probes: 1. /props endpoint → llama.cpp llama-server 2. /v1/chat/completions response.system_fingerprint → "vllm-*" or "sglang-*" 3. CONTAINER name pattern as last-resort fallback Cached as $ENGINE_KIND and surfaced in the script header. Step 2 (Genesis check) and step 8 (MTP acceptance via SpecDecoding log scrape) now skip with engine-aware messages on llamacpp/sglang/unknown instead of failing on missing docker or missing log format. Net: a contributor on a host-build llama.cpp endpoint sees: [2/8] Genesis patches applied ... ⊘ llama.cpp engine — Genesis is vLLM-only, not applicable (skipped) instead of the previous misleading "no Genesis marker in logs". ## soak-test.sh `docker` becomes soft-required. New `CONTAINER=none` (or implicit when docker isn't in PATH) puts the script into HOST_MODE which: - Skips all `docker ps`/`docker port`/`docker stats`/`docker inspect` - Uses URL env var directly (fall back to localhost:8020) - Tracks VRAM via bare `nvidia-smi --query-gpu=memory.used` (no docker stats) Smoke-tested against running gemma-mtp endpoint with CONTAINER=none: boots cleanly, "[soak] host mode: CONTAINER=none — skipping docker checks", VRAM tracking + per-turn decode runs via HTTP. ## Not in this commit (deferred) verify-stress.sh has the same engine-coupling pattern but the docker references are mostly diagnostic *hints* in error messages (telling users where to find logs). Tracked in #87 as a follow-up; not blocking host-build users today. ## Test plan - [x] vLLM compose path: no regression (engine=vllm detected, all docker-dependent checks ran as before) - [x] CONTAINER=none + vLLM endpoint: host mode triggers, docker checks skipped, all HTTP-based checks ran - [ ] llama.cpp host build: would route to "llamacpp" engine class via /props endpoint, skip Genesis + MTP-log checks with clear messages. @lamentofhighborne or any future host-build contributor can validate.
This commit is contained in:
+37
-9
@@ -81,13 +81,28 @@ cd "$REPO_ROOT"
|
||||
log() { printf '[soak] %s\n' "$*"; }
|
||||
die() { log "ERROR: $*"; exit 2; }
|
||||
need() { command -v "$1" >/dev/null 2>&1 || die "'$1' not found in PATH"; }
|
||||
soft_need() { command -v "$1" >/dev/null 2>&1; }
|
||||
|
||||
need curl
|
||||
need docker
|
||||
need nvidia-smi
|
||||
need python3
|
||||
[[ -x "$HELPER" || -f "$HELPER" ]] || die "missing helper: $HELPER"
|
||||
|
||||
# docker is soft-required: only needed for container-mode tracking
|
||||
# (docker stats + docker logs scrape). Host engines (e.g. llama.cpp host
|
||||
# build, see #85, #87) use CONTAINER=none and run without docker.
|
||||
HAVE_DOCKER=0
|
||||
if soft_need docker; then HAVE_DOCKER=1; fi
|
||||
if [[ "${CONTAINER:-}" == "none" ]]; then
|
||||
HOST_MODE=1
|
||||
elif [[ "$HAVE_DOCKER" == "0" ]]; then
|
||||
log "docker not in PATH — running in host mode (CONTAINER=none implied)"
|
||||
HOST_MODE=1
|
||||
CONTAINER="none"
|
||||
else
|
||||
HOST_MODE=0
|
||||
fi
|
||||
|
||||
auto_container() {
|
||||
docker ps --format '{{.Names}}' 2>/dev/null \
|
||||
| grep -E '^(vllm-qwen36-27b|vllm-gemma-4-31b)' \
|
||||
@@ -124,8 +139,10 @@ capture_state() {
|
||||
local label="$1"
|
||||
nvidia-smi --query-gpu=index,name,memory.used,memory.total,utilization.gpu,power.draw,temperature.gpu \
|
||||
--format=csv,noheader,nounits > "${SOAK_OUTPUT}/nvidia-smi-${label}.csv" 2>/dev/null || true
|
||||
docker stats --no-stream --format '{{json .}}' "$CONTAINER" \
|
||||
> "${SOAK_OUTPUT}/docker-stats-${label}.jsonl" 2>/dev/null || true
|
||||
if [[ "$HOST_MODE" == "0" ]]; then
|
||||
docker stats --no-stream --format '{{json .}}' "$CONTAINER" \
|
||||
> "${SOAK_OUTPUT}/docker-stats-${label}.jsonl" 2>/dev/null || true
|
||||
fi
|
||||
}
|
||||
|
||||
finish() {
|
||||
@@ -137,13 +154,24 @@ finish() {
|
||||
trap finish EXIT
|
||||
trap 'log "interrupted"; exit 2' INT TERM
|
||||
|
||||
CONTAINER="${CONTAINER:-$(auto_container)}"
|
||||
[[ -n "$CONTAINER" ]] || die "no running vllm-qwen36-27b* container found; set CONTAINER=..."
|
||||
docker inspect "$CONTAINER" >/dev/null 2>&1 || die "container '$CONTAINER' not found"
|
||||
[[ "$(docker inspect -f '{{.State.Running}}' "$CONTAINER" 2>/dev/null || echo false)" == "true" ]] \
|
||||
|| die "container '$CONTAINER' is not running"
|
||||
if [[ "$HOST_MODE" == "1" ]]; then
|
||||
log "host mode: CONTAINER=none — skipping docker checks (URL must be set or auto-detected)"
|
||||
CONTAINER="none"
|
||||
else
|
||||
CONTAINER="${CONTAINER:-$(auto_container)}"
|
||||
[[ -n "$CONTAINER" ]] || die "no running vllm-qwen36-27b*/vllm-gemma-4-31b* container found; set CONTAINER=... or CONTAINER=none for host engines"
|
||||
docker inspect "$CONTAINER" >/dev/null 2>&1 || die "container '$CONTAINER' not found (use CONTAINER=none for host engine builds)"
|
||||
[[ "$(docker inspect -f '{{.State.Running}}' "$CONTAINER" 2>/dev/null || echo false)" == "true" ]] \
|
||||
|| die "container '$CONTAINER' is not running"
|
||||
fi
|
||||
|
||||
ENDPOINT="${ENDPOINT:-${URL:-$(endpoint_from_container "$CONTAINER")}}"
|
||||
if [[ "$HOST_MODE" == "1" ]]; then
|
||||
# Host mode: URL must be set explicitly (or fall back to localhost:8020).
|
||||
# We can't sniff a port from a container that doesn't exist.
|
||||
ENDPOINT="${ENDPOINT:-${URL:-http://localhost:8020}}"
|
||||
else
|
||||
ENDPOINT="${ENDPOINT:-${URL:-$(endpoint_from_container "$CONTAINER")}}"
|
||||
fi
|
||||
mkdir -p "$SOAK_OUTPUT"
|
||||
|
||||
MODELS_JSON="${SOAK_OUTPUT}/models.json"
|
||||
|
||||
+60
-5
@@ -65,13 +65,46 @@ pass() { printf " \033[32m✓\033[0m %s\n" "$1"; }
|
||||
fail() { printf " \033[31m✗\033[0m %s\n" "$1"; printf " \033[33m→\033[0m %s\n" "$2"; return 1; }
|
||||
skip() { printf " \033[33m⊘\033[0m %s (skipped)\n" "$1"; }
|
||||
|
||||
# ---- Engine detection ---------------------------------------------------
|
||||
# Returns one of: vllm | llamacpp | sglang | unknown
|
||||
# Used to gate engine-coupled checks (Genesis markers, MTP-acceptance log
|
||||
# scrape) so non-vLLM engines (especially llama.cpp host builds without
|
||||
# Docker) get clean skips rather than misleading failures or fail-paths
|
||||
# that the user can't act on. Surfaced by @lamentofhighborne in #85, fixed
|
||||
# per #87. Engine class is detected ONCE at startup and cached.
|
||||
detect_engine() {
|
||||
# Hint 1: llama-server's /props endpoint (vLLM doesn't ship it)
|
||||
if curl -sf -m 3 "${URL}/props" >/dev/null 2>&1; then
|
||||
echo "llamacpp"; return 0
|
||||
fi
|
||||
# Hint 2: vLLM's chat-completion response includes system_fingerprint
|
||||
# like "vllm-0.20.2rc1.dev9+g01d4d1ad3-tp2-c9120464".
|
||||
local fp
|
||||
fp="$(curl -sf -m 5 "${URL}/v1/chat/completions" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d "{\"model\":\"${MODEL}\",\"messages\":[{\"role\":\"user\",\"content\":\"hi\"}],\"max_tokens\":1}" 2>/dev/null \
|
||||
| python3 -c "import sys,json; d=json.load(sys.stdin); print(d.get('system_fingerprint','') or '')" 2>/dev/null)"
|
||||
case "$fp" in
|
||||
vllm-*) echo "vllm"; return 0 ;;
|
||||
sglang-*) echo "sglang"; return 0 ;;
|
||||
esac
|
||||
# Hint 3: container name pattern as a fallback (cheap, no extra HTTP)
|
||||
case "$CONTAINER" in
|
||||
vllm-*) echo "vllm"; return 0 ;;
|
||||
llama-cpp-*) echo "llamacpp"; return 0 ;;
|
||||
esac
|
||||
echo "unknown"
|
||||
}
|
||||
ENGINE_KIND="$(detect_engine)"
|
||||
|
||||
FAILED=0
|
||||
run_check() {
|
||||
local label="$1"; shift
|
||||
if "$@"; then :; else FAILED=$((FAILED + 1)); fi
|
||||
}
|
||||
|
||||
echo "Running FULL functional test against ${URL} (model=${MODEL}, container=${CONTAINER})"
|
||||
echo "Running FULL functional test against ${URL}"
|
||||
echo " model=${MODEL} container=${CONTAINER} engine=${ENGINE_KIND}"
|
||||
echo ""
|
||||
|
||||
# --------------------------------------------------------------------
|
||||
@@ -93,12 +126,20 @@ run_check "server" check_server
|
||||
# --------------------------------------------------------------------
|
||||
check_patches() {
|
||||
echo "[2/8] Genesis patches applied ..."
|
||||
# Genesis is a vLLM-only patcher. Skip cleanly on other engines instead of
|
||||
# leaving the user wondering whether "no Genesis marker" means a real
|
||||
# problem or a category error.
|
||||
case "$ENGINE_KIND" in
|
||||
llamacpp) skip "llama.cpp engine — Genesis is vLLM-only, not applicable"; return 0 ;;
|
||||
sglang) skip "SGLang engine — Genesis is vLLM-only, not applicable"; return 0 ;;
|
||||
unknown) ;; # fall through; might still be vLLM under a non-standard container name
|
||||
esac
|
||||
if ! command -v docker >/dev/null 2>&1; then
|
||||
skip "docker not in PATH"
|
||||
skip "docker not in PATH (host engine build?)"
|
||||
return 0
|
||||
fi
|
||||
if ! docker inspect "${CONTAINER}" >/dev/null 2>&1; then
|
||||
skip "container '${CONTAINER}' not found"
|
||||
skip "container '${CONTAINER}' not found (host engine build? CONTAINER=none for host endpoints)"
|
||||
return 0
|
||||
fi
|
||||
# Anchors updated 2026-05-02 for Genesis v7.14+ logging conventions (the old
|
||||
@@ -376,12 +417,26 @@ run_check "output_quality" check_output_quality
|
||||
# --------------------------------------------------------------------
|
||||
check_mtp_acceptance() {
|
||||
echo "[8/8] MTP acceptance length threshold ..."
|
||||
# Spec-decode metrics extraction is engine-specific:
|
||||
# vLLM emits "SpecDecoding metrics: Mean acceptance length: N.NN" to stdout
|
||||
# llama.cpp llama-server doesn't emit a "Mean acceptance length" line; spec
|
||||
# metrics are inferred from per-slot accept counts in the response timings
|
||||
# (engine-internal, not exposed via OpenAI API)
|
||||
# SGLang has its own format
|
||||
# For non-vLLM engines we skip rather than fail — the per-engine spec-decode
|
||||
# validation is the user's responsibility (e.g. llama.cpp users run their own
|
||||
# verify-full-mtp.sh adaptations like @lamentofhighborne's, until #87 lands a
|
||||
# generalized harness).
|
||||
case "$ENGINE_KIND" in
|
||||
llamacpp) skip "llama.cpp engine — MTP acceptance check is vLLM-log-format-specific (run engine-side verification separately)"; return 0 ;;
|
||||
sglang) skip "SGLang engine — MTP acceptance check is vLLM-log-format-specific"; return 0 ;;
|
||||
esac
|
||||
if ! command -v docker >/dev/null 2>&1; then
|
||||
skip "docker not in PATH"
|
||||
skip "docker not in PATH (host engine build? — see #87 for generalized harness work)"
|
||||
return 0
|
||||
fi
|
||||
if ! docker inspect "${CONTAINER}" >/dev/null 2>&1; then
|
||||
skip "container '${CONTAINER}' not found"
|
||||
skip "container '${CONTAINER}' not found (CONTAINER=none for host endpoints)"
|
||||
return 0
|
||||
fi
|
||||
|
||||
|
||||
Reference in New Issue
Block a user