Files
club-3090/scripts/tests/test-stream-toolcall-probe.sh
T
99ced8ea74 verify-full: add streaming tool-call check + a reusable streaming probe (#404)
#145 / vLLM #39056 are STREAMING-only tool-call bugs, and our gates were
structurally blind to them: benchlocal-cli runs every pack non-streaming
(benchlocal-cli#68, closed wrong-layer), and verify-full had check_tools
(non-streaming) and check_streaming (no tools) as SEPARATE checks — the bug
lives in their untested intersection (tools × streaming × thinking-on). That's
how the MTP-streaming drop (#39598, un-mitigated since Genesis retired) shipped
silently into the dual composes.

- scripts/verify-full.sh: new [6/9] check_streaming_tools — stream:true + tools
  + tool_choice=auto + enable_thinking=true, reassembles the SSE deltas, and
  asserts delta.tool_calls (get_weather) + finish_reason=tool_calls + no
  <tool_call> leak into delta.content. Uses tool_choice=auto (the clean path on
  v0.22.0); hard-fails on the leak signature (the #145/#39056 class). NB:
  tool_choice=required + MTP is a known-open drop (#39598) — the check notes it
  but uses auto so it's a green gate on shipped composes. Gated by SKIP_TOOLS.
  Renumbered the suite 8 -> 9 checks. Live-validated: vllm/dual all 9 pass.
- scripts/stream-toolcall-probe.py: the deep-sweep instrument behind the #400
  investigation (thinking-on, multi-prompt, auto+required, SSE reassembly,
  PASS/DROP/OTHER classification, exit-1 on any DROP). It's what proved
  qwen3_coder == qwen3_xml and isolated the #39598 MTP-gating. Run it twice
  (control vs candidate) for a streaming A/B.
- scripts/tests/test-stream-toolcall-probe.sh: offline mock-SSE gate (clean +
  #145-drop modes) — no GPU needed.

Full suite green (test-compose-registry-disk fails pre-existing on an unrelated
untracked model dir).

Co-authored-by: noonghunna <[email protected]>
Co-authored-by: Claude Opus 4.8 <[email protected]>
2026-06-14 04:17:04 +05:00

89 lines
4.2 KiB
Bash
Executable File

#!/usr/bin/env bash
# Offline test for scripts/stream-toolcall-probe.py. Stands up a mock SSE server
# that streams either a CLEAN tool-call or the #145 DROP signature, and asserts
# the probe classifies + exit-codes each correctly. No GPU / real model needed.
set -euo pipefail
ROOT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../.." && pwd)"
cd "$ROOT_DIR"
PORT=8771
SRV_PID=""
tmplog="$(mktemp)"
cleanup() {
[[ -n "$SRV_PID" ]] && kill "$SRV_PID" 2>/dev/null || true
rm -f "$tmplog"
}
trap cleanup EXIT
# Mock OpenAI streaming endpoint. MOCK_MODE=clean → proper delta.tool_calls +
# finish_reason:tool_calls. MOCK_MODE=drop → tool_call XML in delta.content +
# finish_reason:stop (the club-3090#145 streaming-drop signature).
start_server() {
MOCK_MODE="$1" python3 - "$PORT" <<'PY' &
import json, os, sys
from http.server import BaseHTTPRequestHandler, HTTPServer
MODE = os.environ.get("MOCK_MODE", "clean")
def sse(obj): return f"data: {json.dumps(obj)}\n\n".encode()
class H(BaseHTTPRequestHandler):
def log_message(self, *a): pass
def do_GET(self):
# /v1/models health
self.send_response(200); self.send_header("Content-Type","application/json"); self.end_headers()
self.wfile.write(json.dumps({"data":[{"id":"mock"}]}).encode())
def do_POST(self):
n = int(self.headers.get("Content-Length", 0)); self.rfile.read(n)
self.send_response(200)
self.send_header("Content-Type", "text/event-stream"); self.end_headers()
if MODE == "clean":
frames = [
{"choices":[{"index":0,"delta":{"role":"assistant"},"finish_reason":None}]},
{"choices":[{"index":0,"delta":{"tool_calls":[{"index":0,"id":"c1","type":"function","function":{"name":"read_file","arguments":""}}]},"finish_reason":None}]},
{"choices":[{"index":0,"delta":{"tool_calls":[{"index":0,"function":{"arguments":"{\"path\":"}}]},"finish_reason":None}]},
{"choices":[{"index":0,"delta":{"tool_calls":[{"index":0,"function":{"arguments":" \"/etc/hosts\"}"}}]},"finish_reason":None}]},
{"choices":[{"index":0,"delta":{},"finish_reason":"tool_calls"}]},
]
else: # drop — XML leaks into content, no tool_calls, finish stop
frames = [
{"choices":[{"index":0,"delta":{"role":"assistant"},"finish_reason":None}]},
{"choices":[{"index":0,"delta":{"content":"<tool_call>\n{\"name\": \"read_file\", "},"finish_reason":None}]},
{"choices":[{"index":0,"delta":{"content":"\"arguments\": {\"path\": \"/etc/hosts\"}}\n</tool_call>"},"finish_reason":None}]},
{"choices":[{"index":0,"delta":{},"finish_reason":"stop"}]},
]
for f in frames:
self.wfile.write(sse(f)); self.wfile.flush()
self.wfile.write(b"data: [DONE]\n\n"); self.wfile.flush()
HTTPServer(("127.0.0.1", int(sys.argv[1])), H).serve_forever()
PY
SRV_PID=$!
for _ in $(seq 1 50); do
curl -sf "http://127.0.0.1:${PORT}/v1/models" >/dev/null 2>&1 && return 0
sleep 0.1
done
echo "mock server didn't come up" >&2; exit 1
}
run_probe() {
# tiny sweep: required only, 1 repeat — the mock answers all prompts identically
python3 scripts/stream-toolcall-probe.py --url "http://127.0.0.1:${PORT}" \
--model mock --thinking off --tool-choice required --repeat 1 --timeout 10 "$@"
}
# --- CLEAN: every request should PASS, exit 0 ---
start_server clean
set +e; run_probe >"$tmplog" 2>&1; rc=$?; set -e
kill "$SRV_PID" 2>/dev/null || true; SRV_PID=""
if [[ $rc -ne 0 ]]; then echo "FAIL: clean mode exited $rc (expected 0)"; cat "$tmplog"; exit 1; fi
grep -q "DROP 0" "$tmplog" || { echo "FAIL: clean mode reported a DROP"; cat "$tmplog"; exit 1; }
echo "clean mode: probe PASSed, 0 drops, exit 0 ✓"
# --- DROP: every request should be flagged DROP, exit 1 ---
start_server drop
set +e; run_probe >"$tmplog" 2>&1; rc=$?; set -e
kill "$SRV_PID" 2>/dev/null || true; SRV_PID=""
if [[ $rc -ne 1 ]]; then echo "FAIL: drop mode exited $rc (expected 1)"; cat "$tmplog"; exit 1; fi
grep -q "STREAMING TOOL-CALL DROPS" "$tmplog" || { echo "FAIL: drop mode didn't flag the #145 signature"; cat "$tmplog"; exit 1; }
echo "drop mode: probe flagged DROP, exit 1 ✓"
echo "test-stream-toolcall-probe: ok"