Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
22bf2e9398 | ||
|
|
99a0b66224 | ||
|
|
ef770322f4 | ||
|
|
301083491b |
@@ -0,0 +1,20 @@
|
||||
## Rig bench submission
|
||||
|
||||
> ⚠ Most bench submissions go through an issue (see `CONTRIBUTING.md` "Submitting your bench").
|
||||
> This PR template is for contributors who explicitly chose the direct-PR path.
|
||||
> The maintainer may redirect to an issue thread before merge.
|
||||
|
||||
<!-- This PR was auto-generated by `bash scripts/submit-bench.sh --auto-submit --as-pr --tag <TAG>`. -->
|
||||
<!-- Review the row below; the PR reviewer may move it within the target section. -->
|
||||
|
||||
### New row
|
||||
|
||||
<!-- The generated BENCHMARKS.md row goes here -->
|
||||
|
||||
### Rig
|
||||
|
||||
<!-- Output of `bash scripts/report.sh` (redacted) -->
|
||||
|
||||
### Full results
|
||||
|
||||
See `results/rebench/<TAG>/REPORT.md` for the full per-phase breakdown.
|
||||
@@ -16,6 +16,26 @@ history; SemVer takes over from `v0.3.0` onward.
|
||||
|
||||
---
|
||||
|
||||
## v0.5.3 — 2026-05-13
|
||||
|
||||
|
||||
### ✨ Features
|
||||
|
||||
- feat(scripts): add submit-bench flow ([ef77032](https://github.com/noonghunna/club-3090/commit/ef770322f43724f612a80393f547e5da218b5bf7))
|
||||
|
||||
|
||||
|
||||
[Pin: `git checkout v0.5.3`] · [Full diff](https://github.com/noonghunna/club-3090/compare/v0.5.2...v0.5.3)
|
||||
## v0.5.2 — 2026-05-13
|
||||
|
||||
|
||||
### 🎯 New models + serving paths
|
||||
|
||||
- Add hardware-aware compose preflight ([2698552](https://github.com/noonghunna/club-3090/commit/26985527f75d8da2a32a8a2f985989d5dcf9e89a))
|
||||
|
||||
|
||||
|
||||
[Pin: `git checkout v0.5.2`] · [Full diff](https://github.com/noonghunna/club-3090/compare/v0.5.1...v0.5.2)
|
||||
## v0.5.1 — 2026-05-13
|
||||
|
||||
|
||||
|
||||
@@ -45,6 +45,40 @@ Two GitHub channels, two different shapes of conversation. Picking the right one
|
||||
|
||||
---
|
||||
|
||||
## Submitting your bench
|
||||
|
||||
The matrix is hand-curated — the canonical path is to file an **issue** with your rig + numbers; we'll review, ask clarifying questions, and integrate.
|
||||
|
||||
After running `bash scripts/rebench-full.sh`, generate a paste-ready row:
|
||||
|
||||
```bash
|
||||
bash scripts/submit-bench.sh --tag <your-tag>
|
||||
```
|
||||
|
||||
The script writes `results/rebench/<tag>/BENCHMARKS-row.md`. To submit:
|
||||
|
||||
### Path A — Auto-issue (recommended, requires `gh auth login`)
|
||||
|
||||
```bash
|
||||
bash scripts/submit-bench.sh --tag <your-tag> --auto-submit
|
||||
```
|
||||
|
||||
Opens an issue via `gh issue create` with your rig + row pre-filled.
|
||||
|
||||
### Path B — Manual issue (no tools beyond browser)
|
||||
|
||||
Open https://github.com/noonghunna/club-3090/issues/new?template=numbers-from-your-rig.yml and paste the row + your `rig.txt` into the body.
|
||||
|
||||
### Path C — Direct PR (advanced)
|
||||
|
||||
```bash
|
||||
bash scripts/submit-bench.sh --tag <your-tag> --auto-submit --as-pr
|
||||
```
|
||||
|
||||
For contributors who know the `BENCHMARKS.md` section structure and want to propose the exact row. The maintainer may still redirect to an issue thread for context-gathering before merge — direct PRs aren't a fast-path bypass.
|
||||
|
||||
---
|
||||
|
||||
## Process for non-trivial changes
|
||||
|
||||
1. **Open an issue first** for anything bigger than a typo fix or a one-line measurement contribution. We'll either align on shape or explain why we'd land it differently — saves you a wasted afternoon.
|
||||
|
||||
@@ -0,0 +1,416 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Formatter for one-row BENCHMARKS.md submissions from results/rebench/<tag>/.
|
||||
#
|
||||
# Public functions:
|
||||
# bench_row_format <rebench-tag-dir>
|
||||
# bench_row_section <rebench-tag-dir>
|
||||
# bench_row_fixtures
|
||||
|
||||
_BENCH_ROW_LIB_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)"
|
||||
_BENCH_ROW_ROOT="$(cd -- "${_BENCH_ROW_LIB_DIR}/../.." && pwd)"
|
||||
|
||||
bench_row_fixtures() {
|
||||
local tag
|
||||
for tag in \
|
||||
qwen-int8-pth-n4-2026-05-10 \
|
||||
qwen-bf16-n4-2026-05-11 \
|
||||
qwen-int8-tq3-n3-2026-05-11 \
|
||||
qwen-tq3-mtp-genesis-2026-05-11 \
|
||||
gemma-int8-pth-n4-2026-05-11 \
|
||||
gemma-bf16-n4-2026-05-11; do
|
||||
if [[ -d "${_BENCH_ROW_ROOT}/results/rebench/${tag}" ]]; then
|
||||
printf '%s\n' "${_BENCH_ROW_ROOT}/results/rebench/${tag}"
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
bench_row_section() {
|
||||
_bench_row_python section "$1"
|
||||
}
|
||||
|
||||
bench_row_format() {
|
||||
_bench_row_python row "$1"
|
||||
}
|
||||
|
||||
bench_row_rig_shortname() {
|
||||
_bench_row_python rig-short "$1"
|
||||
}
|
||||
|
||||
_bench_row_python() {
|
||||
local mode="$1"
|
||||
local tag_dir="$2"
|
||||
BENCH_ROW_REPO_ROOT="${_BENCH_ROW_ROOT}" python3 - "$mode" "$tag_dir" <<'PY'
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
MODE = sys.argv[1]
|
||||
TAG_DIR = Path(sys.argv[2]).resolve()
|
||||
ROOT = Path(os.environ.get("BENCH_ROW_REPO_ROOT", ".")).resolve()
|
||||
|
||||
|
||||
def die(msg: str) -> None:
|
||||
print(f"[bench-row] ERROR: {msg}", file=sys.stderr)
|
||||
raise SystemExit(1)
|
||||
|
||||
|
||||
def read_text(path: Path) -> str:
|
||||
try:
|
||||
return path.read_text(errors="replace")
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def read_json(path: Path) -> Any:
|
||||
try:
|
||||
return json.loads(path.read_text(errors="replace"))
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def require_file(name: str) -> Path:
|
||||
path = TAG_DIR / name
|
||||
if not path.is_file():
|
||||
die(f"missing required artifact: {path}")
|
||||
return path
|
||||
|
||||
|
||||
def first_container(blob: Any) -> dict[str, Any]:
|
||||
if isinstance(blob, list) and blob:
|
||||
return blob[0] if isinstance(blob[0], dict) else {}
|
||||
return blob if isinstance(blob, dict) else {}
|
||||
|
||||
|
||||
def flag(cmd: list[str], name: str) -> str:
|
||||
try:
|
||||
i = cmd.index(name)
|
||||
return str(cmd[i + 1])
|
||||
except Exception:
|
||||
return "?"
|
||||
|
||||
|
||||
def rel(path: str) -> str:
|
||||
if not path:
|
||||
return ""
|
||||
p = Path(path)
|
||||
try:
|
||||
return str(p.resolve().relative_to(ROOT))
|
||||
except Exception:
|
||||
return str(p)
|
||||
|
||||
|
||||
def infer_compose_path(container_name: str, served: str, tp: str) -> str:
|
||||
name = container_name.lstrip("/")
|
||||
is_gemma = "gemma" in name or "gemma" in served
|
||||
model_root = "models/gemma-4-31b/vllm/compose" if is_gemma else "models/qwen3.6-27b/vllm/compose"
|
||||
|
||||
mapping = {
|
||||
"dual-int8-tq3": "dual/int8-tq3.yml",
|
||||
"dual-tq3-mtp-genesis": "dual/tq3-mtp-genesis.yml",
|
||||
"dual-tq3-nomtp": "dual/tq3-nomtp.yml",
|
||||
"dual-tq3-mtp": "dual/tq3-mtp.yml",
|
||||
"dual-int8": "dual/int8.yml",
|
||||
"dual-bf16": "dual/bf16.yml",
|
||||
"dual-dflash-noviz": "dual/dflash-noviz.yml",
|
||||
"dual-dflash": "dual/dflash.yml",
|
||||
"dual-turbo": "dual/turbo.yml",
|
||||
"dual": "dual/docker-compose.yml",
|
||||
"minimal": "single/minimal.yml",
|
||||
"tools-text": "single/tools-text.yml",
|
||||
"long-text-no-mtp": "single/long-text-no-mtp.yml",
|
||||
"long-text": "single/long-text.yml",
|
||||
"long-vision": "single/long-vision.yml",
|
||||
}
|
||||
for needle, suffix in mapping.items():
|
||||
if needle in name:
|
||||
return f"{model_root}/{suffix}"
|
||||
if tp == "4":
|
||||
return "models/qwen3.6-27b/vllm/compose/multi4/docker-compose.yml"
|
||||
if tp == "2":
|
||||
return f"{model_root}/dual/docker-compose.yml"
|
||||
return f"{model_root}/single/docker-compose.yml"
|
||||
|
||||
|
||||
def compose_display(compose_path: str, served: str) -> str:
|
||||
path = compose_path.replace("\\", "/")
|
||||
parts = path.split("/")
|
||||
base = parts[-1] if parts else path
|
||||
parent = parts[-2] if len(parts) >= 2 else ""
|
||||
is_gemma = "gemma" in served or "gemma-4-31b" in path
|
||||
|
||||
if base == "docker-compose.yml":
|
||||
if parent == "dual":
|
||||
return "dual.yml"
|
||||
if parent == "multi4":
|
||||
return "dual4.yml"
|
||||
if parent == "single":
|
||||
return "vllm/gemma-mtp-tp1" if is_gemma else "vllm/default"
|
||||
if parent in {"dual", "multi4"} and not base.startswith(f"{parent}-"):
|
||||
return f"{parent}-{base}"
|
||||
return base
|
||||
|
||||
|
||||
def parse_rig(rig_txt: str) -> dict[str, Any]:
|
||||
out: dict[str, Any] = {"gpus": []}
|
||||
for raw in rig_txt.splitlines():
|
||||
line = raw.strip()
|
||||
if not line:
|
||||
continue
|
||||
if line.startswith("hostname:"):
|
||||
out["hostname"] = line.split(":", 1)[1].strip()
|
||||
elif line.startswith("GPU "):
|
||||
gpu = line.split(":", 1)[1].split("(UUID", 1)[0].strip()
|
||||
out["gpus"].append(gpu)
|
||||
elif line.startswith("power_cap_w:"):
|
||||
out["power_cap_w"] = line.split(":", 1)[1].strip()
|
||||
return out
|
||||
|
||||
|
||||
def simplify_gpu(name: str) -> str:
|
||||
name = re.sub(r"^NVIDIA\s+", "", name)
|
||||
name = re.sub(r"^GeForce\s+", "", name)
|
||||
name = re.sub(r"^RTX\s+", "", name)
|
||||
return name.strip()
|
||||
|
||||
|
||||
def rig_shape(rig: dict[str, Any]) -> str:
|
||||
gpus = [simplify_gpu(g) for g in rig.get("gpus") or []]
|
||||
if not gpus:
|
||||
shape = "rig"
|
||||
elif len(set(gpus)) == 1:
|
||||
shape = f"{len(gpus)}× {gpus[0]}"
|
||||
else:
|
||||
shape = " + ".join(gpus)
|
||||
power = str(rig.get("power_cap_w") or "").strip()
|
||||
if power:
|
||||
try:
|
||||
power = f"{float(power):.0f} W/card"
|
||||
except Exception:
|
||||
power = f"{power} W/card"
|
||||
return f"{shape}, {power}"
|
||||
return shape
|
||||
|
||||
|
||||
def rig_cell(rig: dict[str, Any]) -> str:
|
||||
user = os.environ.get("BENCH_ROW_GITHUB_USER", "").strip().lstrip("@") or "your-handle"
|
||||
return f"@{user} ({rig_shape(rig)})"
|
||||
|
||||
|
||||
def short_date(tag: str, report: str) -> str:
|
||||
m = re.search(r"(20\d{2}-\d{2}-\d{2})", tag)
|
||||
if m:
|
||||
return m.group(1)
|
||||
m = re.search(r"\*\*Date:\*\*\s*(20\d{2}-\d{2}-\d{2})", report)
|
||||
return m.group(1) if m else "—"
|
||||
|
||||
|
||||
def kv_display(raw: str, served: str) -> str:
|
||||
raw = (raw or "?").strip("`")
|
||||
lowered = raw.lower()
|
||||
if lowered in {"turboquant_3bit_nc", "tq3"}:
|
||||
return "TQ3"
|
||||
if lowered in {"fp8_e5m2", "fp8", "fp8_e4m3"}:
|
||||
return "fp8"
|
||||
if lowered in {"bfloat16", "bf16"}:
|
||||
return "bf16"
|
||||
if lowered == "auto":
|
||||
return "bf16"
|
||||
if lowered == "int8_per_token_head":
|
||||
return "int8_per_token_head"
|
||||
return raw or "?"
|
||||
|
||||
|
||||
def fmt_ctx(value: str) -> str:
|
||||
try:
|
||||
n = int(str(value).replace(",", ""))
|
||||
except Exception:
|
||||
return str(value or "?")
|
||||
if n >= 1000:
|
||||
return f"{round(n / 1000):.0f}K"
|
||||
return str(n)
|
||||
|
||||
|
||||
def fmt_tps(value: Any) -> str:
|
||||
try:
|
||||
return f"{float(value):.2f}"
|
||||
except Exception:
|
||||
return "?"
|
||||
|
||||
|
||||
def parse_mib(text: str) -> int | None:
|
||||
m = re.search(r"([\d.]+)\s*MiB", str(text))
|
||||
if not m:
|
||||
return None
|
||||
try:
|
||||
return int(float(m.group(1)))
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def peak_vram(internal: dict[str, Any], gpu_count: int) -> str:
|
||||
vals: list[int] = []
|
||||
for g in (((internal.get("bench") or {}).get("gpu_state")) or []):
|
||||
mib = parse_mib(g.get("mem_used", ""))
|
||||
if mib is not None:
|
||||
vals.append(mib)
|
||||
if not vals:
|
||||
return "TBD"
|
||||
suffix = "/card" if gpu_count > 1 else ""
|
||||
return f"{max(vals) / 1024:.1f} GB{suffix}"
|
||||
|
||||
|
||||
def parse_jsonish(value: str) -> dict[str, Any]:
|
||||
try:
|
||||
parsed = json.loads(value)
|
||||
return parsed if isinstance(parsed, dict) else {}
|
||||
except Exception:
|
||||
return {}
|
||||
|
||||
|
||||
def verify_summary(report: str) -> str:
|
||||
m = re.search(r"Verify-stress:\s+\*\*(\d+/\d+)\*\*", report)
|
||||
if m:
|
||||
return m.group(1)
|
||||
m = re.search(r"\*\*Overall:\*\*\s+(PASS|FAIL|\?)", report)
|
||||
return m.group(1) if m else "?"
|
||||
|
||||
|
||||
def soak_note(soak: dict[str, Any]) -> str:
|
||||
verdict = str(soak.get("verdict") or "").upper()
|
||||
silent = str(soak.get("silent_empty") or "")
|
||||
growth = str(soak.get("max_growth_mib") or "")
|
||||
if not verdict:
|
||||
return "Soak: —"
|
||||
if verdict == "PASS" and (silent.startswith("0 ") or silent.startswith("0/") or silent == "0"):
|
||||
return "Soak: ✓ PASS"
|
||||
if verdict == "PASS":
|
||||
return f"Soak: ⚠ borderline ({silent or growth})"
|
||||
return f"Soak: ✗ {verdict}"
|
||||
|
||||
|
||||
def section_name(compose_path: str, served: str, tp: str, container: str) -> str:
|
||||
if "gemma" in served or "gemma" in container or "gemma-4-31b" in compose_path:
|
||||
return "Gemma 4 31B (community-experimental)"
|
||||
path = compose_path.replace("\\", "/")
|
||||
if "llama-cpp" in path or "llama-cpp" in container:
|
||||
return "Single-card (1× RTX 3090) — llama.cpp"
|
||||
if "/multi4/" in path or tp == "4":
|
||||
return "Quad-card (4× RTX 3090, TP=4)"
|
||||
if "/dual/" in path or tp == "2":
|
||||
return "Dual-card (2× RTX 3090, TP=2)"
|
||||
return "Single-card (1× RTX 3090) — vLLM"
|
||||
|
||||
|
||||
def load() -> dict[str, Any]:
|
||||
if not TAG_DIR.is_dir():
|
||||
die(f"tag dir not found: {TAG_DIR}")
|
||||
internal = read_json(require_file("_internal.json"))
|
||||
if not isinstance(internal, dict):
|
||||
die(f"invalid JSON artifact: {TAG_DIR / '_internal.json'}")
|
||||
report = read_text(require_file("REPORT.md"))
|
||||
config_blob = first_container(read_json(require_file("container-config.json")))
|
||||
cfg = config_blob.get("Config") or {}
|
||||
labels = cfg.get("Labels") or {}
|
||||
cmd = cfg.get("Cmd") or []
|
||||
env = cfg.get("Env") or []
|
||||
name = str(config_blob.get("Name") or "").lstrip("/")
|
||||
served = flag(cmd, "--served-model-name")
|
||||
tp = flag(cmd, "--tensor-parallel-size")
|
||||
compose = rel(labels.get("com.docker.compose.project.config_files") or "")
|
||||
if not compose:
|
||||
compose = infer_compose_path(name, served, tp)
|
||||
rig = parse_rig(read_text(require_file("rig.txt")))
|
||||
tag = TAG_DIR.name
|
||||
spec = parse_jsonish(flag(cmd, "--speculative-config"))
|
||||
|
||||
return {
|
||||
"tag": tag,
|
||||
"report": report,
|
||||
"internal": internal,
|
||||
"container": {
|
||||
"name": name,
|
||||
"served": served,
|
||||
"tp": tp,
|
||||
"compose": compose,
|
||||
"kv": flag(cmd, "--kv-cache-dtype"),
|
||||
"max_ctx": flag(cmd, "--max-model-len"),
|
||||
"max_num_seqs": flag(cmd, "--max-num-seqs"),
|
||||
"mem_util": flag(cmd, "--gpu-memory-utilization"),
|
||||
"image": cfg.get("Image", ""),
|
||||
"spec": spec,
|
||||
"genesis": any(str(e).startswith("GENESIS_") for e in env),
|
||||
},
|
||||
"rig": rig,
|
||||
"date": short_date(tag, report),
|
||||
}
|
||||
|
||||
|
||||
def format_row(data: dict[str, Any]) -> str:
|
||||
c = data["container"]
|
||||
internal = data["internal"]
|
||||
bench = internal.get("bench") or {}
|
||||
narrative = bench.get("narrative") or {}
|
||||
code = bench.get("code") or {}
|
||||
mtp = bench.get("mtp") or {}
|
||||
quality = internal.get("quality") or {}
|
||||
aider = internal.get("aider") or {}
|
||||
soak = internal.get("soak") or {}
|
||||
rig = data["rig"]
|
||||
gpu_count = max(len(rig.get("gpus") or []), 1)
|
||||
section = section_name(c["compose"], c["served"], c["tp"], c["name"])
|
||||
verify = verify_summary(data["report"])
|
||||
compose = compose_display(c["compose"], c["served"])
|
||||
kv = kv_display(c["kv"], c["served"])
|
||||
max_ctx = fmt_ctx(c["max_ctx"])
|
||||
tps = f"**{fmt_tps(narrative.get('wall_tps_mean'))} / {fmt_tps(code.get('wall_tps_mean'))}**"
|
||||
peak = peak_vram(internal, gpu_count)
|
||||
spec_n = c["spec"].get("num_speculative_tokens")
|
||||
notes = [soak_note(soak), f"verify-stress {verify}"]
|
||||
if quality.get("total_passed") is not None:
|
||||
notes.append(f"quality {quality.get('total_passed')}/{quality.get('total_total')}")
|
||||
if aider.get("total_count"):
|
||||
notes.append(f"aider {aider.get('passed_count')}/{aider.get('total_count')}")
|
||||
if mtp.get("mean_accept_length") is not None:
|
||||
n_text = f" n={spec_n}" if spec_n is not None else ""
|
||||
notes.append(
|
||||
f"MTP{n_text} AL {float(mtp['mean_accept_length']):.2f}, "
|
||||
f"accept {float(mtp.get('avg_accept_rate', 0)):.1f}%"
|
||||
)
|
||||
if c.get("genesis"):
|
||||
notes.append("Genesis on")
|
||||
|
||||
note_cell = "; ".join(notes) + f". Report: `results/rebench/{data['tag']}/REPORT.md`."
|
||||
|
||||
if section == "Gemma 4 31B (community-experimental)":
|
||||
al = f"{float(mtp['mean_accept_length']):.2f}" if mtp.get("mean_accept_length") is not None else "—"
|
||||
per_pos = str(mtp.get("per_position") or "—")
|
||||
return (
|
||||
f"| `{compose}` | {rig_cell(rig)} | {kv} | {max_ctx} | {tps} | "
|
||||
f"{al} | {per_pos} | {peak} | {data['date']} | {note_cell} |"
|
||||
)
|
||||
|
||||
return (
|
||||
f"| `{compose}` | {rig_cell(rig)} | {kv} | {max_ctx} | {tps} | "
|
||||
f"{peak} | {data['date']} | {note_cell} |"
|
||||
)
|
||||
|
||||
|
||||
data = load()
|
||||
if MODE == "section":
|
||||
c = data["container"]
|
||||
print(section_name(c["compose"], c["served"], c["tp"], c["name"]))
|
||||
elif MODE == "row":
|
||||
print(format_row(data))
|
||||
elif MODE == "rig-short":
|
||||
print(data["tag"])
|
||||
else:
|
||||
die(f"unknown mode: {MODE}")
|
||||
PY
|
||||
}
|
||||
@@ -273,3 +273,6 @@ echo " verify-stress: tail -5 $OUT_DIR/verify-stress.log"
|
||||
echo " quality: grep '^Quality:' $OUT_DIR/quality-full.log"
|
||||
echo " soak: grep -E 'verdict|silent_empty|p50_decode' $OUT_DIR/soak.log"
|
||||
echo " aider: grep 'aider-polyglot-30' $OUT_DIR/aider-polyglot.log"
|
||||
echo
|
||||
echo "To submit your numbers (review then PR):"
|
||||
echo " bash scripts/submit-bench.sh --tag $TAG"
|
||||
|
||||
Executable
+312
@@ -0,0 +1,312 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Generate or submit a BENCHMARKS.md row from results/rebench/<tag>/.
|
||||
#
|
||||
# Usage:
|
||||
# bash scripts/submit-bench.sh --tag <tag>
|
||||
# bash scripts/submit-bench.sh --tag <tag> --auto-submit
|
||||
# bash scripts/submit-bench.sh --tag <tag> --auto-submit --as-pr
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
ROOT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
cd "$ROOT_DIR"
|
||||
|
||||
TAG=""
|
||||
AUTO_SUBMIT=0
|
||||
AS_PR=0
|
||||
SECTION_OVERRIDE=""
|
||||
|
||||
usage() {
|
||||
sed -n '2,18p' "$0" | sed 's/^# \{0,1\}//'
|
||||
exit 0
|
||||
}
|
||||
|
||||
die() {
|
||||
echo "[submit-bench] ERROR: $*" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
log() {
|
||||
echo "[submit-bench] $*"
|
||||
}
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--tag)
|
||||
[[ $# -ge 2 ]] || die "--tag requires a value"
|
||||
TAG="$2"
|
||||
shift 2
|
||||
;;
|
||||
--auto-submit)
|
||||
AUTO_SUBMIT=1
|
||||
shift
|
||||
;;
|
||||
--as-pr)
|
||||
AS_PR=1
|
||||
shift
|
||||
;;
|
||||
--section)
|
||||
[[ $# -ge 2 ]] || die "--section requires a value"
|
||||
SECTION_OVERRIDE="$2"
|
||||
shift 2
|
||||
;;
|
||||
-h|--help)
|
||||
usage
|
||||
;;
|
||||
*)
|
||||
die "unknown arg: $1"
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
[[ -n "$TAG" ]] || die "--tag <tag> is required"
|
||||
|
||||
TAG_DIR="results/rebench/${TAG}"
|
||||
[[ -d "$TAG_DIR" ]] || die "tag dir not found: ${TAG_DIR}"
|
||||
[[ -f "$TAG_DIR/REPORT.md" ]] || die "missing required artifact: ${TAG_DIR}/REPORT.md"
|
||||
[[ -f "$TAG_DIR/_internal.json" ]] || die "missing required artifact: ${TAG_DIR}/_internal.json"
|
||||
[[ -f "$TAG_DIR/container-config.json" ]] || die "missing required artifact: ${TAG_DIR}/container-config.json"
|
||||
[[ -f "$TAG_DIR/rig.txt" ]] || die "missing required artifact: ${TAG_DIR}/rig.txt"
|
||||
|
||||
# shellcheck source=lib/bench-row-formatter.sh
|
||||
source "$ROOT_DIR/scripts/lib/bench-row-formatter.sh"
|
||||
|
||||
github_user_for_row() {
|
||||
if [[ -n "${BENCH_ROW_GITHUB_USER:-}" ]]; then
|
||||
printf '%s' "${BENCH_ROW_GITHUB_USER#@}"
|
||||
return 0
|
||||
fi
|
||||
if [[ "${GH_MOCK:-0}" == "1" ]]; then
|
||||
printf '%s' "${GH_MOCK_USER:-mock-user}"
|
||||
return 0
|
||||
fi
|
||||
if command -v gh >/dev/null 2>&1 && gh auth status >/dev/null 2>&1; then
|
||||
gh api user --jq .login 2>/dev/null || true
|
||||
fi
|
||||
}
|
||||
|
||||
if [[ "$AUTO_SUBMIT" -eq 1 ]]; then
|
||||
GH_USER="$(github_user_for_row)"
|
||||
[[ -n "$GH_USER" ]] || die "not authed with gh. Run: gh auth login"
|
||||
export BENCH_ROW_GITHUB_USER="$GH_USER"
|
||||
fi
|
||||
|
||||
ROW="$(bench_row_format "$TAG_DIR")"
|
||||
SECTION="${SECTION_OVERRIDE:-$(bench_row_section "$TAG_DIR")}"
|
||||
OUTPUT="$TAG_DIR/BENCHMARKS-row.md"
|
||||
printf '%s\n' "$ROW" > "$OUTPUT"
|
||||
|
||||
log "Generated BENCHMARKS row for section: ${SECTION}"
|
||||
log "Wrote: ${OUTPUT}"
|
||||
echo
|
||||
|
||||
valid_sections() {
|
||||
rg -n '^(##|###) ' BENCHMARKS.md | sed 's/^/[submit-bench] /' >&2 || true
|
||||
}
|
||||
|
||||
write_pr_body() {
|
||||
local body_file="$1"
|
||||
local row="$2"
|
||||
local tag="$3"
|
||||
local template=".github/PULL_REQUEST_TEMPLATE/bench-row.md"
|
||||
|
||||
if [[ -f "$template" ]]; then
|
||||
python3 - "$template" "$body_file" "$tag" "$row" <<'PY'
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
template, body_file, tag, row = sys.argv[1:5]
|
||||
text = Path(template).read_text()
|
||||
text = text.replace("<TAG>", tag)
|
||||
text = text.replace("<!-- The generated BENCHMARKS.md row goes here -->", row)
|
||||
text = text.replace(
|
||||
"<!-- Output of `bash scripts/report.sh` (redacted) -->",
|
||||
f"See `results/rebench/{tag}/rig.txt`.",
|
||||
)
|
||||
Path(body_file).write_text(text)
|
||||
PY
|
||||
else
|
||||
{
|
||||
echo "## Rig bench submission"
|
||||
echo
|
||||
echo "### New row"
|
||||
echo
|
||||
echo "$row"
|
||||
echo
|
||||
echo "### Full results"
|
||||
echo
|
||||
echo "See \`results/rebench/${tag}/REPORT.md\`."
|
||||
} > "$body_file"
|
||||
fi
|
||||
}
|
||||
|
||||
write_issue_body() {
|
||||
local body_file="$1"
|
||||
local row="$2"
|
||||
local tag="$3"
|
||||
local section="$4"
|
||||
|
||||
# The repo's numbers-from-your-rig issue template is a structured YAML form
|
||||
# with required textarea/dropdown fields. `gh issue create --template` opens
|
||||
# that interactive form shape, which is not useful once submit-bench has
|
||||
# already generated the structured report. Use a direct markdown body instead.
|
||||
{
|
||||
echo "**Compose / section**: \`${section}\`"
|
||||
echo
|
||||
echo "**Rig**:"
|
||||
echo
|
||||
echo '```text'
|
||||
cat "${TAG_DIR}/rig.txt"
|
||||
echo '```'
|
||||
echo
|
||||
echo "**Proposed BENCHMARKS.md row**:"
|
||||
echo
|
||||
echo "$row"
|
||||
echo
|
||||
echo "**Full report**: \`results/rebench/${tag}/REPORT.md\`"
|
||||
echo
|
||||
echo "**Generated row file**: \`results/rebench/${tag}/BENCHMARKS-row.md\`"
|
||||
} > "$body_file"
|
||||
}
|
||||
|
||||
insert_row() {
|
||||
local section="$1"
|
||||
local row="$2"
|
||||
python3 - "$section" "$row" <<'PY'
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
section = sys.argv[1]
|
||||
row = sys.argv[2]
|
||||
path = Path("BENCHMARKS.md")
|
||||
lines = path.read_text().splitlines()
|
||||
|
||||
heading_idx = None
|
||||
for i, line in enumerate(lines):
|
||||
if line.strip() in {f"## {section}", f"### {section}"}:
|
||||
heading_idx = i
|
||||
break
|
||||
if heading_idx is None:
|
||||
print(f"[submit-bench] ERROR: section not found in BENCHMARKS.md: {section}", file=sys.stderr)
|
||||
raise SystemExit(1)
|
||||
|
||||
table_start = None
|
||||
for i in range(heading_idx + 1, len(lines)):
|
||||
if lines[i].startswith("#"):
|
||||
break
|
||||
if lines[i].startswith("|"):
|
||||
table_start = i
|
||||
break
|
||||
if table_start is None:
|
||||
print(f"[submit-bench] ERROR: no markdown table found below section: {section}", file=sys.stderr)
|
||||
raise SystemExit(1)
|
||||
|
||||
insert_at = table_start
|
||||
for i in range(table_start, len(lines)):
|
||||
line = lines[i]
|
||||
if line.startswith("#"):
|
||||
break
|
||||
if line.startswith("|"):
|
||||
insert_at = i + 1
|
||||
continue
|
||||
if insert_at > table_start:
|
||||
break
|
||||
|
||||
lines.insert(insert_at, row)
|
||||
path.write_text("\n".join(lines) + "\n")
|
||||
PY
|
||||
}
|
||||
|
||||
if [[ "$AUTO_SUBMIT" -ne 1 ]]; then
|
||||
cat <<EOF
|
||||
Inspect at ${OUTPUT}. Three ways to land it (recommended order):
|
||||
|
||||
1. Issue + maintainer integrates (preferred — vetting before merge):
|
||||
bash scripts/submit-bench.sh --tag ${TAG} --auto-submit
|
||||
(opens an issue via \`gh issue create\`)
|
||||
Or, no-gh-needed:
|
||||
https://github.com/noonghunna/club-3090/issues/new?template=numbers-from-your-rig.yml
|
||||
— paste the contents of ${OUTPUT} + ${TAG_DIR}/rig.txt into the body
|
||||
|
||||
2. Direct PR (advanced — for contributors who know the matrix structure):
|
||||
bash scripts/submit-bench.sh --tag ${TAG} --auto-submit --as-pr
|
||||
Note: matrix is hand-curated; direct PRs may get redirected to an
|
||||
issue thread for context-gathering before merge.
|
||||
|
||||
3. Manual edit (zero tools):
|
||||
Paste the row from ${OUTPUT} into BENCHMARKS.md via the GitHub web editor.
|
||||
EOF
|
||||
exit 0
|
||||
fi
|
||||
|
||||
if [[ -n "$SECTION_OVERRIDE" ]]; then
|
||||
if ! rg -q "^(##|###) ${SECTION_OVERRIDE//\//\\/}$" BENCHMARKS.md; then
|
||||
echo "[submit-bench] Known sections:" >&2
|
||||
valid_sections
|
||||
die "section override not found: ${SECTION_OVERRIDE}"
|
||||
fi
|
||||
fi
|
||||
|
||||
PR_TITLE="bench(matrix): @${BENCH_ROW_GITHUB_USER} $(bench_row_rig_shortname "$TAG_DIR")"
|
||||
ISSUE_TITLE="[bench] @${BENCH_ROW_GITHUB_USER} $(bench_row_rig_shortname "$TAG_DIR")"
|
||||
BRANCH_USER="$(printf '%s' "${BENCH_ROW_GITHUB_USER}" | tr -cd '[:alnum:]_.-')"
|
||||
BRANCH_TAG="$(printf '%s' "${TAG}" | tr -cd '[:alnum:]_.-')"
|
||||
BRANCH="bench/${BRANCH_USER}-${BRANCH_TAG}"
|
||||
if [[ "$AS_PR" -eq 1 ]]; then
|
||||
BODY_FILE="$TAG_DIR/PR-body.md"
|
||||
write_pr_body "$BODY_FILE" "$ROW" "$TAG"
|
||||
else
|
||||
BODY_FILE="$TAG_DIR/ISSUE-body.md"
|
||||
write_issue_body "$BODY_FILE" "$ROW" "$TAG" "$SECTION"
|
||||
fi
|
||||
|
||||
if [[ "${GH_MOCK:-0}" == "1" ]]; then
|
||||
MOCK_LOG="$TAG_DIR/auto-submit-mock.log"
|
||||
if [[ "$AS_PR" -eq 1 ]]; then
|
||||
{
|
||||
echo "git switch -c ${BRANCH}"
|
||||
echo "insert BENCHMARKS.md row under: ${SECTION}"
|
||||
echo "git commit -m ${PR_TITLE}"
|
||||
echo "git push -u origin ${BRANCH}"
|
||||
echo "gh pr create --title ${PR_TITLE} --body-file ${BODY_FILE}"
|
||||
} > "$MOCK_LOG"
|
||||
log "GH_MOCK=1 — wrote mocked PR auto-submit commands: ${MOCK_LOG}"
|
||||
log "PR title: ${PR_TITLE}"
|
||||
else
|
||||
{
|
||||
echo "gh issue create --title ${ISSUE_TITLE} --body-file ${BODY_FILE} --label bench-contribution"
|
||||
} > "$MOCK_LOG"
|
||||
log "GH_MOCK=1 — wrote mocked issue auto-submit command: ${MOCK_LOG}"
|
||||
log "Issue title: ${ISSUE_TITLE}"
|
||||
fi
|
||||
exit 0
|
||||
fi
|
||||
|
||||
command -v gh >/dev/null 2>&1 || die "'gh' not found. Install GitHub CLI or submit manually."
|
||||
gh auth status >/dev/null 2>&1 || die "not authed with gh. Run: gh auth login"
|
||||
|
||||
if [[ "$AS_PR" -ne 1 ]]; then
|
||||
ISSUE_URL="$(gh issue create --title "$ISSUE_TITLE" --body-file "$BODY_FILE" --label bench-contribution)"
|
||||
log "Opened issue: ${ISSUE_URL}"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
if ! git diff --quiet -- BENCHMARKS.md; then
|
||||
die "BENCHMARKS.md already has local edits; commit/stash them before --auto-submit"
|
||||
fi
|
||||
|
||||
git fetch origin master >/dev/null 2>&1 || log "WARN: git fetch origin master failed; continuing from current branch"
|
||||
if git show-ref --verify --quiet refs/remotes/origin/master; then
|
||||
git switch -c "$BRANCH" origin/master
|
||||
else
|
||||
git switch -c "$BRANCH"
|
||||
fi
|
||||
insert_row "$SECTION" "$ROW"
|
||||
git add BENCHMARKS.md
|
||||
git commit -m "$PR_TITLE"
|
||||
git push -u origin "$BRANCH"
|
||||
PR_URL="$(gh pr create --title "$PR_TITLE" --body-file "$BODY_FILE")"
|
||||
log "Opened PR: ${PR_URL}"
|
||||
Executable
+111
@@ -0,0 +1,111 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
ROOT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../.." && pwd)"
|
||||
cd "$ROOT_DIR"
|
||||
|
||||
# shellcheck source=../lib/bench-row-formatter.sh
|
||||
source "$ROOT_DIR/scripts/lib/bench-row-formatter.sh"
|
||||
|
||||
assert_contains() {
|
||||
local haystack="$1"
|
||||
local needle="$2"
|
||||
if [[ "$haystack" != *"$needle"* ]]; then
|
||||
echo "ASSERTION FAILED: expected output to contain: $needle" >&2
|
||||
echo "--- output ---" >&2
|
||||
echo "$haystack" >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
assert_columns() {
|
||||
local row="$1"
|
||||
local expected="$2"
|
||||
local cols
|
||||
cols="$(awk -F'|' '{print NF - 2}' <<< "$row")"
|
||||
if [[ "$cols" != "$expected" ]]; then
|
||||
echo "ASSERTION FAILED: expected $expected columns, got $cols" >&2
|
||||
echo "$row" >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
fixtures=()
|
||||
while IFS= read -r fixture; do
|
||||
fixtures+=("$fixture")
|
||||
done < <(bench_row_fixtures)
|
||||
|
||||
if [[ "${#fixtures[@]}" -lt 6 ]]; then
|
||||
echo "ASSERTION FAILED: expected at least 6 submit-bench fixtures, found ${#fixtures[@]}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
for dir in "${fixtures[@]}"; do
|
||||
section="$(bench_row_section "$dir")"
|
||||
row="$(bench_row_format "$dir")"
|
||||
[[ "$row" == \|*\| ]] || {
|
||||
echo "ASSERTION FAILED: row is not a markdown table row for $dir" >&2
|
||||
echo "$row" >&2
|
||||
exit 1
|
||||
}
|
||||
assert_contains "$row" "Report: \`results/rebench/${dir##*/}/REPORT.md\`"
|
||||
if [[ "$section" == "Gemma 4 31B (community-experimental)" ]]; then
|
||||
assert_columns "$row" 10
|
||||
else
|
||||
assert_columns "$row" 8
|
||||
fi
|
||||
done
|
||||
|
||||
tag="qwen-int8-pth-n4-2026-05-10"
|
||||
rm -f "results/rebench/${tag}/BENCHMARKS-row.md" \
|
||||
"results/rebench/${tag}/PR-body.md" \
|
||||
"results/rebench/${tag}/ISSUE-body.md" \
|
||||
"results/rebench/${tag}/auto-submit-mock.log"
|
||||
|
||||
out="$(bash scripts/submit-bench.sh --tag "$tag")"
|
||||
assert_contains "$out" "Generated BENCHMARKS row for section: Dual-card (2× RTX 3090, TP=2)"
|
||||
assert_contains "$out" "Wrote: results/rebench/${tag}/BENCHMARKS-row.md"
|
||||
assert_contains "$out" "1. Issue + maintainer integrates"
|
||||
assert_contains "$out" "2. Direct PR"
|
||||
assert_contains "$out" "3. Manual edit"
|
||||
test -s "results/rebench/${tag}/BENCHMARKS-row.md"
|
||||
|
||||
if out="$(bash scripts/submit-bench.sh --tag does-not-exist 2>&1)"; then
|
||||
echo "ASSERTION FAILED: missing tag unexpectedly succeeded" >&2
|
||||
exit 1
|
||||
fi
|
||||
assert_contains "$out" "tag dir not found: results/rebench/does-not-exist"
|
||||
|
||||
out="$(GH_MOCK=1 GH_MOCK_USER=octocat bash scripts/submit-bench.sh --tag "$tag" --auto-submit)"
|
||||
assert_contains "$out" "Issue title: [bench] @octocat ${tag}"
|
||||
test -s "results/rebench/${tag}/auto-submit-mock.log"
|
||||
assert_contains "$(cat "results/rebench/${tag}/auto-submit-mock.log")" "gh issue create --title [bench] @octocat ${tag}"
|
||||
test -s "results/rebench/${tag}/ISSUE-body.md"
|
||||
assert_contains "$(cat "results/rebench/${tag}/ISSUE-body.md")" "results/rebench/${tag}/REPORT.md"
|
||||
assert_contains "$(cat "results/rebench/${tag}/ISSUE-body.md")" "Proposed BENCHMARKS.md row"
|
||||
|
||||
out="$(GH_MOCK=1 GH_MOCK_USER=octocat bash scripts/submit-bench.sh --tag "$tag" --auto-submit --as-pr)"
|
||||
assert_contains "$out" "PR title: bench(matrix): @octocat ${tag}"
|
||||
test -s "results/rebench/${tag}/PR-body.md"
|
||||
assert_contains "$(cat "results/rebench/${tag}/auto-submit-mock.log")" "gh pr create --title bench(matrix): @octocat ${tag}"
|
||||
assert_contains "$(cat "results/rebench/${tag}/PR-body.md")" "results/rebench/${tag}/REPORT.md"
|
||||
|
||||
tmp_bin="$(mktemp -d)"
|
||||
trap 'rm -rf "$tmp_bin"; rm -f "results/rebench/${tag}/BENCHMARKS-row.md" "results/rebench/${tag}/PR-body.md" "results/rebench/${tag}/ISSUE-body.md" "results/rebench/${tag}/auto-submit-mock.log"' EXIT
|
||||
cat > "${tmp_bin}/gh" <<'MOCK_GH'
|
||||
#!/usr/bin/env bash
|
||||
if [[ "$1" == "auth" && "$2" == "status" ]]; then
|
||||
exit 1
|
||||
fi
|
||||
echo "unexpected gh call: $*" >&2
|
||||
exit 2
|
||||
MOCK_GH
|
||||
chmod +x "${tmp_bin}/gh"
|
||||
|
||||
if out="$(PATH="${tmp_bin}:${PATH}" bash scripts/submit-bench.sh --tag "$tag" --auto-submit 2>&1)"; then
|
||||
echo "ASSERTION FAILED: unauthenticated gh path unexpectedly succeeded" >&2
|
||||
exit 1
|
||||
fi
|
||||
assert_contains "$out" "not authed with gh. Run: gh auth login"
|
||||
|
||||
echo "test-submit-bench: ok"
|
||||
Reference in New Issue
Block a user