diff --git a/.github/workflows/benchmark.yml b/.github/workflows/benchmark.yml
new file mode 100644
index 0000000..218a80c
--- /dev/null
+++ b/.github/workflows/benchmark.yml
@@ -0,0 +1,492 @@
+name: benchmark
+
+# Runs the browser-agent suites against a local llama-server on the RTX
+# worker, keeps every run's results, and updates the GitHub Pages dashboard.
+#
+# Manual trigger only. This repo is public: a push or pull_request trigger
+# would let fork code reach the worker. Inputs show on the run page, so no
+# input may carry a secret -- secrets come from the `rtx-benchmark`
+# environment. An empty input falls back to the environment variable of the
+# same meaning, then to the built-in default.
+#
+# Setup (environment, secrets, variables, Pages): docs/benchmark-action.md.
+on:
+ workflow_dispatch:
+ inputs:
+ worker_host:
+ description: 'Worker SSH host on the tailnet (empty: vars.WORKER_HOST)'
+ required: false
+ type: string
+ default: ''
+ ssh_user:
+ description: 'SSH user on the worker (empty: vars.WORKER_SSH_USER)'
+ required: false
+ type: string
+ default: ''
+ model_preset:
+ description: 'Model preset'
+ required: false
+ type: choice
+ options:
+ - qwen3.5-9b
+ - qwen3.5-4b
+ default: qwen3.5-9b
+ ctx:
+ description: 'Context size for the single-slot phase'
+ required: false
+ type: string
+ default: '32768'
+ suites:
+ description: 'Space-separated: fixture long_context live_web concurrency'
+ required: false
+ type: string
+ default: 'fixture long_context live_web'
+ repetitions:
+ description: 'Repetitions of fixture tasks and live_web runs (capped to fit the job time)'
+ required: false
+ type: string
+ default: '1'
+ restore_production:
+ description: 'Stop the test server and restart the live container at the end'
+ required: false
+ type: boolean
+ default: true
+
+# One run at a time: the worker has one GPU, and gh-pages has one history.
+concurrency:
+ group: rtx-benchmark
+ cancel-in-progress: false
+
+permissions:
+ contents: read
+
+jobs:
+ benchmark:
+ runs-on: ubuntu-latest
+ environment: rtx-benchmark
+ timeout-minutes: 240
+ env:
+ # Inputs reach shell steps only through env, never by ${{ }} inside a
+ # script, so an input value cannot inject shell code. The worker host
+ # and SSH user are not here: only "Resolve settings" sees them, and it
+ # masks them before any other step can print them.
+ IN_MODEL_PRESET: ${{ inputs.model_preset || 'qwen3.5-9b' }}
+ IN_CTX: ${{ inputs.ctx || '32768' }}
+ IN_SUITES: ${{ inputs.suites || 'fixture long_context live_web' }}
+ IN_REPETITIONS: ${{ inputs.repetitions || '1' }}
+ LIVE_CONTAINER: ${{ vars.LIVE_CONTAINER || 'bonsai-llama-1' }}
+ LIVE_HEALTH_URL: ${{ vars.LIVE_HEALTH_URL || 'http://127.0.0.1:8080/health' }}
+ LIVE_ENV_FILE: ${{ vars.LIVE_ENV_FILE }}
+ TEST_IMAGE: ${{ vars.TEST_IMAGE || 'qcm-rtx-validation:922be44' }}
+ MODELS_DIR: ${{ vars.MODELS_DIR }}
+ TEMPLATE_FILE: ${{ vars.TEMPLATE_FILE }}
+ # GPU name for meta.json (run_suites.py reads it); empty means unknown.
+ BENCH_GPU_NAME: ${{ vars.BENCH_GPU_NAME }}
+ TEST_CONTAINER: qcm-bench
+ TEST_PORT: '18080'
+ LOCAL_PORT: '18080'
+ # The 2-slot server for the concurrency suite: 24,576 tokens per slot.
+ PHASE2_CTX: '49152'
+ PHASE2_PARALLEL: '2'
+ # Same Playwright as @playwright/mcp@0.0.82 in start-playwright-mcp.sh.
+ PLAYWRIGHT_VERSION: 1.64.0-alpha-1789764292000
+ # Secrets cannot appear in a step `if:`; these booleans can.
+ HAS_TS_OAUTH: ${{ secrets.TS_OAUTH_CLIENT_ID != '' && secrets.TS_OAUTH_SECRET != '' }}
+ steps:
+ - uses: actions/checkout@v7
+ with:
+ # This job pushes nothing; keep the token out of .git/config.
+ persist-credentials: false
+
+ - uses: actions/setup-python@v7
+ with:
+ python-version: '3.11'
+
+ - uses: actions/setup-node@v7
+ with:
+ node-version: '22'
+
+ - name: Resolve settings
+ env:
+ # A secret is masked in every log line; a variable is plain text
+ # until the add-mask below. Prefer the secrets (see the docs).
+ IN_WORKER_HOST: ${{ inputs.worker_host || secrets.WORKER_HOST || vars.WORKER_HOST }}
+ IN_SSH_USER: ${{ inputs.ssh_user || secrets.WORKER_SSH_USER || vars.WORKER_SSH_USER }}
+ run: |
+ set -euo pipefail
+ # First, before anything can print them: this repo's logs are
+ # public, and later steps (the tailnet ping, ssh) name the worker.
+ [[ -z "$IN_WORKER_HOST" ]] || echo "::add-mask::$IN_WORKER_HOST"
+ [[ -z "$IN_SSH_USER" ]] || echo "::add-mask::$IN_SSH_USER"
+ [[ -n "$IN_WORKER_HOST" ]] || { echo "::error::worker_host input or WORKER_HOST variable is required"; exit 1; }
+ [[ -n "$IN_SSH_USER" ]] || { echo "::error::ssh_user input or WORKER_SSH_USER variable is required"; exit 1; }
+ [[ -n "$MODELS_DIR" ]] || { echo "::error::MODELS_DIR variable is required"; exit 1; }
+ [[ -n "$TEMPLATE_FILE" ]] || { echo "::error::TEMPLATE_FILE variable is required"; exit 1; }
+ [[ "$IN_WORKER_HOST" =~ ^[A-Za-z0-9._:-]+$ ]] || { echo "::error::bad worker_host"; exit 1; }
+ [[ "$IN_SSH_USER" =~ ^[a-z_][a-z0-9_-]*$ ]] || { echo "::error::bad ssh_user"; exit 1; }
+ [[ "$IN_CTX" =~ ^[0-9]+$ ]] || { echo "::error::ctx must be an integer"; exit 1; }
+ [[ "$IN_REPETITIONS" =~ ^[0-9]+$ ]] || { echo "::error::repetitions must be an integer"; exit 1; }
+
+ # The one place a preset maps to a model. Both use the Qwen3.5
+ # template at TEMPLATE_FILE on the worker.
+ case "$IN_MODEL_PRESET" in
+ qwen3.5-9b)
+ model_file=/models/Qwen3.5-9B-GGUF/Qwen3.5-9B-Q4_K_M.gguf
+ model_alias=qwen3.5-9b-q4_k_m ;;
+ qwen3.5-4b)
+ model_file=/models/Qwen3.5-4B-GGUF/Qwen3.5-4B-Q4_K_M.gguf
+ model_alias=qwen3.5-4b-q4_k_m ;;
+ *) echo "::error::unknown model_preset: $IN_MODEL_PRESET"; exit 1 ;;
+ esac
+
+ # Split the suites: concurrency needs a 2-slot server, the rest 1.
+ # set -f: split the list on spaces without expanding globs.
+ set -f
+ phase1=() ; phase2=""
+ for s in $IN_SUITES; do
+ case "$s" in
+ fixture|long_context|live_web) phase1+=("$s") ;;
+ concurrency) phase2=concurrency ;;
+ *) echo "::error::unknown suite: $s"; exit 1 ;;
+ esac
+ done
+ set +f
+ [[ ${#phase1[@]} -gt 0 || -n "$phase2" ]] || { echo "::error::no suites selected"; exit 1; }
+
+ # Step caps from run_suites.py's own job timeouts, so the restore
+ # always fits in the job's 240 minutes. Too many repetitions would
+ # not fit: refuse the run instead of cutting the restore short.
+ budget="$(python3 - "${phase1[*]:-}" "$phase2" "$IN_REPETITIONS" <<'EOF'
+ import sys
+ sys.path.insert(0, "tests/bench")
+ import run_suites as r
+ p1, p2, reps = sys.argv[1].split(), sys.argv[2].split(), int(sys.argv[3])
+ print(r.step_minutes(p1, reps), r.step_minutes(p2, reps), r.max_repetitions(p1, p2))
+ EOF
+ )"
+ read -r phase1_min phase2_min max_reps <<<"$budget"
+ (( 10#$IN_REPETITIONS >= 1 )) || { echo "::error::repetitions must be at least 1"; exit 1; }
+ if (( 10#$IN_REPETITIONS > max_reps )); then
+ echo "::error::repetitions $IN_REPETITIONS does not fit the 240-minute job with these suites; the most is $max_reps"
+ exit 1
+ fi
+
+ {
+ echo "WORKER_HOST=$IN_WORKER_HOST"
+ echo "WORKER_SSH=$IN_SSH_USER@$IN_WORKER_HOST"
+ echo "MODEL_FILE=$model_file"
+ echo "MODEL_ALIAS=$model_alias"
+ echo "PHASE1_SUITES=${phase1[*]}"
+ echo "PHASE2_SUITES=$phase2"
+ # Never 0: a skipped step's cap must still be a valid number.
+ echo "PHASE1_TIMEOUT_MIN=$(( phase1_min > 0 ? phase1_min : 1 ))"
+ echo "PHASE2_TIMEOUT_MIN=$(( phase2_min > 0 ? phase2_min : 1 ))"
+ } >> "$GITHUB_ENV"
+ echo "model: $model_alias ctx: $IN_CTX repetitions: $IN_REPETITIONS (most: $max_reps)"
+ echo "phase 1: ${phase1[*]:-none} (cap ${phase1_min} min) phase 2: ${phase2:-none} (cap ${phase2_min} min)"
+
+ # ubuntu-latest (24.04) ships Google Chrome (runner-images readme),
+ # which is what the Playwright MCP uses (--browser chrome). Install it
+ # only if the image ever lacks it.
+ - name: Ensure Google Chrome
+ run: |
+ set -euo pipefail
+ if command -v google-chrome >/dev/null 2>&1; then
+ google-chrome --version
+ else
+ npx -y "playwright@${PLAYWRIGHT_VERSION}" install --with-deps chrome
+ fi
+
+ - name: Join tailnet (OAuth client)
+ if: env.HAS_TS_OAUTH == 'true'
+ uses: tailscale/github-action@v4
+ with:
+ oauth-client-id: ${{ secrets.TS_OAUTH_CLIENT_ID }}
+ oauth-secret: ${{ secrets.TS_OAUTH_SECRET }}
+ tags: tag:ci
+ hostname: gh-bench-${{ github.run_id }}
+ ping: ${{ env.WORKER_HOST }}
+
+ - name: Join tailnet (auth key)
+ if: env.HAS_TS_OAUTH != 'true'
+ uses: tailscale/github-action@v4
+ with:
+ authkey: ${{ secrets.TS_AUTHKEY }}
+ hostname: gh-bench-${{ github.run_id }}
+ ping: ${{ env.WORKER_HOST }}
+
+ - name: Configure SSH
+ env:
+ SSH_KEY: ${{ secrets.WORKER_SSH_KEY }}
+ KNOWN_HOSTS: ${{ secrets.WORKER_KNOWN_HOSTS }}
+ run: |
+ set -euo pipefail
+ umask 077
+ mkdir -p ~/.ssh
+ printf '%s\n' "$SSH_KEY" > ~/.ssh/bench_key
+ printf '%s\n' "$KNOWN_HOSTS" > ~/.ssh/known_hosts
+ cat > ~/.ssh/config <<'EOF'
+ Host *
+ IdentityFile ~/.ssh/bench_key
+ IdentitiesOnly yes
+ UserKnownHostsFile ~/.ssh/known_hosts
+ StrictHostKeyChecking yes
+ BatchMode yes
+ ServerAliveInterval 30
+ ServerAliveCountMax 4
+ # One shared connection for worker.sh calls: --vram-cmd polls
+ # every 2 s, and down should not pay a fresh handshake per try.
+ ControlMaster auto
+ ControlPath ~/.ssh/cm-%C
+ ControlPersist 10m
+ EOF
+ ssh "$WORKER_SSH" true
+ echo "SSH to the worker works"
+
+ - name: Start test server (1 slot)
+ if: env.PHASE1_SUITES != ''
+ timeout-minutes: 10
+ env:
+ CTX: ${{ env.IN_CTX }}
+ PARALLEL: '1'
+ run: tests/bench/worker.sh up
+
+ - name: Open tunnel to the test server
+ run: |
+ set -euo pipefail
+ # -E and the redirects detach the background ssh from the step's
+ # output pipes, so the step ends when the tunnel is up.
+ log="$RUNNER_TEMP/tunnel.log"
+ # Its own connection, not the shared master: the tunnel must own
+ # its forward and live until the job ends.
+ if ! ssh -f -N -E "$log" -o ExitOnForwardFailure=yes \
+ -o ControlMaster=no -o ControlPath=none \
+ -L "${LOCAL_PORT}:127.0.0.1:${TEST_PORT}" "$WORKER_SSH" \
+ /dev/null; then
+ cat "$log" >&2
+ exit 1
+ fi
+ echo "tunnel: 127.0.0.1:${LOCAL_PORT} -> worker 127.0.0.1:${TEST_PORT}"
+
+ - name: Run single-slot suites
+ if: env.PHASE1_SUITES != ''
+ timeout-minutes: ${{ fromJSON(env.PHASE1_TIMEOUT_MIN) }}
+ run: |
+ set -euo pipefail
+ # shellcheck disable=SC2086 # PHASE1_SUITES is a word list on purpose
+ python3 tests/bench/run_suites.py \
+ --base-url "http://127.0.0.1:${LOCAL_PORT}" \
+ --model "$MODEL_ALIAS" \
+ --ctx "$IN_CTX" \
+ --parallel 1 \
+ --suites $PHASE1_SUITES \
+ --repetitions "$IN_REPETITIONS" \
+ --out-dir bench-out/raw \
+ --vram-cmd "tests/bench/worker.sh vram"
+
+ # `down` is not called between phases: the live container stays
+ # stopped, and `up` replaces the 1-slot test container.
+ # Phase 2 may fail without failing the job: the merge step records the
+ # failure under "concurrency" in errors.json, and phase 1's results
+ # still get summarized and published.
+ - name: Restart test server (2 slots)
+ id: up2
+ if: env.PHASE2_SUITES != ''
+ continue-on-error: true
+ timeout-minutes: 10
+ env:
+ CTX: ${{ env.PHASE2_CTX }}
+ PARALLEL: ${{ env.PHASE2_PARALLEL }}
+ run: tests/bench/worker.sh up
+
+ - name: Run concurrency suite
+ id: concurrency
+ if: env.PHASE2_SUITES != '' && steps.up2.outcome == 'success'
+ continue-on-error: true
+ timeout-minutes: ${{ fromJSON(env.PHASE2_TIMEOUT_MIN) }}
+ run: |
+ set -euo pipefail
+ python3 tests/bench/run_suites.py \
+ --base-url "http://127.0.0.1:${LOCAL_PORT}" \
+ --model "$MODEL_ALIAS" \
+ --ctx "$PHASE2_CTX" \
+ --parallel "$PHASE2_PARALLEL" \
+ --suites concurrency \
+ --repetitions "$IN_REPETITIONS" \
+ --out-dir bench-out/raw-concurrency \
+ --vram-cmd "tests/bench/worker.sh vram"
+
+ # collect.py reads one raw directory. Fold the 2-slot phase into it:
+ # its report files move over, errors merge, and meta.json lists every
+ # suite. meta.json keeps the 1-slot phase's ctx/parallel when both ran;
+ # suite_settings records the server each suite really used, so the
+ # dashboard plots concurrency VRAM at the 2-slot ctx.
+ - name: Merge phase reports
+ if: always() && env.PHASE2_SUITES != '' && env.MODEL_ALIAS != ''
+ env:
+ UP2_OUTCOME: ${{ steps.up2.outcome }}
+ RUN2_OUTCOME: ${{ steps.concurrency.outcome }}
+ run: |
+ set -euo pipefail
+ mkdir -p bench-out/raw
+ python3 - <<'EOF'
+ import datetime, json, os, pathlib, shutil
+ src, dst = pathlib.Path("bench-out/raw-concurrency"), pathlib.Path("bench-out/raw")
+ def load(p):
+ try:
+ return json.loads(p.read_text())
+ except (OSError, ValueError):
+ return {}
+ src_meta = load(src / "meta.json")
+ if src.is_dir():
+ for path in src.iterdir():
+ if path.name in ("meta.json", "errors.json"):
+ continue
+ target = dst / path.name
+ if path.is_dir():
+ shutil.copytree(path, target, dirs_exist_ok=True)
+ else:
+ shutil.copy2(path, target)
+ errors = {**load(dst / "errors.json"), **load(src / "errors.json")}
+ if os.environ["UP2_OUTCOME"] == "failure":
+ errors["concurrency"] = "the 2-slot test server did not start (worker.sh up failed)"
+ elif os.environ["UP2_OUTCOME"] != "success":
+ errors["concurrency"] = "not run: an earlier step failed"
+ elif os.environ["RUN2_OUTCOME"] == "failure" and "concurrency" not in errors:
+ errors["concurrency"] = "the concurrency step failed or hit its time cap"
+ if errors:
+ (dst / "errors.json").write_text(json.dumps(errors, indent=2) + "\n")
+ ctx, parallel = int(os.environ["PHASE2_CTX"]), int(os.environ["PHASE2_PARALLEL"])
+ # No phase-1 meta (concurrency only): start from phase 2's, or a
+ # minimal one when phase 2 never wrote any, so the failure still
+ # reaches the dashboard.
+ meta = load(dst / "meta.json") or src_meta or {
+ "model": os.environ["MODEL_ALIAS"], "ctx": ctx, "parallel": parallel,
+ "gpu": os.environ.get("BENCH_GPU_NAME") or None,
+ "commit": os.environ.get("GITHUB_SHA", "")[:7] or None,
+ "run_id": os.environ.get("GITHUB_RUN_ID"),
+ "created_utc": datetime.datetime.now(datetime.timezone.utc).isoformat(),
+ "suites": []}
+ meta["suites"] = list(dict.fromkeys(meta.get("suites", []) + src_meta.get("suites", []) + ["concurrency"]))
+ settings = meta.setdefault("suite_settings", {})
+ settings.update(src_meta.get("suite_settings", {}))
+ settings["concurrency"] = {"ctx": ctx, "parallel": parallel}
+ (dst / "meta.json").write_text(json.dumps(meta, indent=2) + "\n")
+ EOF
+
+ - name: Summarize
+ if: always()
+ run: |
+ set -euo pipefail
+ if [[ ! -f bench-out/raw/meta.json ]]; then
+ echo "::warning::no raw reports to summarize"
+ exit 0
+ fi
+ python3 tests/bench/collect.py summarize bench-out/raw --out bench-out/summary.json
+ cat bench-out/summary.json
+
+ - name: Upload raw reports
+ if: always()
+ uses: actions/upload-artifact@v7
+ with:
+ # Holds the saved live_web images too (live-images/).
+ name: bench-raw-${{ github.run_id }}
+ path: bench-out/raw
+ if-no-files-found: ignore
+ retention-days: 90
+
+ - name: Upload summary
+ if: always()
+ uses: actions/upload-artifact@v7
+ with:
+ name: bench-summary
+ path: bench-out/summary.json
+ if-no-files-found: ignore
+ retention-days: 90
+
+ - name: Close tunnel
+ if: always()
+ run: pkill -f "${LOCAL_PORT}:127.0.0.1:${TEST_PORT}" || true
+
+ # Always last. A restore failure fails the job (and so skips publish).
+ # Skipped only when settings never resolved: then nothing was stopped.
+ # worker.sh retries the live start itself; one more whole pass here
+ # covers a longer network drop.
+ - name: Restore production (worker.sh down)
+ if: always() && inputs.restore_production && env.WORKER_SSH != ''
+ # Two full worker.sh down passes take about 11 minutes at worst;
+ # run_suites.RESERVED_MINUTES keeps this cap free in the job.
+ timeout-minutes: 20
+ run: |
+ set -euo pipefail
+ if ! tests/bench/worker.sh down; then
+ echo "::warning::worker.sh down failed; retrying once in 20 s"
+ sleep 20
+ tests/bench/worker.sh down
+ fi
+
+ publish:
+ needs: benchmark
+ if: success()
+ runs-on: ubuntu-latest
+ permissions:
+ contents: write
+ pages: write
+ id-token: write
+ environment:
+ name: github-pages
+ url: ${{ steps.deploy.outputs.page_url }}
+ steps:
+ - uses: actions/checkout@v7
+
+ - uses: actions/setup-python@v7
+ with:
+ python-version: '3.11'
+
+ - uses: actions/download-artifact@v8
+ with:
+ name: bench-summary
+ path: bench-out
+
+ - name: Check out gh-pages into site/
+ run: |
+ set -euo pipefail
+ if git ls-remote --exit-code --heads origin gh-pages >/dev/null; then
+ git fetch --depth=1 origin gh-pages
+ git worktree add -B gh-pages site FETCH_HEAD
+ else
+ git worktree add --orphan -b gh-pages site
+ fi
+
+ - name: Update history and site
+ run: |
+ set -euo pipefail
+ mkdir -p site/data/runs
+ cp bench-out/summary.json "site/data/runs/${GITHUB_RUN_ID}.json"
+ # The dashboard files; bench-site/data is a local preview sample.
+ rsync -a --exclude data/ bench-site/ site/
+ touch site/.nojekyll
+ python3 tests/bench/collect.py index site/data/runs --out site/data/index.json
+
+ - name: Push gh-pages
+ run: |
+ set -euo pipefail
+ cd site
+ git config user.name "github-actions[bot]"
+ git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
+ git add -A
+ if git diff --cached --quiet; then
+ echo "nothing to commit"
+ else
+ git commit -m "benchmark run ${GITHUB_RUN_ID} (${GITHUB_SHA::7})"
+ git push origin gh-pages
+ fi
+
+ - uses: actions/configure-pages@v6
+
+ - uses: actions/upload-pages-artifact@v5
+ with:
+ path: site
+
+ - id: deploy
+ uses: actions/deploy-pages@v5
diff --git a/README.md b/README.md
index 16212ce..664ebe5 100644
--- a/README.md
+++ b/README.md
@@ -179,6 +179,8 @@ On macOS, update by hand: `git pull` then `make restart`.
- [docs/updating.md](docs/updating.md) — how updates and the weekly timer work.
- [docs/codex.md](docs/codex.md) — using Codex CLI with this server as the model.
- [docs/claude-code.md](docs/claude-code.md) — same for Claude Code, which also gets MCP.
+- [docs/benchmarks.md](docs/benchmarks.md) — browser-agent benchmark findings, all in one place.
+- [docs/benchmark-action.md](docs/benchmark-action.md) — the manual GitHub Actions job that runs the benchmark on the RTX worker.
## License
diff --git a/bench-site/app.js b/bench-site/app.js
new file mode 100644
index 0000000..7b374d9
--- /dev/null
+++ b/bench-site/app.js
@@ -0,0 +1,405 @@
+/* RTX benchmark dashboard — plain JS, no build step, no external libraries.
+ * Reads data/index.json (written by tests/bench/collect.py) and renders:
+ * - latest-run suite cards
+ * - pass-rate-over-runs line chart (one line per suite)
+ * - median-time-over-runs line chart (one line per suite)
+ * - peak-VRAM-vs-context scatter, one point per suite at the ctx of the
+ * server that suite ran on, with an 8,192 MiB reference line
+ * - a runs table (date, commit link, model, ctx, suites)
+ * Charts are drawn as inline SVG built with the DOM API — no canvas
+ * libraries, no chart frameworks. Colour is never the only signal: every
+ * series has a legend entry with its name, every point has a native
+ * tooltip (
), and every card/table cell repeats the number as text.
+ */
+(function () {
+ "use strict";
+
+ var REPO_URL = "https://github.com/KSEGIT/QuickCleverModel";
+ var VRAM_CAP_MIB = 8192;
+
+ // Fixed order, fixed colour per suite — never reassigned based on what a
+ // given run happens to contain (dataviz: "colour follows the entity").
+ var SUITES = [
+ { key: "fixture", label: "Fixture", color: "--series-fixture" },
+ { key: "long_context", label: "Long context", color: "--series-long-context" },
+ { key: "live_web", label: "Live web", color: "--series-live-web" },
+ { key: "concurrency_serial", label: "Concurrency (serial)", color: "--series-concurrency-serial" },
+ { key: "concurrency_parallel", label: "Concurrency (parallel)", color: "--series-concurrency-parallel" }
+ ];
+
+ function suiteMeta(key) {
+ for (var i = 0; i < SUITES.length; i++) {
+ if (SUITES[i].key === key) return SUITES[i];
+ }
+ return { key: key, label: key, color: "--text-secondary" };
+ }
+
+ function cssVar(name) {
+ return getComputedStyle(document.documentElement).getPropertyValue(name).trim();
+ }
+
+ function el(tag, attrs, children) {
+ var isSvg = tag === "svg" || tag === "polyline" || tag === "line" ||
+ tag === "circle" || tag === "text" || tag === "g" || tag === "rect" || tag === "title";
+ var node = isSvg
+ ? document.createElementNS("http://www.w3.org/2000/svg", tag)
+ : document.createElement(tag);
+ for (var key in (attrs || {})) {
+ if (key === "class") node.setAttribute("class", attrs[key]);
+ else if (key === "text") node.textContent = attrs[key];
+ else node.setAttribute(key, attrs[key]);
+ }
+ (children || []).forEach(function (c) { if (c) node.appendChild(c); });
+ return node;
+ }
+
+ function fmtNum(n, digits) {
+ if (n === null || n === undefined) return "–"; // en dash for "no data"
+ return n.toLocaleString(undefined, { maximumFractionDigits: digits === undefined ? 0 : digits });
+ }
+
+ function fmtSeconds(n) {
+ if (n === null || n === undefined) return "–";
+ return fmtNum(n, 1) + " s";
+ }
+
+ function fmtDate(iso) {
+ if (!iso) return "–";
+ var d = new Date(iso);
+ if (isNaN(d.getTime())) return iso;
+ return d.toISOString().slice(0, 16).replace("T", " ") + " UTC";
+ }
+
+ function shortCommit(sha) {
+ if (!sha) return null;
+ return sha.length > 7 ? sha.slice(0, 7) : sha;
+ }
+
+ // The context size of the server a suite ran on. The concurrency suites
+ // run on their own 2-slot server, so a suite's own ctx wins; older
+ // summaries without it fall back to the run's ctx.
+ function suiteCtx(run, suite) {
+ return suite.ctx != null ? suite.ctx : run.ctx;
+ }
+
+ // ---------------------------------------------------------------------
+ // Latest-run cards
+ // ---------------------------------------------------------------------
+
+ function renderCards(run) {
+ var grid = el("div", { class: "card-grid" });
+ var keys = Object.keys(run.suites || {});
+ if (keys.length === 0) {
+ grid.appendChild(el("p", { text: "The latest run requested no suites." }));
+ return grid;
+ }
+ SUITES.concat(keys.filter(function (k) { return !SUITES.some(function (s) { return s.key === k; }); })
+ .map(function (k) { return { key: k, label: k, color: "--text-secondary" }; }))
+ .forEach(function (meta) {
+ var suite = run.suites[meta.key];
+ if (!suite) return;
+ var card = el("div", { class: "suite-card" });
+ var swatch = el("span", { class: "swatch", style: "background:" + cssVar(meta.color) + ";" });
+ card.appendChild(el("p", { class: "suite-name" }, [swatch, document.createTextNode(meta.label)]));
+
+ var passOk = suite.total > 0 && suite.passed === suite.total;
+ var pill = el("span", {
+ class: "status-pill " + (passOk ? "status-good" : "status-critical"),
+ text: (passOk ? "PASS " : "") + suite.passed + "/" + suite.total
+ });
+ var dl = el("dl", {}, [
+ el("dt", { text: "Pass rate" }), el("dd", {}, [pill, document.createTextNode(" " + fmtNum(suite.pass_rate, 1) + "%")]),
+ el("dt", { text: "Median time" }), el("dd", { text: fmtSeconds(suite.median_seconds) }),
+ el("dt", { text: "Peak VRAM" }), el("dd", { text: suite.peak_vram_mib != null ? fmtNum(suite.peak_vram_mib) + " MiB" : "–" })
+ ]);
+ card.appendChild(dl);
+ if (suite.error) {
+ card.appendChild(el("p", { class: "error-text", text: "Error: " + suite.error }));
+ }
+ grid.appendChild(card);
+ });
+ return grid;
+ }
+
+ // ---------------------------------------------------------------------
+ // Generic multi-series line chart (SVG, DOM API, no libraries)
+ // ---------------------------------------------------------------------
+
+ function drawLineChart(runs, opts) {
+ // opts: { valueFn(suite) -> number|null, yFormat(n) -> string, yMax? }
+ var W = 640, H = 260;
+ var marginL = 46, marginR = 16, marginT = 16, marginB = 34;
+ var plotW = W - marginL - marginR, plotH = H - marginT - marginB;
+
+ var seriesData = SUITES.map(function (meta) {
+ var points = runs.map(function (run, i) {
+ var suite = run.suites && run.suites[meta.key];
+ var v = suite ? opts.valueFn(suite) : null;
+ return { i: i, v: (v === undefined ? null : v), run: run };
+ });
+ return { meta: meta, points: points };
+ }).filter(function (s) { return s.points.some(function (p) { return p.v !== null; }); });
+
+ var allVals = [];
+ seriesData.forEach(function (s) { s.points.forEach(function (p) { if (p.v !== null) allVals.push(p.v); }); });
+ var yMax = opts.yMax !== undefined ? opts.yMax : Math.max.apply(null, allVals.concat([0])) * 1.15;
+ if (yMax <= 0) yMax = 1;
+ var yMin = 0;
+
+ var n = runs.length;
+ function xAt(i) { return n <= 1 ? marginL + plotW / 2 : marginL + (plotW * i) / (n - 1); }
+ function yAt(v) { return marginT + plotH - ((v - yMin) / (yMax - yMin)) * plotH; }
+
+ var svg = el("svg", { class: "chart", viewBox: "0 0 " + W + " " + H, role: "img", "aria-label": opts.ariaLabel || "" });
+
+ // gridlines + y labels
+ var ticks = 4;
+ for (var t = 0; t <= ticks; t++) {
+ var v = yMax * t / ticks;
+ var y = yAt(v);
+ svg.appendChild(el("line", { class: "gridline", x1: marginL, x2: W - marginR, y1: y, y2: y }));
+ svg.appendChild(el("text", { x: marginL - 8, y: y + 3, "text-anchor": "end", "font-size": "10", text: opts.yFormat ? opts.yFormat(v) : fmtNum(v) }));
+ }
+ // x axis baseline + run date labels
+ svg.appendChild(el("line", { class: "axis-line", x1: marginL, x2: W - marginR, y1: marginT + plotH, y2: marginT + plotH }));
+ runs.forEach(function (run, i) {
+ var x = xAt(i);
+ svg.appendChild(el("line", { class: "axis-line", x1: x, x2: x, y1: marginT + plotH, y2: marginT + plotH + 4 }));
+ var label = (run.created_utc || "").slice(5, 10) || String(i + 1);
+ svg.appendChild(el("text", { x: x, y: H - 8, "text-anchor": "middle", "font-size": "10", text: label }));
+ });
+
+ // series lines + points
+ seriesData.forEach(function (s) {
+ var color = cssVar(s.meta.color);
+ var seg = [];
+ function flush() {
+ if (seg.length > 1) {
+ var pts = seg.map(function (p) { return xAt(p.i) + "," + yAt(p.v); }).join(" ");
+ svg.appendChild(el("polyline", { class: "series-line", points: pts, stroke: color }));
+ }
+ seg = [];
+ }
+ s.points.forEach(function (p) {
+ if (p.v === null) { flush(); return; }
+ seg.push(p);
+ });
+ flush();
+ s.points.forEach(function (p) {
+ if (p.v === null) return;
+ var cx = xAt(p.i), cy = yAt(p.v);
+ var title = el("title", { text: s.meta.label + " — " + fmtDate(p.run.created_utc) + ": " + (opts.yFormat ? opts.yFormat(p.v) : fmtNum(p.v)) });
+ svg.appendChild(el("circle", { class: "series-point", cx: cx, cy: cy, r: 4, fill: color }, [title]));
+ });
+ });
+
+ var wrap = el("div", { class: "chart-wrap" }, [svg]);
+ var legend = el("div", { class: "legend" }, seriesData.map(function (s) {
+ return el("span", { class: "legend-item" }, [
+ el("span", { class: "swatch", style: "background:" + cssVar(s.meta.color) + ";" }),
+ document.createTextNode(s.meta.label)
+ ]);
+ }));
+ if (seriesData.length === 0) {
+ return el("div", {}, [el("p", { class: "panel-note", text: "No suites with this data yet." })]);
+ }
+ var container = el("div", {}, [wrap, legend]);
+ return container;
+ }
+
+ // ---------------------------------------------------------------------
+ // Peak VRAM vs context scatter
+ // ---------------------------------------------------------------------
+
+ function drawVramScatter(runs) {
+ var W = 640, H = 260;
+ var marginL = 54, marginR = 16, marginT = 16, marginB = 34;
+ var plotW = W - marginL - marginR, plotH = H - marginT - marginB;
+
+ var points = [];
+ runs.forEach(function (run) {
+ Object.keys(run.suites || {}).forEach(function (key) {
+ var suite = run.suites[key];
+ var ctx = suiteCtx(run, suite);
+ if (ctx == null || suite.peak_vram_mib == null) return;
+ points.push({ ctx: ctx, vram: suite.peak_vram_mib, run: run, meta: suiteMeta(key), parallel: suite.parallel });
+ });
+ });
+
+ if (points.length === 0) {
+ return el("p", { class: "panel-note", text: "No suite yet has both a context size and a peak VRAM reading." });
+ }
+
+ var xs = points.map(function (p) { return p.ctx; });
+ var ys = points.map(function (p) { return p.vram; }).concat([VRAM_CAP_MIB]);
+ var xMax = Math.max.apply(null, xs) * 1.1;
+ var xMin = 0;
+ var yMax = Math.max.apply(null, ys) * 1.1;
+ var yMin = 0;
+
+ function xAt(v) { return marginL + ((v - xMin) / (xMax - xMin)) * plotW; }
+ function yAt(v) { return marginT + plotH - ((v - yMin) / (yMax - yMin)) * plotH; }
+
+ var svg = el("svg", { class: "chart", viewBox: "0 0 " + W + " " + H, role: "img", "aria-label": "Peak VRAM against context size, in mebibytes" });
+
+ var ticks = 4;
+ for (var t = 0; t <= ticks; t++) {
+ var yv = yMax * t / ticks;
+ var y = yAt(yv);
+ svg.appendChild(el("line", { class: "gridline", x1: marginL, x2: W - marginR, y1: y, y2: y }));
+ svg.appendChild(el("text", { x: marginL - 8, y: y + 3, "text-anchor": "end", "font-size": "10", text: fmtNum(yv) }));
+ }
+ for (var xt = 0; xt <= ticks; xt++) {
+ var xv = xMax * xt / ticks;
+ var x = xAt(xv);
+ svg.appendChild(el("text", { x: x, y: H - 8, "text-anchor": "middle", "font-size": "10", text: fmtNum(xv) }));
+ }
+ svg.appendChild(el("line", { class: "axis-line", x1: marginL, x2: W - marginR, y1: marginT + plotH, y2: marginT + plotH }));
+ svg.appendChild(el("line", { class: "axis-line", x1: marginL, x2: marginL, y1: marginT, y2: marginT + plotH }));
+
+ // 8,192 MiB reference line
+ var refY = yAt(VRAM_CAP_MIB);
+ if (refY >= marginT && refY <= marginT + plotH) {
+ svg.appendChild(el("line", { class: "ref-line", x1: marginL, x2: W - marginR, y1: refY, y2: refY }));
+ svg.appendChild(el("text", { x: W - marginR, y: refY - 4, "text-anchor": "end", "font-size": "10", text: "8,192 MiB" }));
+ }
+
+ points.forEach(function (p) {
+ var cx = xAt(p.ctx), cy = yAt(p.vram);
+ var slots = p.parallel != null ? ", " + p.parallel + (p.parallel === 1 ? " slot" : " slots") : "";
+ var title = el("title", { text: p.meta.label + " — " + (p.run.model || "run") + ", " + fmtDate(p.run.created_utc) +
+ " — ctx " + fmtNum(p.ctx) + slots + ", peak " + fmtNum(p.vram) + " MiB" });
+ svg.appendChild(el("circle", { class: "series-point", cx: cx, cy: cy, r: 5, fill: cssVar(p.meta.color) }, [title]));
+ });
+
+ var shown = SUITES.filter(function (meta) {
+ return points.some(function (p) { return p.meta.key === meta.key; });
+ });
+ var legend = el("div", { class: "legend" }, shown.map(function (meta) {
+ return el("span", { class: "legend-item" }, [
+ el("span", { class: "swatch", style: "background:" + cssVar(meta.color) + ";" }),
+ document.createTextNode(meta.label)
+ ]);
+ }));
+
+ return el("div", {}, [
+ el("div", { class: "chart-wrap" }, [svg]),
+ legend,
+ el("p", { class: "panel-note", text: "Dashed line: 8,192 MiB, the RTX 3070 Ti's VRAM budget." })
+ ]);
+ }
+
+ // ---------------------------------------------------------------------
+ // Runs table
+ // ---------------------------------------------------------------------
+
+ function renderTable(runs) {
+ var table = el("table", { class: "runs-table" });
+ var thead = el("thead", {}, [el("tr", {}, [
+ el("th", { text: "Date" }), el("th", { text: "Commit" }), el("th", { text: "Model" }),
+ el("th", { text: "Ctx" }), el("th", { text: "Suites" })
+ ])]);
+ var tbody = el("tbody");
+ // Newest first for the table, even though the chart x-axis reads oldest-first.
+ runs.slice().reverse().forEach(function (run) {
+ var commit = shortCommit(run.commit);
+ var commitCell = commit
+ ? el("td", {}, [el("a", { href: REPO_URL + "/commit/" + run.commit, text: commit })])
+ : el("td", { text: "–" });
+ var tags = el("span", { class: "suite-tags" }, Object.keys(run.suites || {}).map(function (key) {
+ var suite = run.suites[key];
+ var meta = suiteMeta(key);
+ var ctx = suiteCtx(run, suite);
+ // Name the ctx only where it differs from the run's Ctx column.
+ var ctxNote = ctx != null && ctx !== run.ctx ? " · ctx " + fmtNum(ctx) : "";
+ return el("span", { class: "suite-tag", text: meta.label + " · " + fmtNum(suite.pass_rate, 0) + "%" + ctxNote });
+ }));
+ var dateText = fmtDate(run.created_utc);
+ if (run.sample) dateText += " (sample)";
+ var row = el("tr", {}, [
+ el("td", { text: dateText }),
+ commitCell,
+ el("td", { text: run.model || "–" }),
+ el("td", { text: run.ctx != null ? fmtNum(run.ctx) : "–" }),
+ el("td", {}, [tags])
+ ]);
+ tbody.appendChild(row);
+ });
+ table.appendChild(thead);
+ table.appendChild(tbody);
+ return el("div", { class: "table-scroll" }, [table]);
+ }
+
+ // ---------------------------------------------------------------------
+ // Page assembly
+ // ---------------------------------------------------------------------
+
+ function panel(title, note, body) {
+ var children = [el("h2", { text: title })];
+ if (note) children.push(el("p", { class: "panel-note", text: note }));
+ children.push(body);
+ return el("section", { class: "panel" }, children);
+ }
+
+ function render(index) {
+ var app = document.getElementById("app");
+ var headerNote = document.getElementById("header-note");
+ app.textContent = "";
+
+ var runs = (index && index.runs) || [];
+ var hasSample = runs.some(function (r) { return r.sample; });
+
+ if (hasSample) {
+ headerNote.textContent = "";
+ var headerRow = el("p", {}, [
+ document.createTextNode(runs.length + " run" + (runs.length === 1 ? "" : "s") + " recorded. "),
+ el("span", { class: "badge badge-sample", text: "Sample data — no real run yet" })
+ ]);
+ headerNote.appendChild(headerRow);
+ } else if (runs.length > 0) {
+ headerNote.textContent = runs.length + " run" + (runs.length === 1 ? "" : "s") + " recorded. Latest: " + fmtDate(runs[runs.length - 1].created_utc) + ".";
+ } else {
+ headerNote.textContent = "No runs recorded yet.";
+ }
+
+ if (runs.length === 0) {
+ app.appendChild(el("div", { class: "empty-note" }, [
+ el("p", { text: "No runs yet." }),
+ el("p", { text: "Run the “benchmark” workflow from the Actions tab to populate this dashboard." })
+ ]));
+ return;
+ }
+
+ var latest = runs[runs.length - 1];
+ app.appendChild(panel("Latest run", "Suite results for the most recent run (" + fmtDate(latest.created_utc) + ").", renderCards(latest)));
+
+ app.appendChild(panel("Pass rate over runs", "One line per suite; a suite with no data for a run is skipped, not zero.",
+ drawLineChart(runs, { valueFn: function (s) { return s.total > 0 ? s.pass_rate : null; }, yFormat: function (v) { return fmtNum(v) + "%"; }, yMax: 100, ariaLabel: "Pass rate per suite across runs, 0 to 100 percent" })));
+
+ app.appendChild(panel("Median time over runs", "Median wall-clock seconds per suite.",
+ drawLineChart(runs, { valueFn: function (s) { return s.median_seconds; }, yFormat: function (v) { return fmtNum(v, 0) + "s"; }, ariaLabel: "Median seconds per suite across runs" })));
+
+ app.appendChild(panel("Peak VRAM vs. context", "Each point is one suite in one run: the context size of the server that suite ran on, and its highest VRAM use.",
+ drawVramScatter(runs)));
+
+ app.appendChild(panel("All runs", null, renderTable(runs)));
+ }
+
+ function renderError(message) {
+ var app = document.getElementById("app");
+ document.getElementById("header-note").textContent = "Could not load run data.";
+ app.textContent = "";
+ app.appendChild(el("div", { class: "empty-note" }, [
+ el("p", { text: "Could not load data/index.json." }),
+ el("p", { text: String(message) })
+ ]));
+ }
+
+ fetch("data/index.json")
+ .then(function (r) {
+ if (!r.ok) throw new Error("HTTP " + r.status);
+ return r.json();
+ })
+ .then(render)
+ .catch(function (err) { renderError(err && err.message ? err.message : err); });
+})();
diff --git a/bench-site/data/index.json b/bench-site/data/index.json
new file mode 100644
index 0000000..a200db4
--- /dev/null
+++ b/bench-site/data/index.json
@@ -0,0 +1,101 @@
+{
+ "schema": 1,
+ "runs": [
+ {
+ "schema": 1,
+ "run_id": "2026092501",
+ "created_utc": "2026-09-25T20:01:34+00:00",
+ "commit": "9f1c2ab",
+ "model": "qwen3.5-9b-q4_k_m",
+ "ctx": 32768,
+ "gpu": "RTX 3070 Ti",
+ "suites": {
+ "fixture": {
+ "passed": 1,
+ "total": 1,
+ "pass_rate": 100.0,
+ "median_seconds": 15.7,
+ "peak_vram_mib": 6269,
+ "peak_prompt_tokens": 4874,
+ "ctx": 32768,
+ "parallel": 1
+ },
+ "live_web": {
+ "passed": 1,
+ "total": 2,
+ "pass_rate": 50.0,
+ "median_seconds": 49.5,
+ "peak_vram_mib": 6120,
+ "peak_prompt_tokens": 11124,
+ "ctx": 32768,
+ "parallel": 1
+ }
+ },
+ "sample": true
+ },
+ {
+ "schema": 1,
+ "run_id": "2026092502",
+ "created_utc": "2026-09-25T21:10:00+00:00",
+ "commit": "b2ac901",
+ "model": "qwen3.5-9b-q4_k_m",
+ "ctx": 49152,
+ "gpu": "RTX 3070 Ti",
+ "suites": {
+ "concurrency_serial": {
+ "passed": 3,
+ "total": 4,
+ "pass_rate": 75.0,
+ "median_seconds": 16.3,
+ "peak_vram_mib": 6823,
+ "peak_prompt_tokens": 6801,
+ "ctx": 49152,
+ "parallel": 2
+ },
+ "concurrency_parallel": {
+ "passed": 2,
+ "total": 4,
+ "pass_rate": 50.0,
+ "median_seconds": 29.7,
+ "peak_vram_mib": 6823,
+ "peak_prompt_tokens": 6659,
+ "ctx": 49152,
+ "parallel": 2
+ }
+ },
+ "sample": true
+ },
+ {
+ "schema": 1,
+ "run_id": "2026092503",
+ "created_utc": "2026-09-26T05:50:24+00:00",
+ "commit": "c77e410",
+ "model": "qwen3.5-9b-q4_k_m",
+ "ctx": 65536,
+ "gpu": "RTX 3070 Ti",
+ "suites": {
+ "long_context": {
+ "passed": 1,
+ "total": 1,
+ "pass_rate": 100.0,
+ "median_seconds": 26.1,
+ "peak_vram_mib": 7323,
+ "peak_prompt_tokens": 59906,
+ "ctx": 65536,
+ "parallel": 1
+ },
+ "live_web": {
+ "passed": 2,
+ "total": 3,
+ "pass_rate": 66.7,
+ "median_seconds": 69.5,
+ "peak_vram_mib": 7100,
+ "peak_prompt_tokens": 11124,
+ "ctx": 65536,
+ "parallel": 1
+ }
+ },
+ "sample": true
+ }
+ ]
+}
diff --git a/bench-site/index.html b/bench-site/index.html
new file mode 100644
index 0000000..1839739
--- /dev/null
+++ b/bench-site/index.html
@@ -0,0 +1,38 @@
+
+
+
+
+
+RTX benchmark dashboard
+
+
+
+
+
+
+
+
RTX benchmark dashboard
+
Loading run data…
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
diff --git a/bench-site/style.css b/bench-site/style.css
new file mode 100644
index 0000000..5f82126
--- /dev/null
+++ b/bench-site/style.css
@@ -0,0 +1,364 @@
+/* RTX benchmark dashboard — plain CSS, no build step.
+ Palette: validated default from the dataviz skill (references/palette.md).
+ Light is the default; dark applies automatically via prefers-color-scheme. */
+
+:root {
+ color-scheme: light;
+
+ --page-bg: #f9f9f7;
+ --surface: #fcfcfb;
+ --text-primary: #0b0b0b;
+ --text-secondary: #52514e;
+ --text-muted: #898781;
+ --gridline: #e1e0d9;
+ --baseline: #c3c2b7;
+ --border: rgba(11, 11, 11, 0.10);
+
+ --status-good: #0ca30c;
+ --status-critical: #d03b3b;
+
+ /* Categorical, slots 1-5, fixed order — never reassigned per run. */
+ --series-fixture: #2a78d6; /* blue */
+ --series-long-context: #eb6834; /* orange */
+ --series-live-web: #1baf7a; /* aqua */
+ --series-concurrency-serial: #eda100; /* yellow */
+ --series-concurrency-parallel: #e87ba4;/* magenta*/
+
+ --font-sans: system-ui, -apple-system, "Segoe UI", sans-serif;
+}
+
+@media (prefers-color-scheme: dark) {
+ :root {
+ color-scheme: dark;
+
+ --page-bg: #0d0d0d;
+ --surface: #1a1a19;
+ --text-primary: #ffffff;
+ --text-secondary: #c3c2b7;
+ --text-muted: #898781;
+ --gridline: #2c2c2a;
+ --baseline: #383835;
+ --border: rgba(255, 255, 255, 0.10);
+
+ --status-good: #0ca30c;
+ --status-critical: #e66767;
+
+ --series-fixture: #3987e5;
+ --series-long-context: #d95926;
+ --series-live-web: #199e70;
+ --series-concurrency-serial: #c98500;
+ --series-concurrency-parallel: #d55181;
+ }
+}
+
+* {
+ box-sizing: border-box;
+}
+
+html, body {
+ margin: 0;
+ padding: 0;
+}
+
+body {
+ background: var(--page-bg);
+ color: var(--text-primary);
+ font-family: var(--font-sans);
+ font-size: 16px;
+ line-height: 1.45;
+ /* Never scroll the whole page sideways, even if a child misbehaves. */
+ overflow-x: hidden;
+}
+
+.page {
+ max-width: 1000px;
+ margin: 0 auto;
+ padding: 16px;
+}
+
+header.site-header {
+ margin-bottom: 16px;
+}
+
+header.site-header h1 {
+ font-size: 1.4rem;
+ margin: 0 0 4px;
+}
+
+header.site-header p {
+ margin: 0;
+ color: var(--text-secondary);
+ font-size: 0.9rem;
+}
+
+.badge {
+ display: inline-flex;
+ align-items: center;
+ gap: 4px;
+ border: 1px solid var(--border);
+ border-radius: 999px;
+ padding: 2px 10px;
+ font-size: 0.78rem;
+ font-weight: 600;
+ color: var(--text-secondary);
+ background: var(--surface);
+ white-space: nowrap;
+}
+
+.badge.badge-sample {
+ color: #7a5b00;
+ border-color: #eda100;
+}
+@media (prefers-color-scheme: dark) {
+ .badge.badge-sample {
+ color: #f0c04a;
+ border-color: #c98500;
+ }
+}
+
+.empty-note {
+ border: 1px dashed var(--baseline);
+ border-radius: 8px;
+ padding: 24px;
+ text-align: center;
+ color: var(--text-secondary);
+}
+
+section.panel {
+ background: var(--surface);
+ border: 1px solid var(--border);
+ border-radius: 10px;
+ padding: 16px;
+ margin-bottom: 20px;
+}
+
+section.panel h2 {
+ font-size: 1.05rem;
+ margin: 0 0 4px;
+}
+
+section.panel .panel-note {
+ margin: 0 0 12px;
+ color: var(--text-secondary);
+ font-size: 0.85rem;
+}
+
+/* Latest-run suite cards */
+.card-grid {
+ display: grid;
+ grid-template-columns: repeat(auto-fit, minmax(180px, 1fr));
+ gap: 12px;
+}
+
+.suite-card {
+ border: 1px solid var(--border);
+ border-radius: 8px;
+ padding: 12px;
+}
+
+.suite-card .suite-name {
+ font-weight: 600;
+ font-size: 0.92rem;
+ margin: 0 0 8px;
+ display: flex;
+ align-items: center;
+ gap: 6px;
+}
+
+.suite-card .swatch {
+ width: 10px;
+ height: 10px;
+ border-radius: 2px;
+ flex: 0 0 auto;
+ display: inline-block;
+}
+
+.suite-card dl {
+ margin: 0;
+ display: grid;
+ grid-template-columns: auto auto;
+ gap: 2px 8px;
+ font-size: 0.85rem;
+}
+
+.suite-card dt {
+ color: var(--text-secondary);
+}
+
+.suite-card dd {
+ margin: 0;
+ text-align: right;
+ font-variant-numeric: tabular-nums;
+}
+
+.status-pill {
+ font-weight: 700;
+ font-size: 0.78rem;
+ border-radius: 4px;
+ padding: 1px 6px;
+ display: inline-block;
+}
+
+.status-pill.status-good {
+ color: var(--status-good);
+ border: 1px solid var(--status-good);
+}
+
+.status-pill.status-critical {
+ color: var(--status-critical);
+ border: 1px solid var(--status-critical);
+}
+
+.suite-card .error-text {
+ margin-top: 8px;
+ font-size: 0.78rem;
+ color: var(--status-critical);
+ word-break: break-word;
+}
+
+/* Charts */
+.chart-wrap {
+ width: 100%;
+ overflow: hidden;
+}
+
+svg.chart {
+ width: 100%;
+ height: auto;
+ display: block;
+ font-family: var(--font-sans);
+}
+
+svg.chart text {
+ fill: var(--text-secondary);
+}
+
+svg.chart .axis-line {
+ stroke: var(--baseline);
+ stroke-width: 1;
+}
+
+svg.chart .gridline {
+ stroke: var(--gridline);
+ stroke-width: 1;
+}
+
+svg.chart .ref-line {
+ stroke: var(--text-muted);
+ stroke-width: 1;
+ stroke-dasharray: 4 3;
+}
+
+svg.chart .series-line {
+ fill: none;
+ stroke-width: 2;
+}
+
+svg.chart .series-point {
+ stroke: var(--surface);
+ stroke-width: 1.5;
+}
+
+svg.chart .end-label {
+ font-size: 11px;
+ font-weight: 600;
+}
+
+.legend {
+ display: flex;
+ flex-wrap: wrap;
+ gap: 10px 16px;
+ margin-top: 8px;
+ font-size: 0.8rem;
+ color: var(--text-secondary);
+}
+
+.legend-item {
+ display: inline-flex;
+ align-items: center;
+ gap: 6px;
+}
+
+.legend-item .swatch {
+ width: 10px;
+ height: 10px;
+ border-radius: 2px;
+ display: inline-block;
+}
+
+/* Runs table: the table itself may scroll sideways in its own box; the
+ page never does. */
+.table-scroll {
+ overflow-x: auto;
+ border: 1px solid var(--border);
+ border-radius: 8px;
+}
+
+table.runs-table {
+ width: 100%;
+ border-collapse: collapse;
+ font-size: 0.85rem;
+ min-width: 520px;
+}
+
+table.runs-table th,
+table.runs-table td {
+ padding: 8px 10px;
+ text-align: left;
+ border-bottom: 1px solid var(--gridline);
+ white-space: nowrap;
+}
+
+table.runs-table th {
+ color: var(--text-secondary);
+ font-weight: 600;
+ font-size: 0.78rem;
+ text-transform: uppercase;
+ letter-spacing: 0.02em;
+}
+
+table.runs-table tbody tr:last-child td {
+ border-bottom: none;
+}
+
+table.runs-table a {
+ color: var(--series-fixture);
+}
+
+.suite-tags {
+ display: flex;
+ gap: 4px;
+ flex-wrap: wrap;
+}
+
+.suite-tag {
+ font-size: 0.72rem;
+ border-radius: 4px;
+ padding: 1px 6px;
+ border: 1px solid var(--border);
+ color: var(--text-secondary);
+ white-space: nowrap;
+}
+
+footer.site-footer {
+ color: var(--text-muted);
+ font-size: 0.78rem;
+ text-align: center;
+ padding: 12px 0 24px;
+}
+
+/* Phone width: stack cards fully, keep type legible, no page-level
+ horizontal scroll (the table box handles its own overflow). */
+@media (max-width: 480px) {
+ .page {
+ padding: 12px;
+ }
+
+ header.site-header h1 {
+ font-size: 1.2rem;
+ }
+
+ .card-grid {
+ grid-template-columns: 1fr;
+ }
+}
diff --git a/docs/benchmark-action.md b/docs/benchmark-action.md
new file mode 100644
index 0000000..71e8396
--- /dev/null
+++ b/docs/benchmark-action.md
@@ -0,0 +1,238 @@
+# RTX benchmark action
+
+This page explains the GitHub Actions job that runs the browser-agent
+benchmark on the RTX worker, and how to set it up and use it. The workflow
+file is `.github/workflows/benchmark.yml`.
+
+## What it does
+
+- You start it by hand, from the Actions tab. It does not run on its own.
+- It joins the worker's tailnet, connects over SSH, and swaps the worker's
+ live model container for a test server for the length of the run.
+- It runs the browser-agent suites you choose against that test server.
+- It puts the live container back at the end — even if a suite failed — and
+ checks that it answers before the job finishes.
+- It saves the results, and shows them on a small dashboard site built by
+ GitHub Pages.
+
+### Why a manual trigger only
+
+This repository is public. A workflow that runs on `push` or on a pull
+request would also run for a pull request from a fork, with no review. That
+must never reach the worker over SSH. So the workflow only has a
+`workflow_dispatch` (manual) trigger.
+
+GitHub itself lets anyone with write access to the repository start a
+`workflow_dispatch` run. In practice, only the repository owner should start
+one, because a run stops the live model for everyone (see "Stopping
+production, on purpose" below).
+
+## One-time setup
+
+A repository owner does these steps once, before the first run.
+
+### 1. Tailscale
+
+1. In the Tailscale admin console, create an OAuth client and give it the
+ tag `tag:ci`.
+2. Add a Tailscale ACL rule that lets `tag:ci` reach the worker on port 22
+ only. Do not give it any wider access.
+3. Store the OAuth client ID and secret as `rtx-benchmark` environment
+ secrets (see the table below). An auth key works too, as a fallback.
+
+### 2. A dedicated SSH key
+
+1. Make a new SSH key pair. Use it only for this job. Do not reuse a key you
+ use anywhere else.
+2. Add the public key to the worker's `authorized_keys` for the account this
+ job will use.
+3. Run `ssh-keyscan ` by hand, once, from a machine you trust.
+ Look at the output and check it names the worker's real host key before
+ you store it. Do not skip this check.
+4. Store the private key and the `ssh-keyscan` output as `rtx-benchmark`
+ environment secrets.
+
+Never write the worker's real Tailscale IP address or its SSH username in a
+doc, an issue, or a commit message. Use a placeholder such as `100.x.y.z` or
+`` instead. The setup below keeps both out of the workflow's
+visible inputs for the same reason.
+
+### 3. Environment secrets and variables
+
+Create a GitHub Environment named `rtx-benchmark`, and add these secrets and
+variables to it. Put them on the environment, not on the repository.
+
+**Limit the environment to `main`.** In **Settings → Environments →
+rtx-benchmark**, set **Deployment branches and tags** to **Selected
+branches** and add only `main`. Without this rule, a user with write access
+can push a changed workflow to another branch and run it. That run gets the
+SSH key.
+
+**Secrets** (never shown on the run page, even in this public repository):
+
+| Secret | Holds |
+| --- | --- |
+| `TS_OAUTH_CLIENT_ID` | The Tailscale OAuth client ID from step 1 |
+| `TS_OAUTH_SECRET` | The Tailscale OAuth client secret (or use `TS_AUTHKEY` instead, with a Tailscale auth key) |
+| `WORKER_SSH_KEY` | The private half of the dedicated SSH key from step 2 |
+| `WORKER_KNOWN_HOSTS` | The checked `ssh-keyscan` output for the worker |
+| `WORKER_HOST` | Optional, but best: the worker's tailnet name or address (see below) |
+| `WORKER_SSH_USER` | Optional, but best: the SSH account the job connects as |
+
+The logs of this public repository are public. GitHub hides a secret's
+value in every log line. So store `WORKER_HOST` and `WORKER_SSH_USER` as
+secrets. You can use variables of the same names instead. The first step
+("Resolve settings") then masks them, but a variable is plain text before
+that step, so it can show in that step's header.
+
+**Variables** (defaults for a run; not secret, but still worth keeping out of
+public docs where they'd be specific to your worker):
+
+| Variable | Meaning | Default |
+| --- | --- | --- |
+| `WORKER_HOST` | The worker's tailnet name or address, if not a secret | none — set it here or as a secret |
+| `WORKER_SSH_USER` | The SSH account the job connects as, if not a secret | none — set it here or as a secret |
+| `BENCH_GPU_NAME` | The GPU name the dashboard shows for each run, for example `RTX 3070 Ti` | none — the run records no GPU name |
+| `LIVE_CONTAINER` | The production container the job pauses and restores | `bonsai-llama-1` |
+| `LIVE_HEALTH_URL` | Health-check URL for the live container, checked on the worker | `http://127.0.0.1:8080/health` |
+| `LIVE_ENV_FILE` | Worker path to the live container's env file, if it needs one | none — no key is sent if unset |
+| `TEST_IMAGE` | The pinned test image the benchmark runs | `qcm-rtx-validation:922be44` |
+| `MODELS_DIR` | Worker path to the model files | none — must be set |
+| `TEMPLATE_FILE` | Worker path to the chat template for the test server | none — must be set |
+
+A `workflow_dispatch` run also takes `worker_host` and `ssh_user` inputs that
+can override these values. Leave them blank when you start a run — the
+job then uses the variables above. If you type a value instead, it shows on
+the run's page, and this repository is public.
+
+**`LIVE_ENV_FILE` must be readable by the SSH user, without `sudo`.** The
+restore step reads the live server's API key from this file, on the worker,
+to check that the live container answers after it restarts. If the SSH user
+cannot read the file, the check sends no key. The live server can then
+answer 401, and the job reports the live container as unhealthy — even
+though the container itself is running. Set file permissions so the
+dedicated SSH account (from step 2) can read this file directly.
+
+### 4. Turn on GitHub Pages
+
+1. Go to **Settings → Pages**.
+2. Under **Build and deployment**, set **Source** to **GitHub Actions**.
+3. Save. This step needs repository admin rights, so only the owner can do
+ it, and it is a one-time setup step, not something each run repeats.
+4. Go to **Settings → Environments → github-pages** (GitHub creates this
+ environment the first time a Pages deployment runs, or you can create it
+ yourself). Check its **Deployment branches and tags** rule. If it only
+ allows a specific branch (often the default branch), and you dispatch a
+ benchmark run from a different branch, the publish job stalls or is
+ blocked — the environment rule never matches. Add every branch you might
+ dispatch a run from. Since `rtx-benchmark` allows only `main`, allowing
+ `main` here is enough.
+
+## Starting a run
+
+1. Go to the **Actions** tab, open the `benchmark.yml` workflow, and choose
+ **Run workflow**.
+2. Fill in the inputs you want to change. Leave the rest blank to use the
+ defaults.
+
+| Input | Meaning | Default |
+| --- | --- | --- |
+| `worker_host` | Overrides `WORKER_HOST` for this run only | (environment variable) |
+| `ssh_user` | Overrides `WORKER_SSH_USER` for this run only | (environment variable) |
+| `model_preset` | Which model to benchmark: `qwen3.5-9b` or `qwen3.5-4b` | `qwen3.5-9b` |
+| `ctx` | Context size, in tokens | `32768` |
+| `suites` | Space-separated list of suites to run: `fixture`, `long_context`, `live_web`, `concurrency` | `fixture long_context live_web` |
+| `repetitions` | How many times to repeat the `fixture` tasks and the `live_web` run. `long_context` and `concurrency` ignore it. | `1` |
+| `restore_production` | Restart the live container when the run ends | `true` |
+
+3. Anyone with write access to the repository can start a run, but only the
+ repository owner should — see "Why a manual trigger only" above.
+
+### The `repetitions` limit
+
+The job has 240 minutes. It must always keep time to put the live container
+back. So the first step works out the longest time each suite step can
+take, from the suites you chose and `repetitions`. Each suite step gets
+that time as its own limit. If the total does not fit, the run stops at
+once, with an error that gives the highest `repetitions` that fits.
+
+- With the default suites, the most is 2, with or without `concurrency`.
+ With `fixture` alone, the most is 3.
+- Without `fixture`, the limit is much higher (26 for `live_web` alone).
+
+`tests/bench/run_suites.py` holds the numbers (`step_minutes`,
+`max_repetitions`), so the workflow and the suites always agree.
+4. Click **Run workflow** and wait. A run can take a while, because it
+ downloads nothing new but does run real browser tasks against a live
+ model, one suite at a time.
+
+### About `restore_production`
+
+Leave this at its default, `true`, for almost every run. When it is `true`,
+the job always puts the live container back at the end, even if a suite
+failed. When it is `false`, the job leaves the test server running and does
+**not** restart the live container automatically — you must restart it
+yourself. Only set it to `false` if you plan to inspect the test server by
+hand right after the run.
+
+## What happens during a run
+
+1. The job joins the tailnet, then connects to the worker over SSH.
+2. It stops the live container (`bonsai-llama-1` by default) and starts the
+ test server in its place, bound to the worker's loopback address only.
+3. It runs the suites you chose:
+
+ | Suite | Server needs | What it runs | Pass means |
+ | --- | --- | --- | --- |
+ | `fixture` | 1 slot | Core browser tasks, repeated | Each task's own check passes |
+ | `long_context` | 1 slot | A prompt near the context limit, asking the model to quote one exact line | The line quoted is exactly right |
+ | `live_web` | 1 slot | A real DuckDuckGo image search that must save one real image file | An image file is actually saved |
+ | `concurrency` | 2 slots | Four job-application runs: one at a time, then two at a time in pairs | Each run's own check passes |
+
+ For `concurrency`, the job restarts the test server first, with 2 slots
+ and a 49,152-token context, because that suite needs two agents running
+ at once. If this restart or the `concurrency` suite fails, the job goes
+ on: it records the error under `concurrency`, and still publishes the
+ other suites' results.
+4. It saves raw results, with any saved images, as one workflow artifact,
+ so you can look at them even without the dashboard.
+5. At the end, whether or not a suite failed, it stops the test server and
+ starts the live container again (unless you set `restore_production` to
+ `false`). It checks the live container's health endpoint. If that check
+ does not return HTTP 200, the job fails — that tells you production is
+ not back up cleanly, instead of hiding the problem. This step retries on
+ its own: starting the live container is tried up to 5 times, and if the
+ whole step still fails, the workflow waits 20 seconds and runs it a
+ second time. This covers a short network drop between the runner and the
+ worker.
+6. On success, a separate publish job downloads the run summary that this
+ job uploaded as an artifact, then writes it to the `gh-pages` branch
+ (`data/runs/.json`), rebuilds the run index (`data/index.json`),
+ and updates the Pages site from `bench-site/`. If the benchmark job never
+ reaches its "Upload summary" step, there is nothing for the publish job
+ to download, and publishing does not happen.
+
+Only one run can happen at a time. If you start a second run while one is
+already going, it waits its turn instead of running alongside the first.
+
+## Reading the dashboard
+
+Open the repository's Pages URL — **Settings → Pages** shows it once a run
+has published. The dashboard shows:
+
+- The latest run's pass rate, median time, and peak VRAM, for each suite.
+- A pass-rate history chart, per suite, across all runs.
+- A median-time history chart.
+- A peak-VRAM-versus-context chart, with a line at 8,192 MiB (the worker's
+ total VRAM). It has one point for each suite in each run. A point uses
+ the context size of the server that suite ran on. So the `concurrency`
+ points show 49,152 tokens, even when the other suites in that run used a
+ different `ctx`.
+- A table of every run, with its date, commit, model, context, and suites.
+
+## Stopping production, on purpose
+
+A run stops the real chat model for as long as the run takes. Anyone using
+the live chat during that window sees it go down, then come back once the
+run ends. Pick a quiet time to start a run, and tell anyone else who might
+be using the live model first.
diff --git a/docs/benchmarks.md b/docs/benchmarks.md
new file mode 100644
index 0000000..1ef6c34
--- /dev/null
+++ b/docs/benchmarks.md
@@ -0,0 +1,181 @@
+# Benchmark findings, all in one place
+
+This page brings together the browser-agent benchmark findings recorded so
+far. It is a summary. For full detail, raw report paths, and the caveats
+behind each number, read the source pages:
+
+- [docs/browser-agent-validation.md](browser-agent-validation.md) — RTX and
+ Metal browser-agent test log, in date order.
+- [docs/runtime-validation.md](runtime-validation.md) — native API and model
+ checks (Metal host).
+- [docs/playwright-agent-benchmark.md](playwright-agent-benchmark.md) — what
+ the benchmark tool tests and how to run it.
+- [docs/browser-agent-models.md](browser-agent-models.md) — candidate models,
+ their intended roles, and the checks a model must pass before promotion.
+
+None of the numbers below are invented. Each one comes from a run recorded in
+those pages. Most are single runs or small repeat counts, not a measured
+success rate. Treat every result as evidence, not as a final ranking, unless
+the text says otherwise.
+
+## Hardware
+
+| Item | Detail |
+| --- | --- |
+| RTX worker | `firesand-worker` — RTX 3070 Ti, 8,192 MiB VRAM, NVIDIA driver 595.84, about 60 GiB system RAM, runs the stack in Docker |
+| Test image | Pinned Prism `922be44` (full revision `922be44aa6ac81b46f092716351cddff1c1733a7`), CUDA 12.4.1, built for Ampere (`sm_86`) |
+| Live (production) image | Prism `4dd165625`, as recorded at the time of testing |
+| Comparison host | Apple M5, Metal, 24 GB unified memory |
+
+Metal figures are **not** RTX VRAM figures. Apple's unified memory and an
+NVIDIA card's VRAM are different things; a number from one host never stands
+in for the other.
+
+Every RTX check paused only the live model container for the length of its
+own test, used a loopback-only test port, and restored the live container
+afterward. Its authenticated `/health` endpoint was checked after each
+restore.
+
+## Models tried
+
+| Model | Role | RTX browser result so far |
+| --- | --- | --- |
+| `qwen3.5-9b-q4_k_m` | Primary fast Playwright agent; provisional everyday choice | Passed 12/12 in the matched core-workflow repeat. Passed 7/10, then 8/10, in the broader ten-case set; the failures were JSON formatting or a context limit, not wrong facts. |
+| `qwen3.5-4b-q4_k_m` | Very fast worker, for simple deterministic workflows | Passed 12/12 in the matched core-workflow repeat, about 17% faster overall than 9B on that narrow set. Passed 5/6 selected workflows in an earlier single run; of four more cases, only wrong-state recovery passed, the rest had correct facts but failed strict JSON. |
+| `qwen3.6-35b-a3b` (Q4_K_XL) | Complex agent / planner candidate | First pass, original prompt: 4/6. After a prompt fix: basic form passed in 17.2 s with zero tool errors, but extraction JSON was still fenced and an injected failure was retried before inspection, so those two tasks still failed. |
+| `qwen3.6-35b-a3b` (IQ4_XS) | Same role, alternate quant | Passed basic form, three-step form, and the synthetic job application, all with zero tool errors. Too few trials yet to choose over Q4_K_XL. |
+| `gemma4-e4b` | Fast full-GPU challenger | Failed its first live task. After a prompt fix, passed 5 of 10 single-run cases. Not recommended as a browser default despite fast token generation. |
+| `granite-4.1-8b-q4_k_m` | Structured tool-call baseline | Passed 6 of 10 single-run cases. Useful as a structured-output comparison, not yet a faster reliable default. |
+
+## Results
+
+### Fixture suite: core-workflow repeats, 2026-09-25
+
+Both Qwen3.5 models ran the same four core tasks (basic form, three-step
+form, injected-error recovery, synthetic job application) three times each,
+at 8K context, on the pinned test image. This is 12 trials per model, not a
+production-wide reliability estimate.
+
+| Result | Qwen3.5 4B | Qwen3.5 9B |
+| --- | ---: | ---: |
+| Successful / total | 12 / 12 | 12 / 12 |
+| Total wall time | 127.37 s | 152.71 s |
+| Median / p95 successful task | 8.53 / 18.43 s | 10.78 / 22.16 s |
+| Model inference / browser execution | 100.4 / 27.0 s | 128.3 / 24.3 s |
+| LLM turns | 144 | 129 |
+| Tool errors / wrong arguments | 9 / 3 | 6 / 0 |
+| Mixed cached-turn prompt / generation speed | 1,372 / 135 t/s | 1,155 / 88 t/s |
+
+Qwen3.5 4B finished this set about 17% faster overall. Qwen3.5 9B used fewer
+turns and made no wrong-argument calls. The broader ten-case runs still favor
+9B for strict extraction and decisions; both models have JSON-formatting
+failures, but 4B failed more of those cases.
+
+### 32K context, one agent, 2026-09-25
+
+Qwen3.5 9B loaded with one 32,768-token slot, f16 GPU K/V cache.
+
+| Setting | Value |
+| --- | --- |
+| Idle VRAM | 6,257 MiB |
+| VRAM during browser work | 6,269 / 8,192 MiB |
+| Synthetic job application | PASS, 15.7 s |
+| Large-page task | Prompt reached 15,568 tokens; finished in 11.1 s with no context or out-of-memory error; failed the strict-output check because the answer had prose plus a fenced JSON block |
+
+This proves the model can load at 32K and accept a large request. It does not
+prove strict-extraction reliability or peak memory once a browser also runs
+on the worker's GPU.
+
+### Two 24K slots, concurrency, 2026-09-25
+
+The same model loaded with two 24,576-token slots (49,152 total), so two
+agents could run at once.
+
+| Setting | Value |
+| --- | --- |
+| Idle VRAM | 6,811 MiB |
+| VRAM during simultaneous work | 6,823 / 8,192 MiB |
+| Concurrent runs | Two pairs of synthetic-application agents (4 agents total); 2 / 4 passed |
+| Passing run times | 24.6 s and 24.9 s — slower than the 15.7 s single-slot run above |
+| Failure cause | Both failures filled the Name field with "Ada Lovelace" instead of "Ada," reached "Application incomplete," then hit the 30-turn cap trying to recover |
+
+### Serial control, same two-slot server, 2026-09-25
+
+A matched control ran the same server, image, and flags, but with one agent
+at a time instead of two.
+
+| Setting | Value |
+| --- | --- |
+| Result | 3 / 4 passed |
+| Passing run times | 17.7 s, 14.9 s, 15.0 s (18 turns each, peak prompt about 4,870 tokens) |
+| One failure | Same Name-field error as above; hit the 30-turn cap after 25.3 s |
+
+The Name-field error happens with one agent alone, not only with two agents
+at once, so it is a task error, not a concurrency error. Four trials each way
+are too few to say whether concurrency makes it more common (2/4 versus 1/4
+failures). Running two agents at once did add about 9 seconds to each
+passing run.
+
+### 64K context, 2026-09-26
+
+Qwen3.5 9B loaded with one 65,536-token slot, f16 GPU K/V cache, all layers
+on the GPU.
+
+| Setting | Value |
+| --- | --- |
+| Idle VRAM | 7,313 / 8,192 MiB |
+| Prompt size tested | 59,906 tokens |
+| Result | PASS, 26.1 s — the model quoted the requested line correctly |
+| Peak VRAM | 7,323 MiB (about 870 MiB free) |
+
+870 MiB free is enough for inference alone. A browser running on the same GPU
+at the same time is not measured here.
+
+### Live web: DuckDuckGo image search, 2026-09-26
+
+A one-off harness gave the model real headless Chrome through Playwright MCP
+0.0.82: search DuckDuckGo for oranges, open Images, pick a picture, find its
+full-size URL, and call a `save_image` tool. That tool only accepts real
+JPEG/PNG/GIF/WebP bytes, and navigation stayed on duckduckgo.com.
+
+| Run | Context | Result | Time | Note |
+| --- | --- | --- | --- | --- |
+| 1 | 32K | FAIL | 55.3 s | Context overflow; DuckDuckGo showed no images |
+| 2 | 32K | PASS | 43.7 s | Saved an orange photo (verywellhealth.com) |
+| 3 | 64K | FAIL | 133.6 s | Found image URLs but wrote its plan as text instead of calling `save_image` |
+| 4 | 64K | PASS | 69.5 s | Saved an orange photo (wallpapers.com) |
+| 5 | 64K | PASS | 42.7 s | Saved the same photo as run 2 |
+
+With the context fixes described below, 3 of 4 runs passed. The one failure
+was a model error (it did not call the save tool), not a context error.
+
+## Known limits
+
+| Limit | What happens | Status |
+| --- | --- | --- |
+| Name-field error | The model fills the Name field with "Ada Lovelace" instead of "Ada." The form reports "Application incomplete," and the agent exhausts its 30-turn cap trying to recover. | Not fixed. Happens alone and alongside another agent, at about the same rate. Not a concurrency bug. |
+| Fenced JSON | The model gets the facts and the decision right, but wraps its JSON answer in a Markdown code fence. The strict-JSON check does not accept a fence, so the task fails even though the answer is correct. | Not fixed. Seen across Qwen3.5 9B, Qwen3.5 4B, and Qwen3.6 35B A3B. Needs a prompt change or a parser that strips fences. |
+| DuckDuckGo headless block | Headless Chrome's default user agent gets "No images found" from DuckDuckGo, even though a normal browser gets results. | Fixed. Set a normal Chrome user agent with `--user-agent`. |
+| Snapshot size | The DuckDuckGo Images page snapshot is about 175,000 characters, mostly ad-tracking links. A full run with all of that text overflows a 32K context after about 22 turns. | Fixed for this task. Shorten links over 300 characters and keep only the newest large page view in history. Runs then peak at 14.6K–24.3K prompt tokens. |
+| Kernel/driver trap | After a worker reboot to a new kernel, the matching NVIDIA driver module was missing, so the live model container could not start. | Fixed. Install the matching `linux-modules-nvidia` package and load the module. Check `nvidia-smi` after any kernel update, before a benchmark run. |
+
+## Recommendations
+
+- Keep `qwen3.5-9b-q4_k_m` as the **provisional** everyday browser agent. Use
+ `qwen3.5-4b-q4_k_m` for short, simple, deterministic workflows. Do not
+ switch models between browser actions in the same task.
+- For long job-application work, use one 32K Qwen3.5 9B agent and queue
+ additional jobs rather than run two agents at once. Two 24K agents are a
+ memory-feasible experiment, not a reliable production setting yet.
+- Do not carry the 32K result over to other model presets, and do not treat
+ it as proof of GPU headroom once a browser also runs on the worker.
+- Do not promote Gemma 4 E4B as a fast default. Its API probes pass, but its
+ browser task pass rate is too low.
+- Granite 4.1 8B is a good structured-output comparison, but not yet a faster
+ reliable browser default.
+- Run more repeated strict-JSON and long-page trials before naming a final
+ winner between Qwen3.5 9B and 4B, and before promoting Qwen3.6 35B A3B or
+ choosing between its Q4_K_XL and IQ4_XS quants.
+
+See [docs/browser-agent-validation.md](browser-agent-validation.md) for the
+full log, including the earlier Apple Metal checks and every raw-report path.
diff --git a/docs/playwright-agent-benchmark.md b/docs/playwright-agent-benchmark.md
index 224dcd2..061a67a 100644
--- a/docs/playwright-agent-benchmark.md
+++ b/docs/playwright-agent-benchmark.md
@@ -56,6 +56,10 @@ tools to the model. Use `--force-first-tool` only for a separate required-choice
comparison; the rest of each task uses `tool_choice=auto`. The default is auto
for every turn.
+See [docs/benchmarks.md](benchmarks.md) for consolidated findings across
+models, and [docs/benchmark-action.md](benchmark-action.md) for the manual
+GitHub Actions job that runs this suite on the RTX worker.
+
The benchmark uses the production MCP launcher with `--snapshot-mode none` and
`--image-responses omit`. `browser_snapshot` and `browser_find` are available
for deliberate inspection. It fails if an ordinary tool response contains an
diff --git a/tests/bench/__init__.py b/tests/bench/__init__.py
new file mode 100644
index 0000000..e69de29
diff --git a/tests/bench/collect.py b/tests/bench/collect.py
new file mode 100644
index 0000000..1a4873e
--- /dev/null
+++ b/tests/bench/collect.py
@@ -0,0 +1,292 @@
+#!/usr/bin/env python3
+"""Turn a raw benchmark report directory into a run summary, and roll a
+directory of run summaries into an index — both per contracts.md.
+
+Stdlib only. Every read is tolerant of a missing or corrupt file: a suite
+that never wrote its report simply stays out of the summary (or, if the
+run's meta.json asked for it or errors.json explains why it is missing, it
+appears with zero counts and an "error"), and a corrupt run-summary file is
+skipped when building the index rather than failing the whole build.
+
+ collect.py summarize RAW_DIR --out SUMMARY.json
+ collect.py index RUNS_DIR --out INDEX.json
+"""
+import argparse
+import glob
+import json
+import os
+import re
+import statistics
+import sys
+
+SCHEMA = 1
+
+# Suite keys as they appear in a run summary's "suites" map.
+SUITE_KEYS = ("fixture", "long_context", "live_web",
+ "concurrency_serial", "concurrency_parallel")
+
+# meta.json's "suites" (requested) and errors.json's keys name the suite as
+# run_suites.py sees it, where "concurrency" covers both concurrency phases.
+_ALIASES = {"concurrency": ("concurrency_serial", "concurrency_parallel")}
+
+
+def _expand_aliases(names):
+ expanded = set()
+ for name in names:
+ expanded.update(_ALIASES.get(name, (name,)))
+ return expanded
+
+
+def load_json(path):
+ """Parse path as JSON; return None if it is missing, unreadable, or not
+ valid JSON. Never raises."""
+ try:
+ with open(path, "r", encoding="utf-8") as f:
+ return json.load(f)
+ except (OSError, json.JSONDecodeError, UnicodeDecodeError):
+ return None
+
+
+def _natural_key(path):
+ """Sort .../live_web-2.json before .../live_web-10.json."""
+ match = re.search(r"(\d+)(?=\.json$)", path)
+ return (int(match.group(1)) if match else -1, path)
+
+
+def _glob_reports(dir_path, pattern):
+ """Load every file matching pattern inside dir_path, in numeric-suffix
+ order, dropping any file that is missing or fails to parse."""
+ paths = sorted(glob.glob(os.path.join(dir_path, pattern)), key=_natural_key)
+ reports = []
+ for path in paths:
+ data = load_json(path)
+ if data is not None:
+ reports.append(data)
+ return reports
+
+
+def median_or_none(values):
+ values = [v for v in values if v is not None]
+ if not values:
+ return None
+ return statistics.median(values)
+
+
+def pass_rate(passed, total):
+ if not total:
+ return 0.0
+ return round(passed / total * 100, 1)
+
+
+def _round_or_none(value, digits=1):
+ return None if value is None else round(value, digits)
+
+
+def _int_or_none(value):
+ return None if value is None else int(value)
+
+
+def _vram_peak(dir_path, *candidate_names):
+ """peak_mib from the first vram-*.json in candidate_names that exists."""
+ for name in candidate_names:
+ data = load_json(os.path.join(dir_path, name))
+ if data is not None:
+ return _int_or_none(data.get("peak_mib"))
+ return None
+
+
+def _oracle_stats(results):
+ """passed/total/median_seconds/peak_prompt_tokens/peak_vram_mib for a
+ flat list of playwright_agent_bench.py-shaped result dicts (fixture.json
+ and the concurrency-*.json reports share this shape): status is an exact
+ "PASS"/"FAIL" oracle verdict, tokens live at turns[].usage.prompt_tokens.
+ """
+ passed = sum(1 for r in results if r.get("status") == "PASS")
+ total = len(results)
+ seconds = median_or_none([r.get("seconds") for r in results])
+ tokens = [
+ t.get("usage", {}).get("prompt_tokens")
+ for r in results
+ for t in r.get("turns", [])
+ ]
+ tokens = [t for t in tokens if t is not None]
+ peak_tokens = max(tokens) if tokens else None
+ vram_from_results = [r.get("peak_vram_mib") for r in results if r.get("peak_vram_mib") is not None]
+ peak_vram = max(vram_from_results) if vram_from_results else None
+ return passed, total, seconds, peak_tokens, peak_vram
+
+
+def _live_web_stats(reports):
+ """live_image_agent.py reports: status starts with "PASS" when an image
+ was saved, tokens live flat at turns[].prompt_tokens."""
+ passed = sum(1 for r in reports if str(r.get("status", "")).startswith("PASS"))
+ total = len(reports)
+ seconds = median_or_none([r.get("seconds") for r in reports])
+ tokens = [
+ t.get("prompt_tokens")
+ for r in reports
+ for t in r.get("turns", [])
+ ]
+ tokens = [t for t in tokens if t is not None]
+ peak_tokens = max(tokens) if tokens else None
+ return passed, total, seconds, peak_tokens
+
+
+def _suite_dict(passed, total, seconds, peak_tokens, peak_vram, error=None,
+ settings=None):
+ settings = settings or {}
+ suite = {
+ "passed": passed,
+ "total": total,
+ "pass_rate": pass_rate(passed, total),
+ "median_seconds": _round_or_none(seconds),
+ "peak_vram_mib": _int_or_none(peak_vram),
+ "peak_prompt_tokens": _int_or_none(peak_tokens),
+ # The server this suite ran against (the phases use different ones).
+ "ctx": settings.get("ctx"),
+ "parallel": settings.get("parallel"),
+ }
+ if error:
+ suite["error"] = error
+ return suite
+
+
+def summarize(dir_path):
+ """Build the run-summary dict for the raw report directory dir_path."""
+ meta = load_json(os.path.join(dir_path, "meta.json")) or {}
+ errors = load_json(os.path.join(dir_path, "errors.json")) or {}
+
+ requested = _expand_aliases(meta.get("suites") or [])
+ erroring = _expand_aliases(errors.keys())
+
+ def error_for(suite_key, top_level_key):
+ return errors.get(suite_key) or errors.get(top_level_key)
+
+ suite_settings = meta.get("suite_settings") or {}
+
+ def settings_for(top_level_key):
+ """meta.json's suite_settings entry for the suite as run_suites.py
+ names it; each field falls back to the run's own ctx/parallel."""
+ own = suite_settings.get(top_level_key) or {}
+ return {"ctx": own.get("ctx", meta.get("ctx")),
+ "parallel": own.get("parallel", meta.get("parallel"))}
+
+ suites = {}
+
+ # fixture
+ fixture_report = load_json(os.path.join(dir_path, "fixture.json"))
+ fixture_present = fixture_report is not None
+ if "fixture" in requested or fixture_present or "fixture" in erroring:
+ if fixture_present:
+ passed, total, seconds, peak_tokens, peak_vram = _oracle_stats(
+ fixture_report.get("results", []))
+ else:
+ passed, total, seconds, peak_tokens, peak_vram = 0, 0, None, None, None
+ if peak_vram is None:
+ peak_vram = _vram_peak(dir_path, "vram-fixture.json")
+ suites["fixture"] = _suite_dict(
+ passed, total, seconds, peak_tokens, peak_vram,
+ error=error_for("fixture", "fixture"), settings=settings_for("fixture"))
+
+ # long_context
+ lc_report = load_json(os.path.join(dir_path, "long_context.json"))
+ lc_present = lc_report is not None
+ if "long_context" in requested or lc_present or "long_context" in erroring:
+ if lc_present:
+ passed = 1 if lc_report.get("status") == "PASS" else 0
+ total = 1
+ seconds = lc_report.get("seconds")
+ peak_tokens = lc_report.get("prompt_tokens")
+ else:
+ passed, total, seconds, peak_tokens = 0, 0, None, None
+ peak_vram = _vram_peak(dir_path, "vram-long_context.json")
+ suites["long_context"] = _suite_dict(
+ passed, total, seconds, peak_tokens, peak_vram,
+ error=error_for("long_context", "long_context"), settings=settings_for("long_context"))
+
+ # live_web (one or more live_web-.json repetitions)
+ live_reports = _glob_reports(dir_path, "live_web-*.json")
+ if "live_web" in requested or live_reports or "live_web" in erroring:
+ passed, total, seconds, peak_tokens = _live_web_stats(live_reports)
+ peak_vram = _vram_peak(dir_path, "vram-live_web.json")
+ suites["live_web"] = _suite_dict(
+ passed, total, seconds, peak_tokens, peak_vram,
+ error=error_for("live_web", "live_web"), settings=settings_for("live_web"))
+
+ # concurrency: serial and parallel phases, each its own suite key.
+ # run_suites.py samples VRAM once for the whole concurrency suite (both
+ # phases run back to back under one VramSampler) and writes a single
+ # vram-concurrency.json, not a per-phase file — so both phases share
+ # that one peak.
+ concurrency_peak_vram = _vram_peak(dir_path, "vram-concurrency.json")
+ for phase, pattern in (
+ ("concurrency_serial", "concurrency-serial-*.json"),
+ ("concurrency_parallel", "concurrency-parallel-*.json"),
+ ):
+ phase_reports = _glob_reports(dir_path, pattern)
+ results = [r for report in phase_reports for r in report.get("results", [])]
+ if phase in requested or phase_reports or phase in erroring:
+ if phase_reports:
+ passed, total, seconds, peak_tokens, peak_vram = _oracle_stats(results)
+ else:
+ passed, total, seconds, peak_tokens, peak_vram = 0, 0, None, None, None
+ if peak_vram is None:
+ peak_vram = concurrency_peak_vram
+ suites[phase] = _suite_dict(
+ passed, total, seconds, peak_tokens, peak_vram,
+ error=error_for(phase, "concurrency"),
+ settings=settings_for("concurrency"))
+
+ return {
+ "schema": SCHEMA,
+ "run_id": meta.get("run_id"),
+ "created_utc": meta.get("created_utc"),
+ "commit": meta.get("commit"),
+ "model": meta.get("model"),
+ "ctx": meta.get("ctx"),
+ "gpu": meta.get("gpu"),
+ "suites": suites,
+ }
+
+
+def build_index(runs_dir):
+ """Roll every *.json run-summary file in runs_dir into the index."""
+ runs = []
+ for path in sorted(glob.glob(os.path.join(runs_dir, "*.json"))):
+ data = load_json(path)
+ if isinstance(data, dict) and "suites" in data:
+ runs.append(data)
+ runs.sort(key=lambda r: r.get("created_utc") or "")
+ return {"schema": SCHEMA, "runs": runs}
+
+
+def _write_json(obj, out_path):
+ os.makedirs(os.path.dirname(os.path.abspath(out_path)) or ".", exist_ok=True)
+ with open(out_path, "w", encoding="utf-8") as f:
+ json.dump(obj, f, indent=2)
+ f.write("\n")
+
+
+def main(argv=None):
+ parser = argparse.ArgumentParser(prog="collect.py", description=__doc__)
+ sub = parser.add_subparsers(dest="command", required=True)
+
+ p_summarize = sub.add_parser("summarize", help="summarize one raw report directory")
+ p_summarize.add_argument("dir", help="raw report directory (run_suites.py --out-dir)")
+ p_summarize.add_argument("--out", required=True, help="path to write the run summary JSON")
+
+ p_index = sub.add_parser("index", help="rebuild the index from a directory of run summaries")
+ p_index.add_argument("dir", help="directory of run-summary JSON files")
+ p_index.add_argument("--out", required=True, help="path to write index.json")
+
+ args = parser.parse_args(argv)
+
+ if args.command == "summarize":
+ _write_json(summarize(args.dir), args.out)
+ elif args.command == "index":
+ _write_json(build_index(args.dir), args.out)
+ return 0
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/tests/bench/run_suites.py b/tests/bench/run_suites.py
new file mode 100755
index 0000000..8d47c1a
--- /dev/null
+++ b/tests/bench/run_suites.py
@@ -0,0 +1,479 @@
+#!/usr/bin/env python3
+"""Run the RTX-worker benchmark suites against an already-running server.
+
+Runs each requested suite as a subprocess of the existing test scripts
+(tests/playwright_agent_bench.py, tests/long_context_probe.py,
+tests/live_image_agent.py), samples VRAM every 2 seconds through an external
+command (typically `tests/bench/worker.sh vram`, run over SSH by the
+caller), and writes the raw report files described in
+.superpowers/sdd/2026-09-26-benchmark-action/contracts.md to --out-dir.
+
+Run from the repository root: python3 tests/bench/run_suites.py ...
+"""
+import argparse
+import datetime
+import json
+import math
+import os
+from pathlib import Path
+import signal
+import subprocess
+import sys
+import threading
+import time
+from types import SimpleNamespace
+
+ROOT = Path(__file__).resolve().parents[2]
+SUITE_ORDER = ("fixture", "long_context", "live_web", "concurrency")
+# The exact flags recorded from the validated concurrency runs: one
+# job_application task, no repeats, a 24,576-token per-slot budget (half of
+# the 49,152-token, --parallel 2 context the worker is restarted with).
+CONCURRENCY_FLAGS = ("--tasks", "job_application", "--repetitions", "1",
+ "--task-timeout", "120", "--max-tokens", "512",
+ "--context", "24576", "--reasoning", "off")
+
+# Per-suite job timeouts scale with the work each suite actually schedules
+# instead of one flat number: a flat 900s killed normal fixture runs (10
+# cases per repetition x up to 300s each is already 3,000s for a single
+# repetition) while leaving a genuinely hung single-request suite waiting
+# needlessly long. These mirror the *other* scripts' own defaults, which
+# run_suites.py does not override for fixture/long_context/live_web.
+FIXTURE_CASES = 10 # playwright_agent_bench.py's default 9 tasks, +1 extra
+ # variant because conditional_application runs twice.
+DEFAULT_TASK_TIMEOUT = 300 # playwright_agent_bench.py --task-timeout default
+LONG_CONTEXT_REQUEST_TIMEOUT = 120 # long_context_probe.py --request-timeout default
+LIVE_WEB_TASK_TIMEOUT = 300 # live_image_agent.py --task-timeout default
+CONCURRENCY_TASK_TIMEOUT = 120 # matches --task-timeout in CONCURRENCY_FLAGS
+STARTUP_MARGIN = 120 # model/MCP/browser start-up overhead, multi-case suites
+JOB_STARTUP_MARGIN = 60 # same, for suites that are a single request/task
+# The worker.sh `remote()` helper uses `ssh -o ConnectTimeout=15`; the VRAM
+# sample's own subprocess timeout must exceed that or it kills the SSH
+# connection before it even finishes connecting.
+VRAM_SAMPLE_TIMEOUT = 20
+# The concurrency suite: four one-at-a-time runs, then two concurrent pairs.
+CONCURRENCY_SERIAL_RUNS = 4
+CONCURRENCY_PAIRS = 2
+# Worst-case extra time per batch past its job timeout: _kill_process_group
+# waits up to 10 s + 5 s per process, and a pair kills two in turn.
+BATCH_KILL_GRACE = 30
+
+# Workflow time budget (.github/workflows/benchmark.yml). The job has
+# JOB_TIMEOUT_MINUTES in total. RESERVED_MINUTES covers everything that is
+# not a suite step: set-up and tailnet join (~10), both `worker.sh up` steps
+# (10 each, their own step caps), merge/summarize/uploads (~5), and the
+# `worker.sh down` restore step (20, its own step cap). A suite step's cap is
+# its worst case plus STEP_MARGIN_MINUTES, so the step cap only fires if
+# run_suites.py itself hangs.
+JOB_TIMEOUT_MINUTES = 240
+RESERVED_MINUTES = 60
+STEP_MARGIN_MINUTES = 5
+
+
+def job_timeout_for(suite, args):
+ """Pure: the wall-clock cap for one job of `suite`, given how much work
+ run_suites.py itself asked that job to do (repetitions, in particular)."""
+ if suite == "fixture":
+ return FIXTURE_CASES * args.repetitions * DEFAULT_TASK_TIMEOUT + STARTUP_MARGIN
+ if suite == "long_context":
+ return LONG_CONTEXT_REQUEST_TIMEOUT + JOB_STARTUP_MARGIN
+ if suite == "live_web":
+ return LIVE_WEB_TASK_TIMEOUT + JOB_STARTUP_MARGIN
+ if suite == "concurrency":
+ return CONCURRENCY_TASK_TIMEOUT + JOB_STARTUP_MARGIN
+ raise ValueError(f"unknown suite: {suite}")
+
+
+def batch_count(suite, repetitions):
+ """Pure: how many batches build_plan() schedules for `suite`."""
+ if suite in ("fixture", "long_context"):
+ return 1
+ if suite == "live_web":
+ return repetitions
+ if suite == "concurrency":
+ return CONCURRENCY_SERIAL_RUNS + CONCURRENCY_PAIRS
+ raise ValueError(f"unknown suite: {suite}")
+
+
+def worst_case_seconds(suites, repetitions):
+ """Pure: the longest one run_suites.py invocation for `suites` can take
+ before every job has hit its own timeout and been killed."""
+ args = SimpleNamespace(repetitions=repetitions)
+ return sum(batch_count(suite, repetitions) * (job_timeout_for(suite, args) + BATCH_KILL_GRACE)
+ for suite in suites)
+
+
+def step_minutes(suites, repetitions):
+ """Pure: the workflow step cap (whole minutes) for one invocation; 0 when
+ there are no suites (the step is skipped)."""
+ if not suites:
+ return 0
+ return math.ceil(worst_case_seconds(suites, repetitions) / 60) + STEP_MARGIN_MINUTES
+
+
+def fits_job(phase1, phase2, repetitions):
+ """Pure: whether both suite steps plus the reserve fit in the job."""
+ return (step_minutes(phase1, repetitions) + step_minutes(phase2, repetitions)
+ + RESERVED_MINUTES) <= JOB_TIMEOUT_MINUTES
+
+
+def max_repetitions(phase1, phase2, limit=1000):
+ """Pure: the largest repetitions value that still fits the job (0 when
+ even 1 does not)."""
+ best = 0
+ for repetitions in range(1, limit + 1):
+ if not fits_job(phase1, phase2, repetitions):
+ break
+ best = repetitions
+ return best
+
+
+def _kill_process_group(process):
+ """Kill a whole process group, not just the direct child: bench scripts
+ launch Playwright MCP/Chrome as their own children, and Popen.kill()
+ (or subprocess.run(timeout=...)'s internal kill) only reaches the
+ process we spawned directly, orphaning the rest. Requires the process to
+ have been started with start_new_session=True."""
+ try:
+ os.killpg(process.pid, signal.SIGTERM)
+ except OSError:
+ pass
+ try:
+ process.wait(timeout=10)
+ return
+ except subprocess.TimeoutExpired:
+ pass
+ try:
+ os.killpg(process.pid, signal.SIGKILL)
+ except OSError:
+ pass
+ try:
+ process.wait(timeout=5)
+ except subprocess.TimeoutExpired:
+ pass
+
+
+def fixture_command(args, output, root=ROOT):
+ return [sys.executable, str(root / "tests" / "playwright_agent_bench.py"),
+ "--base-url", args.base_url, "--model", args.model, "--output", str(output),
+ "--key-env", args.key_env, "--repetitions", str(args.repetitions),
+ "--context", str(args.ctx), "--reasoning", "off"]
+
+
+def long_context_command(args, output, root=ROOT):
+ return [sys.executable, str(root / "tests" / "long_context_probe.py"),
+ "--base-url", args.base_url, "--model", args.model,
+ "--target-tokens", str(args.ctx), "--output", str(output),
+ "--key-env", args.key_env]
+
+
+def live_web_command(args, output, image_dir, root=ROOT):
+ return [sys.executable, str(root / "tests" / "live_image_agent.py"),
+ "--base-url", args.base_url, "--model", args.model, "--output", str(output),
+ "--key-env", args.key_env, "--image-dir", str(image_dir), "--reasoning", "off"]
+
+
+def concurrency_command(args, output, root=ROOT):
+ return [sys.executable, str(root / "tests" / "playwright_agent_bench.py"),
+ "--base-url", args.base_url, "--model", args.model, "--output", str(output),
+ "--key-env", args.key_env, *CONCURRENCY_FLAGS]
+
+
+def build_plan(args, out_dir, root=ROOT):
+ """Pure: which subprocess commands to run for which requested suites.
+
+ Returns (batches, errors). Each batch is a list of one or more job dicts
+ ({"suite", "output", "cmd"}) meant to be launched together (more than
+ one job in a batch means they run concurrently). `errors` maps a suite
+ name to why it could not even be planned (currently: concurrency without
+ --parallel 2).
+ """
+ batches = []
+ errors = {}
+ requested = set(args.suites)
+ if "fixture" in requested:
+ output = out_dir / "fixture.json"
+ batches.append([{"suite": "fixture", "output": output,
+ "cmd": fixture_command(args, output, root)}])
+ if "long_context" in requested:
+ output = out_dir / "long_context.json"
+ batches.append([{"suite": "long_context", "output": output,
+ "cmd": long_context_command(args, output, root)}])
+ if "live_web" in requested:
+ image_dir = out_dir / "live-images"
+ for n in range(1, args.repetitions + 1):
+ output = out_dir / f"live_web-{n}.json"
+ batches.append([{"suite": "live_web", "output": output,
+ "cmd": live_web_command(args, output, image_dir, root)}])
+ if "concurrency" in requested:
+ if args.parallel != 2:
+ errors["concurrency"] = f"concurrency requires --parallel 2 (got {args.parallel})"
+ else:
+ for n in range(1, CONCURRENCY_SERIAL_RUNS + 1):
+ output = out_dir / f"concurrency-serial-{n}.json"
+ batches.append([{"suite": "concurrency", "output": output,
+ "cmd": concurrency_command(args, output, root)}])
+ for pair in range(CONCURRENCY_PAIRS):
+ group = []
+ for slot in range(2):
+ n = pair * 2 + slot + 1
+ output = out_dir / f"concurrency-parallel-{n}.json"
+ group.append({"suite": "concurrency", "output": output,
+ "cmd": concurrency_command(args, output, root)})
+ batches.append(group)
+ return batches, errors
+
+
+def parse_vram_sample(text):
+ """Pure: the worker prints the MiB integer, possibly after SSH banner
+ noise on earlier lines. None means the sample failed or was unreadable."""
+ if not text:
+ return None
+ lines = [line.strip() for line in text.strip().splitlines() if line.strip()]
+ if not lines:
+ return None
+ try:
+ return int(lines[-1])
+ except ValueError:
+ return None
+
+
+def vram_peak(samples):
+ """Pure: the maximum of the samples that parsed; None if none did."""
+ values = [value for value in samples if isinstance(value, int)]
+ return max(values) if values else None
+
+
+def build_meta(existing, args, commit, gpu, run_id, now):
+ """Pure: meta.json for this out-dir. A second run_suites.py invocation
+ against the same --out-dir (the workflow's non-concurrency then
+ concurrency phases) merges in rather than starting over: the suites list
+ is unioned and created_utc is kept from the first invocation.
+ suite_settings records the server ctx/parallel each suite actually ran
+ with, since the two phases use different servers."""
+ existing = existing or {}
+ suites = list(dict.fromkeys((existing.get("suites") or []) + list(args.suites)))
+ suite_settings = dict(existing.get("suite_settings") or {})
+ for suite in args.suites:
+ suite_settings[suite] = {"ctx": args.ctx, "parallel": args.parallel}
+ return {"model": args.model, "ctx": args.ctx, "parallel": args.parallel,
+ "gpu": gpu, "commit": commit, "run_id": run_id,
+ "created_utc": existing.get("created_utc") or now, "suites": suites,
+ "suite_settings": suite_settings}
+
+
+class VramSampler:
+ """Samples an external command every `interval` seconds in a background
+ thread. `cmd` is a shell command string (run via the shell, typically
+ over SSH); pass a `runner` to inject a fake sampler for tests."""
+
+ def __init__(self, cmd, interval=2.0, runner=None):
+ self.cmd = cmd
+ self.interval = interval
+ self.runner = runner or self._shell_runner
+ self.stop = threading.Event()
+ self.samples = []
+ self._frozen_count = None
+ self._process_lock = threading.Lock()
+ self._process = None
+ self.thread = threading.Thread(target=self._run, daemon=True)
+
+ def _shell_runner(self):
+ try:
+ process = subprocess.Popen(self.cmd, shell=True, start_new_session=True,
+ stdout=subprocess.PIPE, stderr=subprocess.DEVNULL,
+ text=True)
+ except OSError:
+ return None
+ with self._process_lock:
+ self._process = process
+ try:
+ stdout, _ = process.communicate(timeout=VRAM_SAMPLE_TIMEOUT)
+ except subprocess.TimeoutExpired:
+ _kill_process_group(process)
+ return None
+ finally:
+ with self._process_lock:
+ self._process = None
+ return parse_vram_sample(stdout)
+
+ def _run(self):
+ while not self.stop.is_set():
+ self.samples.append(self.runner())
+ self.stop.wait(self.interval)
+
+ def __enter__(self):
+ if self.cmd is not None:
+ self.thread.start()
+ return self
+
+ def __exit__(self, *exc_info):
+ self.stop.set()
+ # An in-flight sample can block for up to VRAM_SAMPLE_TIMEOUT; do not
+ # wait that long here — kill it so shutdown stays bounded.
+ with self._process_lock:
+ process = self._process
+ if process is not None:
+ _kill_process_group(process)
+ if self.thread.is_alive():
+ self.thread.join(timeout=5)
+ # Freeze the sample count now: a sample that still lands after this
+ # (the thread refusing to die, or a scheduling fluke) belongs to
+ # whatever runs next, not to the report we are about to write.
+ self._frozen_count = len(self.samples)
+
+ def report(self):
+ count = self._frozen_count if self._frozen_count is not None else len(self.samples)
+ values = [value for value in self.samples[:count] if isinstance(value, int)]
+ return {"samples_mib": values, "peak_mib": vram_peak(values)}
+
+
+def _finalize(job, timed_out, timeout):
+ """After a subprocess has been waited on (and killed if it timed out),
+ decide whether its output counts as a kept result or a crash. A timeout
+ is not by itself a reason to discard a report: playwright_agent_bench.py
+ writes fixture.json after every completed case, so a job killed near its
+ deadline can still have a fully valid, merely incomplete, report — keep
+ it, and just note that it timed out."""
+ if job["output"].exists():
+ try:
+ json.loads(job["output"].read_text())
+ except (OSError, ValueError) as error:
+ return f"invalid report JSON: {error}"
+ if timed_out:
+ return f"job timed out after {timeout:.0f}s; kept the partial report"
+ return None
+ if timed_out:
+ return f"job timed out after {timeout:.0f}s"
+ return "subprocess exited without writing a report"
+
+
+def run_job(job, timeout=None):
+ """Run one subprocess to completion in its own process group, so a
+ timeout can kill Playwright/Chrome children too, not just the Python
+ process. Returns None on success (including when the benchmark itself
+ recorded a FAIL — that is a normal result, not a crash) or a short
+ reason string when the job crashed or timed out."""
+ try:
+ process = subprocess.Popen(job["cmd"], start_new_session=True)
+ except OSError as error:
+ return f"{type(error).__name__}: {error}"
+ timed_out = False
+ try:
+ process.wait(timeout=timeout)
+ except subprocess.TimeoutExpired:
+ _kill_process_group(process)
+ timed_out = True
+ return _finalize(job, timed_out, timeout)
+
+
+def run_batch(batch, timeout=None):
+ """Run every job in a batch (concurrently when there is more than one),
+ each in its own process group. The batch's deadline starts when the
+ batch starts, not afresh at each job's wait() — otherwise two jobs in a
+ pair could together run for up to 2x timeout instead of sharing it, the
+ same way `for job in batch: job.wait(timeout=timeout)` would. Returns
+ (suite, reason) — reason is the first crash/timeout found, or None."""
+ suite = batch[0]["suite"]
+ if len(batch) == 1:
+ return suite, run_job(batch[0], timeout)
+ processes = []
+ reason = None
+ for job in batch:
+ try:
+ processes.append((job, subprocess.Popen(job["cmd"], start_new_session=True)))
+ except OSError as error:
+ reason = reason or f"{type(error).__name__}: {error}"
+ deadline = None if timeout is None else time.monotonic() + timeout
+ for job, process in processes:
+ remaining = None if deadline is None else max(0, deadline - time.monotonic())
+ timed_out = False
+ try:
+ process.wait(timeout=remaining)
+ except subprocess.TimeoutExpired:
+ _kill_process_group(process)
+ timed_out = True
+ job_reason = _finalize(job, timed_out, timeout)
+ if job_reason and reason is None:
+ reason = job_reason
+ return suite, reason
+
+
+def get_commit(root):
+ try:
+ result = subprocess.run(["git", "rev-parse", "--short", "HEAD"], cwd=root,
+ capture_output=True, text=True, timeout=5, check=True)
+ commit = result.stdout.strip()
+ if commit:
+ return commit
+ except (OSError, subprocess.SubprocessError):
+ pass
+ return os.environ.get("GITHUB_SHA", "")[:7] or None
+
+
+def run(args):
+ out_dir = Path(args.out_dir).resolve()
+ out_dir.mkdir(parents=True, exist_ok=True)
+ batches, errors = build_plan(args, out_dir)
+ ran = 0
+ for suite in SUITE_ORDER:
+ if suite not in args.suites or suite in errors:
+ continue
+ suite_batches = [batch for batch in batches if batch[0]["suite"] == suite]
+ timeout = args.job_timeout if args.job_timeout else job_timeout_for(suite, args)
+ sampler = VramSampler(args.vram_cmd)
+ crash = None
+ with sampler:
+ for batch in suite_batches:
+ ran += len(batch)
+ _, reason = run_batch(batch, timeout=timeout)
+ if reason and crash is None:
+ crash = reason
+ break
+ (out_dir / f"vram-{suite}.json").write_text(json.dumps(sampler.report(), indent=2) + "\n")
+ if crash:
+ errors[suite] = crash
+ if errors:
+ (out_dir / "errors.json").write_text(json.dumps(errors, indent=2) + "\n")
+ meta_path = out_dir / "meta.json"
+ existing = None
+ if meta_path.exists():
+ try:
+ existing = json.loads(meta_path.read_text())
+ except (OSError, ValueError):
+ existing = None
+ gpu = args.gpu or os.environ.get("BENCH_GPU_NAME") or None
+ run_id = args.run_id or os.environ.get("GITHUB_RUN_ID") or str(int(time.time()))
+ now = datetime.datetime.now(datetime.timezone.utc).isoformat()
+ meta = build_meta(existing, args, get_commit(ROOT), gpu, run_id, now)
+ meta_path.write_text(json.dumps(meta, indent=2) + "\n")
+ print(f"run_suites: ran {ran} job(s); errors: {sorted(errors) or 'none'}", flush=True)
+ return 0 if ran else 1
+
+
+def main():
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument("--base-url", required=True)
+ parser.add_argument("--model", required=True)
+ parser.add_argument("--ctx", type=int, required=True)
+ parser.add_argument("--parallel", type=int, default=1)
+ parser.add_argument("--suites", nargs="+", required=True,
+ choices=("fixture", "long_context", "live_web", "concurrency"))
+ parser.add_argument("--repetitions", type=int, default=1)
+ parser.add_argument("--out-dir", required=True)
+ parser.add_argument("--vram-cmd", default=None,
+ help="Shell command that prints used VRAM MiB; sampled every 2s")
+ parser.add_argument("--key-env", default="BONSAI_API_KEY")
+ parser.add_argument("--gpu", default=None, help="GPU name for meta.json; falls back to $BENCH_GPU_NAME")
+ parser.add_argument("--run-id", default=None, help="Run id for meta.json; falls back to $GITHUB_RUN_ID")
+ parser.add_argument("--job-timeout", type=float, default=None,
+ help="Override the per-suite wall-clock cap (seconds); by "
+ "default it scales with the suite's own work, see "
+ "job_timeout_for()")
+ args = parser.parse_args()
+ if args.repetitions < 1:
+ parser.error("--repetitions must be positive")
+ raise SystemExit(run(args))
+
+
+if __name__ == "__main__":
+ main()
diff --git a/tests/bench/worker.sh b/tests/bench/worker.sh
new file mode 100755
index 0000000..c8e199f
--- /dev/null
+++ b/tests/bench/worker.sh
@@ -0,0 +1,184 @@
+#!/usr/bin/env bash
+# Control the benchmark llama-server on the RTX worker over SSH.
+#
+# WORKER_SSH=user@host tests/bench/worker.sh up # stop live, start test server
+# WORKER_SSH=user@host tests/bench/worker.sh down # stop test server, restore live
+# WORKER_SSH=user@host tests/bench/worker.sh vram # used GPU memory, MiB
+# WORKER_SSH=user@host tests/bench/worker.sh health # test server /health JSON
+#
+# The GPU has room for one model, so `up` stops the live container first and
+# `down` must always run afterwards: it brings the live container back and
+# fails (exit 1) if its health check does not return 200 in time.
+#
+# The test server listens on the worker's loopback only. Reach it from the
+# caller with: ssh -f -N -L 18080:127.0.0.1:$TEST_PORT "$WORKER_SSH"
+#
+# Every docker/curl/nvidia-smi command runs ON THE WORKER, so paths such as
+# MODELS_DIR, TEMPLATE_FILE and LIVE_ENV_FILE are worker paths.
+#
+# Secrets: the live server wants a bearer key. It is read from the
+# BONSAI_API_KEY= line of LIVE_ENV_FILE on the worker and passed to curl on
+# stdin (-H @-), so it never crosses the SSH link, never appears in argv on
+# either side, and is never printed.
+set -euo pipefail
+
+LIVE_CONTAINER="${LIVE_CONTAINER:-bonsai-llama-1}"
+LIVE_HEALTH_URL="${LIVE_HEALTH_URL:-http://127.0.0.1:8080/health}"
+LIVE_ENV_FILE="${LIVE_ENV_FILE:-}"
+TEST_CONTAINER="${TEST_CONTAINER:-qcm-bench}"
+TEST_IMAGE="${TEST_IMAGE:-qcm-rtx-validation:922be44}"
+TEST_PORT="${TEST_PORT:-18080}"
+CTX="${CTX:-32768}"
+PARALLEL="${PARALLEL:-1}"
+# Tunables, mainly so tests do not wait for real timeouts.
+UP_TIMEOUT="${UP_TIMEOUT:-180}"
+DOWN_TIMEOUT="${DOWN_TIMEOUT:-120}"
+POLL_INTERVAL="${POLL_INTERVAL:-2}"
+START_ATTEMPTS="${START_ATTEMPTS:-5}"
+RETRY_SLEEP="${RETRY_SLEEP:-3}"
+
+die() { echo "worker.sh: $*" >&2; exit 1; }
+
+usage() {
+ echo "usage: WORKER_SSH=user@host $0 up|down|vram|health" >&2
+ exit 2
+}
+
+need() {
+ local name
+ for name in "$@"; do
+ [[ -n "${!name:-}" ]] || die "$name is not set"
+ done
+}
+
+is_int() { [[ "$1" =~ ^[0-9]+$ ]]; }
+
+# Run one shell command string on the worker. BatchMode: never prompt for a
+# password. StrictHostKeyChecking=yes: the host key must already be in
+# known_hosts; an unknown or changed key is a hard failure.
+remote() {
+ ssh -o BatchMode=yes -o StrictHostKeyChecking=yes -o ConnectTimeout=15 \
+ "$WORKER_SSH" "$1"
+}
+
+q() { printf '%q' "$1"; }
+
+# Poll CHECK (a function name) until it succeeds or TIMEOUT seconds pass.
+wait_for() {
+ local timeout="$1" check="$2" deadline=$((SECONDS + $1))
+ while (( SECONDS < deadline )); do
+ if "$check"; then return 0; fi
+ sleep "$POLL_INTERVAL"
+ done
+ "$check" || { echo "worker.sh: $check not ready after ${timeout}s" >&2; return 1; }
+}
+
+test_ready() {
+ local out
+ # Exit 3 from the remote side means the container is gone: with --rm a
+ # crashed llama-server removes itself, so waiting longer cannot help.
+ out="$(remote "docker inspect -f '{{.State.Running}}' $(q "$TEST_CONTAINER") >/dev/null 2>&1 || exit 3; curl -fsS --max-time 5 http://127.0.0.1:$(q "$TEST_PORT")/health" 2>/dev/null)" && return 0
+ local rc=$?
+ if (( rc == 3 )); then die "test container $TEST_CONTAINER exited during start-up"; fi
+ [[ -n "$out" ]] && echo "worker.sh: waiting for test server: $out" >&2
+ return 1
+}
+
+live_ready() {
+ local code
+ code="$(remote "$(live_health_script)" 2>/dev/null)" || true
+ [[ "$code" == "200" ]]
+}
+
+# Worker-side script that prints the live health HTTP status code. The key
+# is read and used on the worker only.
+live_health_script() {
+ local url envfile
+ url="$(q "$LIVE_HEALTH_URL")"
+ envfile="$(q "$LIVE_ENV_FILE")"
+ cat <&2
+ remote "docker stop $(q "$LIVE_CONTAINER") >/dev/null 2>&1 || true"
+
+ # A second `up` (phase 2, other slot count) replaces the running test
+ # container instead of failing on the name clash.
+ echo "worker.sh: starting $TEST_CONTAINER (ctx $CTX, parallel $PARALLEL, $MODEL_ALIAS)" >&2
+ remote "docker rm -f $(q "$TEST_CONTAINER") >/dev/null 2>&1 || true; \
+docker run -d --rm --gpus all --name $(q "$TEST_CONTAINER") \
+-p 127.0.0.1:$(q "$TEST_PORT"):8080 \
+-v $(q "$MODELS_DIR"):/models:ro \
+-v $(q "$TEMPLATE_FILE"):/template.jinja:ro \
+--entrypoint /app/src/llama.cpp-prism/build/bin/llama-server \
+$(q "$TEST_IMAGE") \
+--host 0.0.0.0 --port 8080 \
+--model $(q "$MODEL_FILE") --alias $(q "$MODEL_ALIAS") \
+--chat-template-file /template.jinja \
+--ctx-size $(q "$CTX") --n-gpu-layers 99 --flash-attn auto \
+--cache-type-k f16 --cache-type-v f16 --reasoning off \
+--parallel $(q "$PARALLEL") >/dev/null"
+
+ wait_for "$UP_TIMEOUT" test_ready || die "test server did not become healthy"
+ echo "worker.sh: test server healthy on worker 127.0.0.1:$TEST_PORT" >&2
+}
+
+# The path that must not give up: production depends on it. Each attempt
+# stops the test container (best effort: it may already be gone, or SSH may
+# drop) and starts the live one. `docker start` on a running container is a
+# no-op, so a retry after a lost reply is safe.
+cmd_down() {
+ local attempt started=0
+ is_int "$START_ATTEMPTS" && (( START_ATTEMPTS > 0 )) || START_ATTEMPTS=5
+ for (( attempt = 1; attempt <= START_ATTEMPTS; attempt++ )); do
+ echo "worker.sh: stopping $TEST_CONTAINER, starting $LIVE_CONTAINER (attempt $attempt/$START_ATTEMPTS)" >&2
+ remote "docker stop $(q "$TEST_CONTAINER") >/dev/null 2>&1 || true" \
+ || echo "worker.sh: could not reach worker to stop $TEST_CONTAINER; continuing" >&2
+ if remote "docker start $(q "$LIVE_CONTAINER") >/dev/null"; then
+ started=1
+ break
+ fi
+ if (( attempt < START_ATTEMPTS )); then sleep "$RETRY_SLEEP"; fi
+ done
+ (( started )) || die "could not start $LIVE_CONTAINER after $START_ATTEMPTS attempts"
+ wait_for "$DOWN_TIMEOUT" live_ready || die "live container $LIVE_CONTAINER is not healthy (expected HTTP 200)"
+ echo "worker.sh: live container healthy" >&2
+}
+
+cmd_vram() {
+ local mib
+ mib="$(remote "nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits | head -n 1")"
+ mib="${mib//[[:space:]]/}"
+ is_int "$mib" || die "unexpected nvidia-smi output: $mib"
+ echo "$mib"
+}
+
+cmd_health() {
+ remote "curl -sS --max-time 5 http://127.0.0.1:$(q "$TEST_PORT")/health"
+ echo
+}
+
+main() {
+ [[ $# -eq 1 ]] || usage
+ need WORKER_SSH
+ case "$1" in
+ up) cmd_up ;;
+ down) cmd_down ;;
+ vram) cmd_vram ;;
+ health) cmd_health ;;
+ *) usage ;;
+ esac
+}
+
+main "$@"
diff --git a/tests/fixtures/bench/run-2x24k/concurrency-parallel-1.json b/tests/fixtures/bench/run-2x24k/concurrency-parallel-1.json
new file mode 100644
index 0000000..0534967
--- /dev/null
+++ b/tests/fixtures/bench/run-2x24k/concurrency-parallel-1.json
@@ -0,0 +1,54 @@
+{
+ "results": [
+ {
+ "task": "job_application",
+ "variant": "eligible",
+ "status": "FAIL",
+ "model": "qwen3.5-9b-q4_k_m",
+ "seconds": 34.37005141700001,
+ "peak_vram_mib": null,
+ "turns": [
+ {
+ "number": 1,
+ "llm_seconds": 2.604187374999924,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 61,
+ "prompt_tokens": 2094,
+ "total_tokens": 2155
+ }
+ },
+ {
+ "number": 2,
+ "llm_seconds": 0.6229012079999166,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 15,
+ "prompt_tokens": 2277,
+ "total_tokens": 2292
+ }
+ },
+ {
+ "number": 3,
+ "llm_seconds": 0.8088497500000358,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 27,
+ "prompt_tokens": 2434,
+ "total_tokens": 2461
+ }
+ },
+ {
+ "number": 30,
+ "llm_seconds": 0.6123582500000566,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 30,
+ "prompt_tokens": 6659,
+ "total_tokens": 6689
+ }
+ }
+ ]
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/run-2x24k/concurrency-parallel-2.json b/tests/fixtures/bench/run-2x24k/concurrency-parallel-2.json
new file mode 100644
index 0000000..32b73ae
--- /dev/null
+++ b/tests/fixtures/bench/run-2x24k/concurrency-parallel-2.json
@@ -0,0 +1,54 @@
+{
+ "results": [
+ {
+ "task": "job_application",
+ "variant": "eligible",
+ "status": "PASS",
+ "model": "qwen3.5-9b-q4_k_m",
+ "seconds": 24.596556167000017,
+ "peak_vram_mib": null,
+ "turns": [
+ {
+ "number": 1,
+ "llm_seconds": 2.587898500000051,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 61,
+ "prompt_tokens": 2097,
+ "total_tokens": 2158
+ }
+ },
+ {
+ "number": 2,
+ "llm_seconds": 0.6266710419999981,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 15,
+ "prompt_tokens": 2280,
+ "total_tokens": 2295
+ }
+ },
+ {
+ "number": 3,
+ "llm_seconds": 0.8088767499999676,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 27,
+ "prompt_tokens": 2437,
+ "total_tokens": 2464
+ }
+ },
+ {
+ "number": 18,
+ "llm_seconds": 2.516835915999991,
+ "finish_reason": "stop",
+ "usage": {
+ "completion_tokens": 127,
+ "prompt_tokens": 4886,
+ "total_tokens": 5013
+ }
+ }
+ ]
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/run-2x24k/concurrency-parallel-3.json b/tests/fixtures/bench/run-2x24k/concurrency-parallel-3.json
new file mode 100644
index 0000000..fc9a142
--- /dev/null
+++ b/tests/fixtures/bench/run-2x24k/concurrency-parallel-3.json
@@ -0,0 +1,54 @@
+{
+ "results": [
+ {
+ "task": "job_application",
+ "variant": "eligible",
+ "status": "FAIL",
+ "model": "qwen3.5-9b-q4_k_m",
+ "seconds": 34.41158066599996,
+ "peak_vram_mib": null,
+ "turns": [
+ {
+ "number": 1,
+ "llm_seconds": 1.819642042000055,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 61,
+ "prompt_tokens": 2096,
+ "total_tokens": 2157
+ }
+ },
+ {
+ "number": 2,
+ "llm_seconds": 0.7381971250000561,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 15,
+ "prompt_tokens": 2279,
+ "total_tokens": 2294
+ }
+ },
+ {
+ "number": 3,
+ "llm_seconds": 0.8816241249999166,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 27,
+ "prompt_tokens": 2436,
+ "total_tokens": 2463
+ }
+ },
+ {
+ "number": 30,
+ "llm_seconds": 0.569238749999954,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 30,
+ "prompt_tokens": 6512,
+ "total_tokens": 6542
+ }
+ }
+ ]
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/run-2x24k/concurrency-parallel-4.json b/tests/fixtures/bench/run-2x24k/concurrency-parallel-4.json
new file mode 100644
index 0000000..11bc3c0
--- /dev/null
+++ b/tests/fixtures/bench/run-2x24k/concurrency-parallel-4.json
@@ -0,0 +1,54 @@
+{
+ "results": [
+ {
+ "task": "job_application",
+ "variant": "eligible",
+ "status": "PASS",
+ "model": "qwen3.5-9b-q4_k_m",
+ "seconds": 24.934957333999932,
+ "peak_vram_mib": null,
+ "turns": [
+ {
+ "number": 1,
+ "llm_seconds": 1.8139366660000178,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 61,
+ "prompt_tokens": 2096,
+ "total_tokens": 2157
+ }
+ },
+ {
+ "number": 2,
+ "llm_seconds": 0.7385952920000136,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 15,
+ "prompt_tokens": 2279,
+ "total_tokens": 2294
+ }
+ },
+ {
+ "number": 3,
+ "llm_seconds": 0.8873174159999735,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 27,
+ "prompt_tokens": 2436,
+ "total_tokens": 2463
+ }
+ },
+ {
+ "number": 18,
+ "llm_seconds": 2.5937946250001005,
+ "finish_reason": "stop",
+ "usage": {
+ "completion_tokens": 133,
+ "prompt_tokens": 4930,
+ "total_tokens": 5063
+ }
+ }
+ ]
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/run-2x24k/concurrency-serial-1.json b/tests/fixtures/bench/run-2x24k/concurrency-serial-1.json
new file mode 100644
index 0000000..a1145c3
--- /dev/null
+++ b/tests/fixtures/bench/run-2x24k/concurrency-serial-1.json
@@ -0,0 +1,54 @@
+{
+ "results": [
+ {
+ "task": "job_application",
+ "variant": "eligible",
+ "status": "PASS",
+ "model": "qwen3.5-9b-q4_k_m",
+ "seconds": 17.721931667000717,
+ "peak_vram_mib": null,
+ "turns": [
+ {
+ "number": 1,
+ "llm_seconds": 1.6246261249998497,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 61,
+ "prompt_tokens": 2095,
+ "total_tokens": 2156
+ }
+ },
+ {
+ "number": 2,
+ "llm_seconds": 0.3422860419996141,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 15,
+ "prompt_tokens": 2278,
+ "total_tokens": 2293
+ }
+ },
+ {
+ "number": 3,
+ "llm_seconds": 0.8301474159998179,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 27,
+ "prompt_tokens": 2435,
+ "total_tokens": 2462
+ }
+ },
+ {
+ "number": 18,
+ "llm_seconds": 1.6624806250001711,
+ "finish_reason": "stop",
+ "usage": {
+ "completion_tokens": 128,
+ "prompt_tokens": 4866,
+ "total_tokens": 4994
+ }
+ }
+ ]
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/run-2x24k/concurrency-serial-2.json b/tests/fixtures/bench/run-2x24k/concurrency-serial-2.json
new file mode 100644
index 0000000..cefc09a
--- /dev/null
+++ b/tests/fixtures/bench/run-2x24k/concurrency-serial-2.json
@@ -0,0 +1,54 @@
+{
+ "results": [
+ {
+ "task": "job_application",
+ "variant": "eligible",
+ "status": "PASS",
+ "model": "qwen3.5-9b-q4_k_m",
+ "seconds": 14.885879000000386,
+ "peak_vram_mib": null,
+ "turns": [
+ {
+ "number": 1,
+ "llm_seconds": 1.2157578750002358,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 61,
+ "prompt_tokens": 2096,
+ "total_tokens": 2157
+ }
+ },
+ {
+ "number": 2,
+ "llm_seconds": 0.324584624999261,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 15,
+ "prompt_tokens": 2279,
+ "total_tokens": 2294
+ }
+ },
+ {
+ "number": 3,
+ "llm_seconds": 0.4759779169999092,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 27,
+ "prompt_tokens": 2436,
+ "total_tokens": 2463
+ }
+ },
+ {
+ "number": 18,
+ "llm_seconds": 0.3594307089997528,
+ "finish_reason": "stop",
+ "usage": {
+ "completion_tokens": 15,
+ "prompt_tokens": 4870,
+ "total_tokens": 4885
+ }
+ }
+ ]
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/run-2x24k/concurrency-serial-3.json b/tests/fixtures/bench/run-2x24k/concurrency-serial-3.json
new file mode 100644
index 0000000..e4e102f
--- /dev/null
+++ b/tests/fixtures/bench/run-2x24k/concurrency-serial-3.json
@@ -0,0 +1,54 @@
+{
+ "results": [
+ {
+ "task": "job_application",
+ "variant": "eligible",
+ "status": "PASS",
+ "model": "qwen3.5-9b-q4_k_m",
+ "seconds": 14.964061957999547,
+ "peak_vram_mib": null,
+ "turns": [
+ {
+ "number": 1,
+ "llm_seconds": 1.2127908330003265,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 61,
+ "prompt_tokens": 2095,
+ "total_tokens": 2156
+ }
+ },
+ {
+ "number": 2,
+ "llm_seconds": 0.3264613749997807,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 15,
+ "prompt_tokens": 2278,
+ "total_tokens": 2293
+ }
+ },
+ {
+ "number": 3,
+ "llm_seconds": 0.46990741699937644,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 27,
+ "prompt_tokens": 2435,
+ "total_tokens": 2462
+ }
+ },
+ {
+ "number": 18,
+ "llm_seconds": 0.3557174159996066,
+ "finish_reason": "stop",
+ "usage": {
+ "completion_tokens": 15,
+ "prompt_tokens": 4878,
+ "total_tokens": 4893
+ }
+ }
+ ]
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/run-2x24k/concurrency-serial-4.json b/tests/fixtures/bench/run-2x24k/concurrency-serial-4.json
new file mode 100644
index 0000000..580ef9f
--- /dev/null
+++ b/tests/fixtures/bench/run-2x24k/concurrency-serial-4.json
@@ -0,0 +1,54 @@
+{
+ "results": [
+ {
+ "task": "job_application",
+ "variant": "eligible",
+ "status": "FAIL",
+ "model": "qwen3.5-9b-q4_k_m",
+ "seconds": 25.33343849999983,
+ "peak_vram_mib": null,
+ "turns": [
+ {
+ "number": 1,
+ "llm_seconds": 1.20848979099992,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 61,
+ "prompt_tokens": 2095,
+ "total_tokens": 2156
+ }
+ },
+ {
+ "number": 2,
+ "llm_seconds": 0.3302010830002473,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 15,
+ "prompt_tokens": 2278,
+ "total_tokens": 2293
+ }
+ },
+ {
+ "number": 3,
+ "llm_seconds": 0.4706059580003057,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 27,
+ "prompt_tokens": 2435,
+ "total_tokens": 2462
+ }
+ },
+ {
+ "number": 30,
+ "llm_seconds": 1.3097657500002242,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 91,
+ "prompt_tokens": 6801,
+ "total_tokens": 6892
+ }
+ }
+ ]
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/run-2x24k/meta.json b/tests/fixtures/bench/run-2x24k/meta.json
new file mode 100644
index 0000000..90dd6ca
--- /dev/null
+++ b/tests/fixtures/bench/run-2x24k/meta.json
@@ -0,0 +1,18 @@
+{
+ "model": "qwen3.5-9b-q4_k_m",
+ "ctx": 49152,
+ "parallel": 2,
+ "gpu": "RTX 3070 Ti",
+ "commit": "b2ac901",
+ "run_id": "2026092502",
+ "created_utc": "2026-09-25T21:10:00+00:00",
+ "suites": [
+ "concurrency"
+ ],
+ "suite_settings": {
+ "concurrency": {
+ "ctx": 49152,
+ "parallel": 2
+ }
+ }
+}
diff --git a/tests/fixtures/bench/run-2x24k/vram-concurrency.json b/tests/fixtures/bench/run-2x24k/vram-concurrency.json
new file mode 100644
index 0000000..df98d1c
--- /dev/null
+++ b/tests/fixtures/bench/run-2x24k/vram-concurrency.json
@@ -0,0 +1,11 @@
+{
+ "samples_mib": [
+ 6100,
+ 6300,
+ 6600,
+ 6823,
+ 6700,
+ 6400
+ ],
+ "peak_mib": 6823
+}
diff --git a/tests/fixtures/bench/run-32k/fixture.json b/tests/fixtures/bench/run-32k/fixture.json
new file mode 100644
index 0000000..fc539cd
--- /dev/null
+++ b/tests/fixtures/bench/run-32k/fixture.json
@@ -0,0 +1,54 @@
+{
+ "results": [
+ {
+ "task": "job_application",
+ "variant": "eligible",
+ "status": "PASS",
+ "model": "qwen3.5-9b-q4_k_m",
+ "seconds": 15.729349834000004,
+ "peak_vram_mib": null,
+ "turns": [
+ {
+ "number": 1,
+ "llm_seconds": 1.5937241249999943,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 61,
+ "prompt_tokens": 2094,
+ "total_tokens": 2155
+ }
+ },
+ {
+ "number": 2,
+ "llm_seconds": 0.32503762500004996,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 15,
+ "prompt_tokens": 2277,
+ "total_tokens": 2292
+ }
+ },
+ {
+ "number": 3,
+ "llm_seconds": 0.47630641599994306,
+ "finish_reason": "tool_calls",
+ "usage": {
+ "completion_tokens": 27,
+ "prompt_tokens": 2434,
+ "total_tokens": 2461
+ }
+ },
+ {
+ "number": 18,
+ "llm_seconds": 0.3714084999999159,
+ "finish_reason": "stop",
+ "usage": {
+ "completion_tokens": 15,
+ "prompt_tokens": 4874,
+ "total_tokens": 4889
+ }
+ }
+ ]
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/run-32k/live_web-1.json b/tests/fixtures/bench/run-32k/live_web-1.json
new file mode 100644
index 0000000..62648df
--- /dev/null
+++ b/tests/fixtures/bench/run-32k/live_web-1.json
@@ -0,0 +1,25 @@
+{
+ "created_utc": "2026-09-26T05:43:14.299641+00:00",
+ "model": "qwen3.5-9b-q4_k_m",
+ "task": "live_web image search",
+ "status": "FAIL",
+ "saved": null,
+ "seconds": 55.26,
+ "turns": [
+ {
+ "number": 1,
+ "llm_seconds": 2.0,
+ "prompt_tokens": 4162
+ },
+ {
+ "number": 2,
+ "llm_seconds": 0.42,
+ "prompt_tokens": 4268
+ },
+ {
+ "number": 3,
+ "llm_seconds": 3.56,
+ "prompt_tokens": 11124
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/run-32k/live_web-2.json b/tests/fixtures/bench/run-32k/live_web-2.json
new file mode 100644
index 0000000..8631c81
--- /dev/null
+++ b/tests/fixtures/bench/run-32k/live_web-2.json
@@ -0,0 +1,30 @@
+{
+ "created_utc": "2026-09-26T05:45:27.597694+00:00",
+ "model": "qwen3.5-9b-q4_k_m",
+ "task": "live_web image search",
+ "status": "PASS (image saved; subject needs visual check)",
+ "saved": {
+ "path": "run/live-orange-images/orange-1790401570.jpg",
+ "bytes": 123461,
+ "kind": "jpg",
+ "url": "https://www.verywellhealth.com/thmb/hMLdauaWPybp8_-TTltweU6J8Kw=/1500x0/filters:no_upscale():max_bytes(150000):strip_icc()/VWH-GettyImages-1205638014-6cee1ce220bd45829eaa069b79b7c3b5.jpg"
+ },
+ "seconds": 43.72,
+ "turns": [
+ {
+ "number": 1,
+ "llm_seconds": 1.45,
+ "prompt_tokens": 4162
+ },
+ {
+ "number": 2,
+ "llm_seconds": 0.33,
+ "prompt_tokens": 4268
+ },
+ {
+ "number": 3,
+ "llm_seconds": 3.61,
+ "prompt_tokens": 11124
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/run-32k/meta.json b/tests/fixtures/bench/run-32k/meta.json
new file mode 100644
index 0000000..3121a31
--- /dev/null
+++ b/tests/fixtures/bench/run-32k/meta.json
@@ -0,0 +1,13 @@
+{
+ "model": "qwen3.5-9b-q4_k_m",
+ "ctx": 32768,
+ "parallel": 1,
+ "gpu": "RTX 3070 Ti",
+ "commit": "9f1c2ab",
+ "run_id": "2026092501",
+ "created_utc": "2026-09-25T20:01:34+00:00",
+ "suites": [
+ "fixture",
+ "live_web"
+ ]
+}
diff --git a/tests/fixtures/bench/run-32k/vram-fixture.json b/tests/fixtures/bench/run-32k/vram-fixture.json
new file mode 100644
index 0000000..8dcf9e3
--- /dev/null
+++ b/tests/fixtures/bench/run-32k/vram-fixture.json
@@ -0,0 +1,9 @@
+{
+ "samples_mib": [
+ 5800,
+ 6100,
+ 6269,
+ 6050
+ ],
+ "peak_mib": 6269
+}
diff --git a/tests/fixtures/bench/run-32k/vram-live_web.json b/tests/fixtures/bench/run-32k/vram-live_web.json
new file mode 100644
index 0000000..ad7f3ba
--- /dev/null
+++ b/tests/fixtures/bench/run-32k/vram-live_web.json
@@ -0,0 +1,8 @@
+{
+ "samples_mib": [
+ 6000,
+ 6120,
+ 5990
+ ],
+ "peak_mib": 6120
+}
diff --git a/tests/fixtures/bench/run-64k/live_web-1.json b/tests/fixtures/bench/run-64k/live_web-1.json
new file mode 100644
index 0000000..41aa57d
--- /dev/null
+++ b/tests/fixtures/bench/run-64k/live_web-1.json
@@ -0,0 +1,25 @@
+{
+ "created_utc": "2026-09-26T05:48:02.260270+00:00",
+ "model": "qwen3.5-9b-q4_k_m",
+ "task": "live_web image search",
+ "status": "FAIL",
+ "saved": null,
+ "seconds": 133.6,
+ "turns": [
+ {
+ "number": 1,
+ "llm_seconds": 3.02,
+ "prompt_tokens": 4162
+ },
+ {
+ "number": 2,
+ "llm_seconds": 0.33,
+ "prompt_tokens": 4268
+ },
+ {
+ "number": 3,
+ "llm_seconds": 3.62,
+ "prompt_tokens": 11124
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/run-64k/live_web-2.json b/tests/fixtures/bench/run-64k/live_web-2.json
new file mode 100644
index 0000000..0d13b3d
--- /dev/null
+++ b/tests/fixtures/bench/run-64k/live_web-2.json
@@ -0,0 +1,30 @@
+{
+ "created_utc": "2026-09-26T05:50:24.113256+00:00",
+ "model": "qwen3.5-9b-q4_k_m",
+ "task": "live_web image search",
+ "status": "PASS (image saved; subject needs visual check)",
+ "saved": {
+ "path": "run/live-orange-images/orange-1790401892.jpg",
+ "bytes": 366229,
+ "kind": "jpg",
+ "url": "https://wallpapers.com/images/hd/bunch-of-orange-fruits-emqolx6janwpicfm.jpg"
+ },
+ "seconds": 69.53,
+ "turns": [
+ {
+ "number": 1,
+ "llm_seconds": 1.23,
+ "prompt_tokens": 4162
+ },
+ {
+ "number": 2,
+ "llm_seconds": 0.34,
+ "prompt_tokens": 4268
+ },
+ {
+ "number": 3,
+ "llm_seconds": 3.61,
+ "prompt_tokens": 11124
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/run-64k/live_web-3.json b/tests/fixtures/bench/run-64k/live_web-3.json
new file mode 100644
index 0000000..f9d19ec
--- /dev/null
+++ b/tests/fixtures/bench/run-64k/live_web-3.json
@@ -0,0 +1,30 @@
+{
+ "created_utc": "2026-09-26T05:51:33.794489+00:00",
+ "model": "qwen3.5-9b-q4_k_m",
+ "task": "live_web image search",
+ "status": "PASS (image saved; subject needs visual check)",
+ "saved": {
+ "path": "run/live-orange-images/orange-1790401935.jpg",
+ "bytes": 123461,
+ "kind": "jpg",
+ "url": "https://www.verywellhealth.com/thmb/hMLdauaWPybp8_-TTltweU6J8Kw=/1500x0/filters:no_upscale():max_bytes(150000):strip_icc()/VWH-GettyImages-1205638014-6cee1ce220bd45829eaa069b79b7c3b5.jpg"
+ },
+ "seconds": 42.68,
+ "turns": [
+ {
+ "number": 1,
+ "llm_seconds": 1.22,
+ "prompt_tokens": 4162
+ },
+ {
+ "number": 2,
+ "llm_seconds": 0.33,
+ "prompt_tokens": 4268
+ },
+ {
+ "number": 3,
+ "llm_seconds": 3.58,
+ "prompt_tokens": 11124
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/run-64k/long_context.json b/tests/fixtures/bench/run-64k/long_context.json
new file mode 100644
index 0000000..9a4106c
--- /dev/null
+++ b/tests/fixtures/bench/run-64k/long_context.json
@@ -0,0 +1,8 @@
+{
+ "status": "PASS",
+ "prompt_tokens": 59906,
+ "seconds": 26.1,
+ "target_tokens": 65536,
+ "answer": "Line 29953: The quick benchmark fox jumps over 29953 lazy GPUs.",
+ "reason": null
+}
diff --git a/tests/fixtures/bench/run-64k/meta.json b/tests/fixtures/bench/run-64k/meta.json
new file mode 100644
index 0000000..fa594eb
--- /dev/null
+++ b/tests/fixtures/bench/run-64k/meta.json
@@ -0,0 +1,13 @@
+{
+ "model": "qwen3.5-9b-q4_k_m",
+ "ctx": 65536,
+ "parallel": 1,
+ "gpu": "RTX 3070 Ti",
+ "commit": "c77e410",
+ "run_id": "2026092503",
+ "created_utc": "2026-09-26T05:50:24+00:00",
+ "suites": [
+ "long_context",
+ "live_web"
+ ]
+}
diff --git a/tests/fixtures/bench/run-64k/vram-live_web.json b/tests/fixtures/bench/run-64k/vram-live_web.json
new file mode 100644
index 0000000..088713d
--- /dev/null
+++ b/tests/fixtures/bench/run-64k/vram-live_web.json
@@ -0,0 +1,8 @@
+{
+ "samples_mib": [
+ 6900,
+ 7100,
+ 7000
+ ],
+ "peak_mib": 7100
+}
diff --git a/tests/fixtures/bench/run-64k/vram-long_context.json b/tests/fixtures/bench/run-64k/vram-long_context.json
new file mode 100644
index 0000000..33db778
--- /dev/null
+++ b/tests/fixtures/bench/run-64k/vram-long_context.json
@@ -0,0 +1,9 @@
+{
+ "samples_mib": [
+ 7000,
+ 7200,
+ 7323,
+ 7100
+ ],
+ "peak_mib": 7323
+}
diff --git a/tests/fixtures/bench/run-errors/errors.json b/tests/fixtures/bench/run-errors/errors.json
new file mode 100644
index 0000000..e83d5a6
--- /dev/null
+++ b/tests/fixtures/bench/run-errors/errors.json
@@ -0,0 +1,3 @@
+{
+ "fixture": "llama-server returned 500 on /v1/chat/completions"
+}
diff --git a/tests/fixtures/bench/run-errors/long_context.json b/tests/fixtures/bench/run-errors/long_context.json
new file mode 100644
index 0000000..5d4eaac
--- /dev/null
+++ b/tests/fixtures/bench/run-errors/long_context.json
@@ -0,0 +1,8 @@
+{
+ "status": "PASS",
+ "prompt_tokens": 14700,
+ "seconds": 8.4,
+ "target_tokens": 16384,
+ "answer": "Line 7350: ok",
+ "reason": null
+}
diff --git a/tests/fixtures/bench/run-errors/meta.json b/tests/fixtures/bench/run-errors/meta.json
new file mode 100644
index 0000000..a6e4a74
--- /dev/null
+++ b/tests/fixtures/bench/run-errors/meta.json
@@ -0,0 +1,13 @@
+{
+ "model": "qwen3.5-4b-q4_k_m",
+ "ctx": 16384,
+ "parallel": 1,
+ "gpu": "RTX 3070 Ti",
+ "commit": "aa11bb2",
+ "run_id": "2026092504",
+ "created_utc": "2026-09-24T09:00:00+00:00",
+ "suites": [
+ "fixture",
+ "long_context"
+ ]
+}
diff --git a/tests/fixtures/bench/run-errors/vram-long_context.json b/tests/fixtures/bench/run-errors/vram-long_context.json
new file mode 100644
index 0000000..f6512da
--- /dev/null
+++ b/tests/fixtures/bench/run-errors/vram-long_context.json
@@ -0,0 +1,7 @@
+{
+ "samples_mib": [
+ 4000,
+ 4200
+ ],
+ "peak_mib": 4200
+}
diff --git a/tests/fixtures/bench/run-missing-meta/fixture.json b/tests/fixtures/bench/run-missing-meta/fixture.json
new file mode 100644
index 0000000..1775c84
--- /dev/null
+++ b/tests/fixtures/bench/run-missing-meta/fixture.json
@@ -0,0 +1,21 @@
+{
+ "results": [
+ {
+ "task": "job_application",
+ "variant": "eligible",
+ "status": "PASS",
+ "seconds": 12.3,
+ "peak_vram_mib": null,
+ "turns": [
+ {
+ "number": 1,
+ "usage": {
+ "prompt_tokens": 2000,
+ "completion_tokens": 10,
+ "total_tokens": 2010
+ }
+ }
+ ]
+ }
+ ]
+}
diff --git a/tests/fixtures/bench/runs-index/broken.json b/tests/fixtures/bench/runs-index/broken.json
new file mode 100644
index 0000000..ded346d
--- /dev/null
+++ b/tests/fixtures/bench/runs-index/broken.json
@@ -0,0 +1 @@
+{not valid json,,,
\ No newline at end of file
diff --git a/tests/fixtures/bench/runs-index/run-2026092501.json b/tests/fixtures/bench/runs-index/run-2026092501.json
new file mode 100644
index 0000000..3c82622
--- /dev/null
+++ b/tests/fixtures/bench/runs-index/run-2026092501.json
@@ -0,0 +1,19 @@
+{
+ "schema": 1,
+ "run_id": "2026092501",
+ "created_utc": "2026-09-25T20:01:34+00:00",
+ "commit": "9f1c2ab",
+ "model": "qwen3.5-9b-q4_k_m",
+ "ctx": 32768,
+ "gpu": "RTX 3070 Ti",
+ "suites": {
+ "fixture": {
+ "passed": 1,
+ "total": 1,
+ "pass_rate": 100.0,
+ "median_seconds": 15.7,
+ "peak_vram_mib": 6269,
+ "peak_prompt_tokens": 4870
+ }
+ }
+}
diff --git a/tests/fixtures/bench/runs-index/run-2026092502.json b/tests/fixtures/bench/runs-index/run-2026092502.json
new file mode 100644
index 0000000..d9536eb
--- /dev/null
+++ b/tests/fixtures/bench/runs-index/run-2026092502.json
@@ -0,0 +1,19 @@
+{
+ "schema": 1,
+ "run_id": "2026092502",
+ "created_utc": "2026-09-25T21:10:00+00:00",
+ "commit": "b2ac901",
+ "model": "qwen3.5-9b-q4_k_m",
+ "ctx": 24576,
+ "gpu": "RTX 3070 Ti",
+ "suites": {
+ "concurrency_serial": {
+ "passed": 3,
+ "total": 4,
+ "pass_rate": 75.0,
+ "median_seconds": 16.3,
+ "peak_vram_mib": 6400,
+ "peak_prompt_tokens": 3200
+ }
+ }
+}
diff --git a/tests/fixtures/bench/runs-index/run-2026092503.json b/tests/fixtures/bench/runs-index/run-2026092503.json
new file mode 100644
index 0000000..322aceb
--- /dev/null
+++ b/tests/fixtures/bench/runs-index/run-2026092503.json
@@ -0,0 +1,19 @@
+{
+ "schema": 1,
+ "run_id": "2026092503",
+ "created_utc": "2026-09-26T05:50:24+00:00",
+ "commit": "c77e410",
+ "model": "qwen3.5-9b-q4_k_m",
+ "ctx": 65536,
+ "gpu": "RTX 3070 Ti",
+ "suites": {
+ "long_context": {
+ "passed": 1,
+ "total": 1,
+ "pass_rate": 100.0,
+ "median_seconds": 26.1,
+ "peak_vram_mib": 7323,
+ "peak_prompt_tokens": 59906
+ }
+ }
+}
diff --git a/tests/live_image_agent.py b/tests/live_image_agent.py
new file mode 100755
index 0000000..5219715
--- /dev/null
+++ b/tests/live_image_agent.py
@@ -0,0 +1,248 @@
+#!/usr/bin/env python3
+"""Live-web agent test: DuckDuckGo image search for oranges, save one image.
+
+Run from the repository root: python3 tests/live_image_agent.py ...
+Reuses the benchmark's chat client and MCP launcher settings, but talks to the
+real web. Navigation is limited to duckduckgo.com; the image itself is
+fetched by a harness tool (save_image) that checks it really is an image
+before saving it.
+"""
+import argparse
+import datetime
+import json
+import os
+from pathlib import Path
+import signal
+import socket
+import subprocess
+import sys
+import tempfile
+import time
+import urllib.error
+import re
+import urllib.parse
+import urllib.request
+
+ROOT = Path(__file__).resolve().parents[1]
+sys.path.insert(0, str(ROOT / "tests"))
+import playwright_agent_bench as bench # noqa: E402
+from smoke_playwright import MCP # noqa: E402
+
+START_URL = "https://duckduckgo.com/"
+TASK = ("Use DuckDuckGo to find pictures of oranges (the fruit) and download one. "
+ f"Start at {START_URL}. Search for oranges, switch to the Images results, pick one "
+ "image that clearly shows orange fruit, find the direct URL of the full image file, "
+ "and call save_image with that URL. When the image is saved, reply with one short "
+ "sentence saying what you saved.")
+SYSTEM = (
+ "You are a browser agent using Playwright tools on the real web. "
+ "Operations do not include automatic snapshots: call browser_snapshot or browser_find to "
+ "inspect the page before choosing element refs. Refs are bare IDs: use target \"e6\", "
+ "not \"ref=e6\". Handle tool errors by inspecting state and changing the target; do not "
+ "repeat a failed call with the same arguments. Only navigate to duckduckgo.com pages. "
+ "To download, pass a direct http(s) image URL to save_image."
+)
+SAVE_TOOL = {"type": "function", "function": {
+ "name": "save_image",
+ "description": "Download an image from a direct http(s) URL and save it to disk. "
+ "Fails if the URL does not return an image.",
+ "parameters": {"type": "object", "properties": {
+ "url": {"type": "string", "description": "Direct URL of the image file"}},
+ "required": ["url"], "additionalProperties": False}}}
+TOOLS = bench.ALLOWED_TOOLS - {"browser_file_upload"}
+MAX_TOOL_TEXT = 24000
+LONG_URL = 300
+STALE_LIMIT = 1500
+USER_AGENT = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
+ "(KHTML, like Gecko) Chrome/140.0.0.0 Safari/537.36")
+MAX_IMAGE_BYTES = 15 * 1024 * 1024
+MAGIC = ((b"\xff\xd8\xff", "jpg"), (b"\x89PNG\r\n\x1a\n", "png"), (b"GIF8", "gif"))
+
+
+def image_kind(data):
+ for prefix, kind in MAGIC:
+ if data.startswith(prefix):
+ return kind
+ if data[:4] == b"RIFF" and data[8:12] == b"WEBP":
+ return "webp"
+ return None
+
+
+def save_image(url, out_dir):
+ parsed = urllib.parse.urlparse(url)
+ if parsed.scheme not in ("http", "https") or not parsed.netloc:
+ return None, "Error: url must be an absolute http(s) URL"
+ request = urllib.request.Request(url, headers={
+ "User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 14_0) AppleWebKit/537.36 "
+ "(KHTML, like Gecko) Chrome/140 Safari/537.36",
+ "Accept": "image/avif,image/webp,image/*,*/*;q=0.8"})
+ try:
+ with urllib.request.urlopen(request, timeout=20) as response:
+ data = response.read(MAX_IMAGE_BYTES + 1)
+ ctype = response.headers.get("Content-Type", "")
+ except (urllib.error.URLError, OSError, ValueError) as error:
+ return None, f"Error: download failed: {error}"
+ if len(data) > MAX_IMAGE_BYTES:
+ return None, "Error: file larger than 15 MB"
+ kind = image_kind(data)
+ if not kind:
+ return None, f"Error: URL did not return an image (Content-Type {ctype!r}). Find the direct image file URL."
+ out_dir.mkdir(parents=True, exist_ok=True)
+ path = out_dir / f"orange-{int(time.time())}.{kind}"
+ path.write_bytes(data)
+ return {"path": str(path), "bytes": len(data), "kind": kind, "url": url}, \
+ f"Saved {len(data)} bytes ({kind}) to {path.name}."
+
+
+def shorten_urls(text):
+ return re.sub(r"https?://\S{%d,}" % LONG_URL,
+ lambda m: m.group(0)[:120] + "...[long link shortened]", text)
+
+
+def drop_stale_pages(messages):
+ """Keep only the newest large page view; older ones become a short note."""
+ for message in messages[:-1]:
+ if message.get("role") == "tool" and len(message["content"]) > STALE_LIMIT:
+ message["content"] = message["content"][:400] + "\n[older page view removed to save context]"
+
+
+def allowed_navigation(url):
+ host = urllib.parse.urlparse(url).hostname or ""
+ return host == "duckduckgo.com" or host.endswith(".duckduckgo.com")
+
+
+def run(args):
+ out_dir = (Path(args.image_dir).resolve() if args.image_dir
+ else Path(args.output).resolve().parent / "live-orange-images")
+ with tempfile.TemporaryDirectory(prefix="qcm-live-agent-") as temporary:
+ root = Path(temporary)
+ with socket.socket() as reservation:
+ reservation.bind(("127.0.0.1", 0))
+ port = reservation.getsockname()[1]
+ env = dict(os.environ, PW_MCP_PORT=str(port), PW_MCP_BROWSER="chrome",
+ PW_MCP_OUTPUT_DIR=str(root / "browser"), PW_MCP_SNAPSHOT="none")
+ log = open(root / "mcp.log", "w+")
+ process = subprocess.Popen(["bash", str(ROOT / "start-playwright-mcp.sh"), "--headless",
+ "--port", str(port), "--browser", "chrome", "--isolated", "--snapshot-mode", "none",
+ "--image-responses", "omit", "--output-dir", str(root / "browser"),
+ "--user-agent", USER_AGENT],
+ cwd=root, env=env, stdout=log, stderr=log, start_new_session=True)
+ report = {"created_utc": datetime.datetime.now(datetime.timezone.utc).isoformat(),
+ "model": args.model, "task": TASK, "turns": [], "status": "FAIL",
+ "saved": None, "tool_errors": 0, "blocked_navigations": 0}
+ started = time.monotonic()
+ try:
+ mcp = MCP(port)
+ deadline = time.monotonic() + 60
+ while True:
+ try:
+ mcp.request("initialize", {"protocolVersion": "2025-03-26", "capabilities": {},
+ "clientInfo": {"name": "qcm-live-agent", "version": "1"}})
+ break
+ except urllib.error.URLError:
+ if process.poll() is not None or time.monotonic() > deadline:
+ log.seek(0)
+ raise RuntimeError("Playwright MCP failed to start: " + log.read())
+ time.sleep(.25)
+ mcp.request("notifications/initialized", notification=True)
+ inventory = mcp.request("tools/list")["tools"]
+ declarations = [{"type": "function", "function": {"name": t["name"],
+ "description": t.get("description", ""), "parameters": t["inputSchema"]}}
+ for t in inventory if t["name"] in TOOLS] + [SAVE_TOOL]
+ report["tools"] = sorted(d["function"]["name"] for d in declarations)
+ agent = bench.Agent(args, mcp, declarations)
+ messages = [{"role": "system", "content": SYSTEM}, {"role": "user", "content": TASK}]
+ deadline = started + args.task_timeout
+ for number in range(1, args.max_turns + 1):
+ if time.monotonic() >= deadline:
+ report["reason"] = "task time limit"
+ break
+ message, timing = agent.chat(messages, deadline)
+ calls = message.get("tool_calls", [])
+ turn = {"number": number, "llm_seconds": round(timing["seconds"], 2),
+ "prompt_tokens": timing["usage"].get("prompt_tokens"),
+ "content": message.get("content"), "calls": []}
+ report["turns"].append(turn)
+ messages.append(message)
+ print(f"turn {number}: {timing['usage'].get('prompt_tokens')} prompt tokens, "
+ f"{[c['function']['name'] for c in calls] or 'answer'}", flush=True)
+ if not calls:
+ report["answer"] = message.get("content") or ""
+ break
+ for call in calls:
+ name = call["function"]["name"]
+ kind, arguments = bench.validate_call(name, call["function"].get("arguments"),
+ agent.names, agent.schemas)
+ entry = {"name": name, "arguments": arguments, "kind": kind}
+ if kind != "valid":
+ text = f"Error: {kind} tool call"
+ report["tool_errors"] += 1
+ elif name == "save_image":
+ saved, text = save_image(arguments["url"], out_dir)
+ if saved:
+ report["saved"] = saved
+ else:
+ report["tool_errors"] += 1
+ elif name == "browser_navigate" and not allowed_navigation(arguments.get("url", "")):
+ text = "Error: navigation is limited to duckduckgo.com. Use save_image for the image URL."
+ report["blocked_navigations"] += 1
+ report["tool_errors"] += 1
+ else:
+ raw = mcp.request("tools/call", {"name": name, "arguments": arguments})
+ text = bench.mcp_text(raw)
+ if raw.get("isError"):
+ report["tool_errors"] += 1
+ text = shorten_urls(text)
+ if len(text) > MAX_TOOL_TEXT:
+ entry["truncated_from"] = len(text)
+ text = text[:MAX_TOOL_TEXT] + "\n[output truncated by harness]"
+ entry["result_head"] = text[:600]
+ turn["calls"].append(entry)
+ if len(text) > STALE_LIMIT:
+ drop_stale_pages(messages)
+ messages.append({"role": "tool", "tool_call_id": call["id"], "content": text})
+ else:
+ report["reason"] = "turn cap reached"
+ except (urllib.error.URLError, OSError, ValueError, AssertionError, RuntimeError) as error:
+ report["reason"] = f"{type(error).__name__}: {error}"
+ finally:
+ report["seconds"] = round(time.monotonic() - started, 2)
+ if report["saved"]:
+ report["status"] = "PASS (image saved; subject needs visual check)"
+ Path(args.output).write_text(json.dumps(report, indent=2) + "\n")
+ if process.poll() is None:
+ os.killpg(process.pid, signal.SIGTERM)
+ try:
+ process.wait(timeout=10)
+ except subprocess.TimeoutExpired:
+ os.killpg(process.pid, signal.SIGKILL)
+ log.close()
+ print(json.dumps({k: report.get(k) for k in ("status", "reason", "seconds", "saved",
+ "tool_errors", "blocked_navigations", "answer")},
+ indent=2))
+ return 0 if report["saved"] else 1
+
+
+def main():
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument("--base-url", required=True)
+ parser.add_argument("--model", required=True)
+ parser.add_argument("--output", required=True)
+ parser.add_argument("--key-env", default="BONSAI_API_KEY")
+ parser.add_argument("--image-dir", default=None,
+ help="Directory to save the downloaded image (default: "
+ "live-orange-images next to --output)")
+ parser.add_argument("--task-timeout", type=float, default=300)
+ parser.add_argument("--request-timeout", type=float, default=120)
+ parser.add_argument("--max-turns", type=int, default=40)
+ parser.add_argument("--max-tokens", type=int, default=768)
+ parser.add_argument("--reasoning", choices=("default", "off", "on"), default="off")
+ args = parser.parse_args()
+ args.cache_prompt = True
+ args.cv_path = ""
+ args.output = str(Path(args.output).resolve())
+ raise SystemExit(run(args))
+
+
+if __name__ == "__main__":
+ main()
diff --git a/tests/long_context_probe.py b/tests/long_context_probe.py
new file mode 100755
index 0000000..bd1161e
--- /dev/null
+++ b/tests/long_context_probe.py
@@ -0,0 +1,125 @@
+#!/usr/bin/env python3
+"""Long-context needle probe: fill most of the context, ask for one line back.
+
+Builds a prompt of numbered filler lines sized to ~85% of --target-tokens,
+asks the model to quote one line by number, and checks the exact line comes
+back. Posts a single non-streaming request to /v1/chat/completions with
+temperature 0 and thinking off. A 400 (for example: context too small for the
+prompt) is reported as FAIL with the server's reason, not raised.
+
+Run from the repository root: python3 tests/long_context_probe.py ...
+"""
+import argparse
+import datetime
+import json
+import os
+from pathlib import Path
+import time
+import urllib.error
+import urllib.request
+
+# Measured, not guessed: on the real Qwen3.5-9B server,
+# " ".join(f"Line {i}: the orange crate number {i} holds navel oranges "
+# "from Valencia." for i in range(2700))
+# plus a one-line question came to 59,906 prompt tokens (the model correctly
+# quoted the requested line back). 59906 / 2700 = 22.187... tokens/line.
+# TOKENS_PER_LINE must stay pinned to this measurement (see
+# test_long_context_probe.py's line-count pins for ctx=32768/65536) — do not
+# "round" it to 22 again, that was the bug that made every probe prompt come
+# out 41% over budget and get a 400 back from the server.
+TOKENS_PER_LINE = 22.2
+TARGET_FRACTION = 0.85
+
+
+def make_line(i):
+ """One filler line, ~22.2 measured tokens: unpadded, so its own token
+ count does not drift as the index grows past 6 digits."""
+ return f"Line {i}: the orange crate number {i} holds navel oranges from Valencia."
+
+
+def build_prompt(target_tokens):
+ """Numbered filler lines filling ~85% of target_tokens (space-joined, the
+ same layout as the measurement above), then a question asking to quote
+ the middle line. Returns (prompt, line) where `line` is the exact text
+ the model should quote."""
+ line_count = max(2, int(TARGET_FRACTION * target_tokens / TOKENS_PER_LINE))
+ lines = [make_line(i) for i in range(line_count)]
+ quote_i = line_count // 2
+ line = lines[quote_i]
+ body = " ".join(lines)
+ question = f" Quote line {quote_i} exactly, and only that line, with no extra words."
+ return body + question, line
+
+
+def check_answer(answer, line):
+ """The model's answer must contain the exact target line's text."""
+ return bool(line) and line.strip() in (answer or "")
+
+
+def request_body(model, prompt):
+ return {"model": model, "messages": [{"role": "user", "content": prompt}],
+ "temperature": 0, "max_tokens": 128, "stream": False,
+ "chat_template_kwargs": {"enable_thinking": False}}
+
+
+def call_server(base_url, model, prompt, key_env, timeout):
+ headers = {"Content-Type": "application/json"}
+ key = os.environ.get(key_env, "")
+ if key:
+ headers["Authorization"] = "Bearer " + key
+ request = urllib.request.Request(
+ base_url.rstrip("/").removesuffix("/v1") + "/v1/chat/completions",
+ data=json.dumps(request_body(model, prompt)).encode(), headers=headers)
+ with urllib.request.urlopen(request, timeout=timeout) as response:
+ return json.load(response)
+
+
+def run(args):
+ started = time.monotonic()
+ prompt, line = build_prompt(args.target_tokens)
+ report = {"created_utc": datetime.datetime.now(datetime.timezone.utc).isoformat(),
+ "model": args.model, "target_tokens": args.target_tokens,
+ "status": "FAIL", "prompt_tokens": None, "answer": None, "reason": None}
+ try:
+ payload = call_server(args.base_url, args.model, prompt, args.key_env, args.request_timeout)
+ except urllib.error.HTTPError as error:
+ detail = error.read().decode("utf-8", errors="replace")[:500]
+ error.close()
+ report["reason"] = f"HTTP {error.code}: {detail}"
+ except (urllib.error.URLError, OSError, ValueError) as error:
+ report["reason"] = f"{type(error).__name__}: {error}"
+ else:
+ usage = payload.get("usage", {}) or {}
+ report["prompt_tokens"] = usage.get("prompt_tokens")
+ answer = payload["choices"][0]["message"].get("content") or ""
+ report["answer"] = answer
+ if check_answer(answer, line):
+ report["status"] = "PASS"
+ else:
+ report["reason"] = "the target line was not found in the answer"
+ report["seconds"] = round(time.monotonic() - started, 2)
+ Path(args.output).write_text(json.dumps(report, indent=2) + "\n")
+ print(f"{report['status']} target_tokens={args.target_tokens} "
+ f"prompt_tokens={report['prompt_tokens']} {report['seconds']:.1f}s: {report['reason']}",
+ flush=True)
+ return 0 if report["status"] == "PASS" else 1
+
+
+def main():
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument("--base-url", required=True)
+ parser.add_argument("--model", required=True)
+ parser.add_argument("--target-tokens", type=int, required=True)
+ parser.add_argument("--output", required=True)
+ parser.add_argument("--key-env", default="BONSAI_API_KEY")
+ parser.add_argument("--request-timeout", type=float, default=120)
+ args = parser.parse_args()
+ if args.target_tokens < 2 * TOKENS_PER_LINE:
+ parser.error("target-tokens is too small to build a probe prompt")
+ args.output = str(Path(args.output).resolve())
+ Path(args.output).parent.mkdir(parents=True, exist_ok=True)
+ raise SystemExit(run(args))
+
+
+if __name__ == "__main__":
+ main()
diff --git a/tests/test_bench_collect.py b/tests/test_bench_collect.py
new file mode 100644
index 0000000..3d11fe5
--- /dev/null
+++ b/tests/test_bench_collect.py
@@ -0,0 +1,273 @@
+"""Tests for tests/bench/collect.py — stdlib unittest only, no network.
+
+Covers: summarize() on the real-shaped fixture dirs under tests/fixtures/bench/
+(pass counts, medians, peaks, absent suites, crashed suites) and index()
+(sorting by created_utc, schema, tolerance of a corrupt summary file).
+"""
+import json
+import os
+import subprocess
+import sys
+import tempfile
+import unittest
+
+from tests.bench import collect
+
+ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+FIXTURES = os.path.join(ROOT, "tests", "fixtures", "bench")
+
+
+def fixture_path(*parts):
+ return os.path.join(FIXTURES, *parts)
+
+
+class SummarizeRun32kTest(unittest.TestCase):
+ """32K Qwen3.5 9B: fixture job_application PASS 15.7s; live_web 1/2."""
+
+ def setUp(self):
+ self.summary = collect.summarize(fixture_path("run-32k"))
+
+ def test_top_level_fields_from_meta(self):
+ self.assertEqual(self.summary["schema"], 1)
+ self.assertEqual(self.summary["run_id"], "2026092501")
+ self.assertEqual(self.summary["created_utc"], "2026-09-25T20:01:34+00:00")
+ self.assertEqual(self.summary["commit"], "9f1c2ab")
+ self.assertEqual(self.summary["model"], "qwen3.5-9b-q4_k_m")
+ self.assertEqual(self.summary["ctx"], 32768)
+ self.assertEqual(self.summary["gpu"], "RTX 3070 Ti")
+
+ def test_fixture_suite(self):
+ s = self.summary["suites"]["fixture"]
+ self.assertEqual(s["passed"], 1)
+ self.assertEqual(s["total"], 1)
+ self.assertEqual(s["pass_rate"], 100.0)
+ self.assertEqual(s["median_seconds"], 15.7)
+ self.assertEqual(s["peak_vram_mib"], 6269)
+ self.assertNotIn("error", s)
+
+ def test_live_web_suite_one_of_two_passed(self):
+ s = self.summary["suites"]["live_web"]
+ self.assertEqual(s["passed"], 1)
+ self.assertEqual(s["total"], 2)
+ self.assertEqual(s["pass_rate"], 50.0)
+ self.assertEqual(s["median_seconds"], 49.5)
+ self.assertEqual(s["peak_vram_mib"], 6120)
+
+ def test_suites_without_settings_default_to_the_run_ctx(self):
+ for key in ("fixture", "live_web"):
+ s = self.summary["suites"][key]
+ self.assertEqual((s["ctx"], s["parallel"]), (32768, 1), key)
+
+ def test_long_context_absent_when_not_requested(self):
+ self.assertNotIn("long_context", self.summary["suites"])
+ self.assertNotIn("concurrency_serial", self.summary["suites"])
+ self.assertNotIn("concurrency_parallel", self.summary["suites"])
+
+
+class SummarizeRun2x24kTest(unittest.TestCase):
+ """2x24K: concurrency parallel 2/4, serial 3/4, peak 6,823 MiB."""
+
+ def setUp(self):
+ self.summary = collect.summarize(fixture_path("run-2x24k"))
+
+ def test_concurrency_serial_three_of_four(self):
+ s = self.summary["suites"]["concurrency_serial"]
+ self.assertEqual((s["passed"], s["total"]), (3, 4))
+ self.assertEqual(s["pass_rate"], 75.0)
+ self.assertEqual(s["median_seconds"], 16.3)
+ self.assertEqual(s["peak_vram_mib"], 6823)
+
+ def test_concurrency_parallel_two_of_four(self):
+ s = self.summary["suites"]["concurrency_parallel"]
+ self.assertEqual((s["passed"], s["total"]), (2, 4))
+ self.assertEqual(s["pass_rate"], 50.0)
+ self.assertEqual(s["median_seconds"], 29.7)
+ self.assertEqual(s["peak_vram_mib"], 6823)
+
+ def test_concurrency_phases_share_one_vram_sample(self):
+ """run_suites.py samples VRAM once for the whole concurrency suite
+ (both phases run back to back under one VramSampler) and writes a
+ single vram-concurrency.json — never a per-phase file. Both summary
+ entries must read that same peak, not two independent ones."""
+ serial = self.summary["suites"]["concurrency_serial"]
+ parallel = self.summary["suites"]["concurrency_parallel"]
+ self.assertEqual(serial["peak_vram_mib"], 6823)
+ self.assertEqual(parallel["peak_vram_mib"], 6823)
+ self.assertEqual(serial["peak_vram_mib"], parallel["peak_vram_mib"])
+
+ def test_run_level_peak_is_max_across_suites(self):
+ peaks = [s["peak_vram_mib"] for s in self.summary["suites"].values()]
+ self.assertEqual(max(peaks), 6823)
+
+ def test_concurrency_suites_carry_the_2_slot_server_settings(self):
+ for key in ("concurrency_serial", "concurrency_parallel"):
+ s = self.summary["suites"][key]
+ self.assertEqual((s["ctx"], s["parallel"]), (49152, 2), key)
+
+
+class SummarizeMergedPhasesTest(unittest.TestCase):
+ """A run with both phases: meta.json keeps the 1-slot ctx at the top,
+ and suite_settings says concurrency ran on the 2-slot server. Each
+ suite's ctx must be the one its own server used."""
+
+ def test_each_suite_gets_its_own_server_ctx(self):
+ with tempfile.TemporaryDirectory() as d:
+ meta = {"model": "m", "ctx": 32768, "parallel": 1, "run_id": "1",
+ "created_utc": "2026-09-26T00:00:00+00:00",
+ "suites": ["long_context", "concurrency"],
+ "suite_settings": {"concurrency": {"ctx": 49152, "parallel": 2}}}
+ with open(os.path.join(d, "meta.json"), "w") as f:
+ json.dump(meta, f)
+ summary = collect.summarize(d)
+ self.assertEqual(summary["ctx"], 32768)
+ suites = summary["suites"]
+ self.assertEqual((suites["long_context"]["ctx"], suites["long_context"]["parallel"]), (32768, 1))
+ for key in ("concurrency_serial", "concurrency_parallel"):
+ self.assertEqual((suites[key]["ctx"], suites[key]["parallel"]), (49152, 2), key)
+
+
+class SummarizeRun64kTest(unittest.TestCase):
+ """64K: long_context PASS 59,906 tokens 26.1s, peak 7,323 MiB; live_web 2/3."""
+
+ def setUp(self):
+ self.summary = collect.summarize(fixture_path("run-64k"))
+
+ def test_long_context_suite(self):
+ s = self.summary["suites"]["long_context"]
+ self.assertEqual((s["passed"], s["total"]), (1, 1))
+ self.assertEqual(s["pass_rate"], 100.0)
+ self.assertEqual(s["median_seconds"], 26.1)
+ self.assertEqual(s["peak_prompt_tokens"], 59906)
+ self.assertEqual(s["peak_vram_mib"], 7323)
+
+ def test_live_web_two_of_three(self):
+ s = self.summary["suites"]["live_web"]
+ self.assertEqual((s["passed"], s["total"]), (2, 3))
+ self.assertAlmostEqual(s["pass_rate"], 66.7)
+ self.assertEqual(s["median_seconds"], 69.5)
+
+ def test_run_level_peak_is_long_context(self):
+ peaks = [s["peak_vram_mib"] for s in self.summary["suites"].values()]
+ self.assertEqual(max(peaks), 7323)
+
+
+class SummarizeErrorsTest(unittest.TestCase):
+ """A crashed suite is recorded with 0 passed and its error; other requested
+ suites still summarize normally; suites never requested stay absent."""
+
+ def setUp(self):
+ self.summary = collect.summarize(fixture_path("run-errors"))
+
+ def test_crashed_suite_has_error_and_zero_counts(self):
+ s = self.summary["suites"]["fixture"]
+ self.assertEqual(s["passed"], 0)
+ self.assertEqual(s["total"], 0)
+ self.assertEqual(s["pass_rate"], 0.0)
+ self.assertIsNone(s["median_seconds"])
+ self.assertIsNone(s["peak_vram_mib"])
+ self.assertIn("500", s["error"])
+
+ def test_other_requested_suite_unaffected(self):
+ s = self.summary["suites"]["long_context"]
+ self.assertEqual((s["passed"], s["total"]), (1, 1))
+ self.assertNotIn("error", s)
+
+ def test_never_requested_suite_is_absent(self):
+ self.assertNotIn("live_web", self.summary["suites"])
+ self.assertNotIn("concurrency_serial", self.summary["suites"])
+
+
+class SummarizeMissingFilesTest(unittest.TestCase):
+ """collect.py is tolerant of missing files: no meta.json at all still
+ produces a summary, with null top-level fields and suites inferred from
+ whatever raw report files are actually on disk."""
+
+ def setUp(self):
+ self.summary = collect.summarize(fixture_path("run-missing-meta"))
+
+ def test_missing_meta_gives_null_top_fields(self):
+ self.assertIsNone(self.summary["run_id"])
+ self.assertIsNone(self.summary["created_utc"])
+ self.assertIsNone(self.summary["commit"])
+ self.assertIsNone(self.summary["model"])
+ self.assertIsNone(self.summary["ctx"])
+ self.assertIsNone(self.summary["gpu"])
+ self.assertEqual(self.summary["schema"], 1)
+
+ def test_suite_inferred_from_files_on_disk(self):
+ s = self.summary["suites"]["fixture"]
+ self.assertEqual((s["passed"], s["total"]), (1, 1))
+ self.assertIsNone(s["peak_vram_mib"]) # no vram-fixture.json present
+
+ def test_completely_empty_directory_has_no_suites(self):
+ with tempfile.TemporaryDirectory() as d:
+ summary = collect.summarize(d)
+ self.assertEqual(summary["suites"], {})
+ self.assertEqual(summary["schema"], 1)
+
+
+class SummarizeCliTest(unittest.TestCase):
+ """The `summarize` subcommand writes exactly the JSON collect.summarize()
+ would return, invoked the way the workflow invokes it."""
+
+ def test_summarize_subcommand_writes_out_file(self):
+ with tempfile.TemporaryDirectory() as d:
+ out_path = os.path.join(d, "summary.json")
+ result = subprocess.run(
+ [sys.executable, os.path.join(ROOT, "tests", "bench", "collect.py"),
+ "summarize", fixture_path("run-32k"), "--out", out_path],
+ cwd=ROOT, capture_output=True, text=True,
+ )
+ self.assertEqual(result.returncode, 0, result.stderr)
+ with open(out_path) as f:
+ written = json.load(f)
+ self.assertEqual(written, collect.summarize(fixture_path("run-32k")))
+
+
+class IndexTest(unittest.TestCase):
+ def test_index_sorted_ascending_by_created_utc(self):
+ index = collect.build_index(fixture_path("runs-index"))
+ self.assertEqual(index["schema"], 1)
+ run_ids = [r["run_id"] for r in index["runs"]]
+ self.assertEqual(run_ids, ["2026092501", "2026092502", "2026092503"])
+
+ def test_index_skips_corrupt_summary_file(self):
+ index = collect.build_index(fixture_path("runs-index"))
+ # broken.json must not have produced a crash, and must not appear.
+ self.assertEqual(len(index["runs"]), 3)
+
+ def test_index_of_empty_directory(self):
+ with tempfile.TemporaryDirectory() as d:
+ index = collect.build_index(d)
+ self.assertEqual(index, {"schema": 1, "runs": []})
+
+ def test_index_subcommand_writes_out_file(self):
+ with tempfile.TemporaryDirectory() as d:
+ out_path = os.path.join(d, "index.json")
+ result = subprocess.run(
+ [sys.executable, os.path.join(ROOT, "tests", "bench", "collect.py"),
+ "index", fixture_path("runs-index"), "--out", out_path],
+ cwd=ROOT, capture_output=True, text=True,
+ )
+ self.assertEqual(result.returncode, 0, result.stderr)
+ with open(out_path) as f:
+ written = json.load(f)
+ self.assertEqual(written, collect.build_index(fixture_path("runs-index")))
+
+
+class HelperFunctionTest(unittest.TestCase):
+ def test_median_of_empty_is_none(self):
+ self.assertIsNone(collect.median_or_none([]))
+
+ def test_median_of_values(self):
+ self.assertEqual(collect.median_or_none([1.0, 3.0, 2.0]), 2.0)
+
+ def test_pass_rate_zero_total_is_zero(self):
+ self.assertEqual(collect.pass_rate(0, 0), 0.0)
+
+ def test_pass_rate_rounds_to_one_decimal(self):
+ self.assertEqual(collect.pass_rate(2, 3), 66.7)
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_benchmark_workflow.py b/tests/test_benchmark_workflow.py
new file mode 100644
index 0000000..59a720d
--- /dev/null
+++ b/tests/test_benchmark_workflow.py
@@ -0,0 +1,240 @@
+"""Contract checks on .github/workflows/benchmark.yml.
+
+stdlib unittest, not pytest — see tests/test_cache_viz.py for the convention.
+No PyYAML: the checks read the file line by line, which is enough for the
+few properties that matter here and keeps the suite dependency-free.
+
+Why these properties: the repo is public, and this workflow reaches a real
+machine. A push or pull_request trigger would run fork code on it; an input
+named like a secret would print that secret on the run page; a `down` step
+without always() would leave production stopped after any failure.
+"""
+import os
+import re
+import unittest
+
+ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+WORKFLOW = os.path.join(ROOT, ".github", "workflows", "benchmark.yml")
+
+
+def load():
+ with open(WORKFLOW) as f:
+ return f.read()
+
+
+def top_level_block(text, key):
+ """Lines of a top-level mapping `key:` up to the next top-level key."""
+ lines = text.splitlines()
+ out, inside = [], False
+ for line in lines:
+ if re.match(rf"^{re.escape(key)}:", line):
+ inside = True
+ out.append(line)
+ continue
+ if inside:
+ if line and not line[0].isspace() and not line.startswith("#"):
+ break
+ out.append(line)
+ return out
+
+
+def input_names(text):
+ """Names under on.workflow_dispatch.inputs (six-space indent)."""
+ block = top_level_block(text, "on")
+ names, inside = [], False
+ for line in block:
+ if re.match(r"^ inputs:\s*$", line):
+ inside = True
+ continue
+ if inside:
+ m = re.match(r"^ ([A-Za-z0-9_-]+):\s*$", line)
+ if m:
+ names.append(m.group(1))
+ elif line.strip() and not line.startswith(" "):
+ break
+ return names
+
+
+def steps(text):
+ """Each step as its list of lines (split on ` - `)."""
+ out, cur = [], None
+ for line in text.splitlines():
+ if re.match(r"^ - ", line):
+ if cur:
+ out.append(cur)
+ cur = [line]
+ elif cur is not None:
+ if line and not line.startswith(" ") and line.strip():
+ out.append(cur)
+ cur = None
+ else:
+ cur.append(line)
+ if cur:
+ out.append(cur)
+ return out
+
+
+class BenchmarkWorkflowTest(unittest.TestCase):
+ def setUp(self):
+ self.text = load()
+
+ def test_only_manual_trigger(self):
+ block = top_level_block(self.text, "on")
+ self.assertTrue(block, "no top-level on:")
+ triggers = [m.group(1) for line in block
+ if (m := re.match(r"^ ([A-Za-z_]+):", line))]
+ self.assertEqual(triggers, ["workflow_dispatch"])
+
+ def test_top_level_permissions_read_only(self):
+ block = top_level_block(self.text, "permissions")
+ body = [line.strip() for line in block[1:] if line.strip()]
+ self.assertEqual(body, ["contents: read"])
+
+ def test_one_run_at_a_time(self):
+ block = "\n".join(top_level_block(self.text, "concurrency"))
+ self.assertIn("group: rtx-benchmark", block)
+ self.assertIn("cancel-in-progress: false", block)
+
+ def test_benchmark_job_uses_environment(self):
+ self.assertRegex(self.text, r"\n environment: rtx-benchmark\n")
+
+ def test_inputs_are_the_documented_set(self):
+ self.assertEqual(input_names(self.text), [
+ "worker_host", "ssh_user", "model_preset", "ctx", "suites",
+ "repetitions", "restore_production"])
+
+ def test_no_input_looks_like_a_secret(self):
+ for name in input_names(self.text):
+ for word in ("key", "secret", "token", "password"):
+ self.assertNotIn(word, name.lower(), name)
+
+ def test_down_step_always_runs(self):
+ down = [s for s in steps(self.text)
+ if any("worker.sh down" in line for line in s)]
+ self.assertEqual(len(down), 1, "expected exactly one worker.sh down step")
+ ifs = [line for line in down[0] if line.strip().startswith(("if:", "- if:"))]
+ self.assertEqual(len(ifs), 1)
+ self.assertIn("always()", ifs[0])
+ self.assertIn("inputs.restore_production", ifs[0])
+ # Skip only when settings never resolved (nothing was stopped).
+ self.assertIn("env.WORKER_SSH != ''", ifs[0])
+
+ def test_down_is_last_benchmark_step(self):
+ all_steps = steps(self.text)
+ down = next(i for i, s in enumerate(all_steps)
+ if any("worker.sh down" in line for line in s))
+ publish_start = self.text.index("\n publish:")
+ # Every step before `publish:` must come at or before `down`.
+ offset = 0
+ for i, s in enumerate(all_steps):
+ offset = self.text.index(s[0], offset)
+ if offset < publish_start:
+ self.assertLessEqual(i, down)
+
+ def test_every_action_pinned_to_major(self):
+ uses = re.findall(r"uses:\s*(\S+)", self.text)
+ self.assertTrue(uses)
+ for ref in uses:
+ self.assertRegex(ref, r"^[\w.-]+/[\w.-]+@v\d+$", ref)
+
+ def test_secrets_only_via_secrets_context(self):
+ # Every mention of a secret name is inside ${{ ... secrets.NAME ... }}.
+ for name in ("TS_OAUTH_CLIENT_ID", "TS_OAUTH_SECRET", "TS_AUTHKEY",
+ "WORKER_SSH_KEY", "WORKER_KNOWN_HOSTS"):
+ with self.subTest(secret=name):
+ for m in re.finditer(name, self.text):
+ before = self.text[max(0, m.start() - 8):m.start()]
+ self.assertEqual(before, "secrets.", f"{name} used outside secrets.")
+
+ def test_secrets_never_inlined_into_scripts(self):
+ # ${{ secrets.* }} belongs in with:/env:, never inside run: text,
+ # where it would be pasted into the script body.
+ for s in steps(self.text):
+ in_run = False
+ for line in s:
+ stripped = line.strip()
+ if re.match(r"^(- )?run:", stripped):
+ in_run = True
+ elif re.match(r"^(- )?[a-z-]+:", stripped) and not line.startswith(" "):
+ in_run = False
+ if in_run:
+ self.assertNotIn("secrets.", line)
+
+ def test_inputs_never_inlined_into_scripts(self):
+ for s in steps(self.text):
+ in_run = False
+ for line in s:
+ stripped = line.strip()
+ if re.match(r"^(- )?run:", stripped):
+ in_run = True
+ elif re.match(r"^(- )?[a-z-]+:", stripped) and not line.startswith(" "):
+ in_run = False
+ if in_run:
+ self.assertNotIn("${{", line)
+
+ def test_model_presets(self):
+ self.assertIn("/models/Qwen3.5-9B-GGUF/Qwen3.5-9B-Q4_K_M.gguf", self.text)
+ self.assertIn("qwen3.5-9b-q4_k_m", self.text)
+ self.assertIn("/models/Qwen3.5-4B-GGUF/Qwen3.5-4B-Q4_K_M.gguf", self.text)
+ self.assertIn("qwen3.5-4b-q4_k_m", self.text)
+
+ def test_worker_identity_masked_before_any_step_can_print_it(self):
+ """Public logs: the host and user are masked first thing in Resolve
+ settings, which runs before the tailnet ping, and never echoed."""
+ benchmark = self.text[:self.text.index("\n publish:")]
+ resolve = self.text.index("- name: Resolve settings")
+ mask_host = self.text.index('echo "::add-mask::$IN_WORKER_HOST"')
+ mask_user = self.text.index('echo "::add-mask::$IN_SSH_USER"')
+ first_check = self.text.index('[[ -n "$IN_WORKER_HOST" ]]')
+ self.assertLess(resolve, mask_host)
+ self.assertLess(max(mask_host, mask_user), first_check)
+ self.assertLess(mask_host, self.text.index("ping: ${{ env.WORKER_HOST }}"))
+ # Only Resolve settings holds them; no job-level env entry.
+ self.assertEqual(benchmark.count("IN_WORKER_HOST:"), 1)
+ self.assertEqual(benchmark.count("IN_SSH_USER:"), 1)
+ # No echo prints them, except the mask itself and the lines the
+ # grouped block writes to $GITHUB_ENV (NAME=value, not the log).
+ for m in re.finditer(r'echo "([^"]*)"', benchmark):
+ said = m.group(1)
+ if said.startswith("::add-mask::") or re.match(r"^[A-Z0-9_]+=", said):
+ continue
+ for name in ("$IN_WORKER_HOST", "$IN_SSH_USER", "$WORKER_SSH", "$WORKER_HOST"):
+ self.assertNotIn(name, said)
+
+ def test_suite_steps_have_time_caps(self):
+ for name, cap in (("Run single-slot suites", "fromJSON(env.PHASE1_TIMEOUT_MIN)"),
+ ("Run concurrency suite", "fromJSON(env.PHASE2_TIMEOUT_MIN)")):
+ step = next(s for s in steps(self.text) if name in s[0])
+ self.assertTrue(any("timeout-minutes:" in l and cap in l for l in step), name)
+ for name in ("Start test server (1 slot)", "Restart test server (2 slots)",
+ "Restore production (worker.sh down)"):
+ step = next(s for s in steps(self.text) if name in s[0])
+ self.assertTrue(any("timeout-minutes:" in l for l in step), name)
+
+ def test_phase_2_failure_does_not_fail_the_job(self):
+ for name in ("Restart test server (2 slots)", "Run concurrency suite"):
+ step = next(s for s in steps(self.text) if name in s[0])
+ self.assertTrue(any("continue-on-error: true" in l for l in step), name)
+ merge = next(s for s in steps(self.text) if "Merge phase reports" in s[0])
+ self.assertIn('errors["concurrency"]', "\n".join(merge))
+
+ def test_benchmark_checkout_drops_credentials(self):
+ benchmark = self.text[:self.text.index("\n publish:")]
+ self.assertIn("persist-credentials: false", benchmark)
+
+ def test_images_uploaded_once(self):
+ self.assertNotIn("bench-images-", self.text)
+
+ def test_gpu_name_from_variable(self):
+ self.assertIn("BENCH_GPU_NAME: ${{ vars.BENCH_GPU_NAME }}", self.text)
+
+ def test_publish_job_permissions(self):
+ publish = self.text[self.text.index("\n publish:"):]
+ self.assertIn("needs: benchmark", publish)
+ for perm in ("contents: write", "pages: write", "id-token: write"):
+ self.assertIn(perm, publish)
+ self.assertIn("name: github-pages", publish)
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_live_image_agent.py b/tests/test_live_image_agent.py
new file mode 100644
index 0000000..53e2fd2
--- /dev/null
+++ b/tests/test_live_image_agent.py
@@ -0,0 +1,112 @@
+"""Offline contracts for the live DuckDuckGo image-search agent.
+
+stdlib unittest, not pytest — see tests/test_cache_viz.py for the convention.
+No network: only the pure helpers (image magic-byte sniffing, URL shortening,
+stale-page trimming, navigation allow-list) are exercised here.
+"""
+import importlib.util
+import pathlib
+import unittest
+
+ROOT = pathlib.Path(__file__).resolve().parents[1]
+SPEC = importlib.util.spec_from_file_location(
+ "live_image_agent", ROOT / "tests" / "live_image_agent.py")
+live = importlib.util.module_from_spec(SPEC)
+SPEC.loader.exec_module(live)
+
+
+class ImageKindTest(unittest.TestCase):
+ def test_recognises_jpeg_png_gif_webp_by_magic_bytes(self):
+ self.assertEqual(live.image_kind(b"\xff\xd8\xff\xe0rest-of-jpeg"), "jpg")
+ self.assertEqual(live.image_kind(b"\x89PNG\r\n\x1a\nrest"), "png")
+ self.assertEqual(live.image_kind(b"GIF89a..."), "gif")
+ webp = b"RIFF" + b"\x00\x00\x00\x00" + b"WEBP" + b"rest"
+ self.assertEqual(live.image_kind(webp), "webp")
+
+ def test_rejects_non_image_bytes(self):
+ self.assertIsNone(live.image_kind(b"not an image"))
+ self.assertIsNone(live.image_kind(b""))
+
+ def test_riff_without_webp_fourcc_is_not_an_image(self):
+ avi = b"RIFF" + b"\x00\x00\x00\x00" + b"AVI " + b"rest"
+ self.assertIsNone(live.image_kind(avi))
+
+
+class ShortenUrlsTest(unittest.TestCase):
+ def test_leaves_short_urls_untouched(self):
+ text = "See https://duckduckgo.com/?q=oranges for results."
+ self.assertEqual(live.shorten_urls(text), text)
+
+ def test_shortens_urls_over_300_chars(self):
+ long_url = "https://duckduckgo.com/i.jpg?" + "a" * 300
+ text = f"Image at {long_url} looks good."
+ result = live.shorten_urls(text)
+ self.assertIn("[long link shortened]", result)
+ self.assertLess(len(result), len(text))
+ self.assertTrue(result.startswith("Image at " + long_url[:120]))
+
+ def test_boundary_length_just_under_threshold_is_untouched(self):
+ url = "https://duckduckgo.com/" + "a" * (live.LONG_URL - len("https://duckduckgo.com/") - 1)
+ text = f"See {url} now"
+ self.assertEqual(live.shorten_urls(text), text)
+
+
+class DropStalePagesTest(unittest.TestCase):
+ def test_keeps_only_the_newest_large_tool_result(self):
+ messages = [
+ {"role": "system", "content": "sys"},
+ {"role": "tool", "content": "x" * 2000},
+ {"role": "assistant", "content": "..."},
+ {"role": "tool", "content": "y" * 2000},
+ ]
+ live.drop_stale_pages(messages)
+ self.assertIn("[older page view removed to save context]", messages[1]["content"])
+ self.assertLess(len(messages[1]["content"]), 500)
+ # The newest tool message (last in the list) is never touched by
+ # drop_stale_pages, even though it is also large.
+ self.assertEqual(messages[3]["content"], "y" * 2000)
+
+ def test_short_tool_messages_are_left_alone(self):
+ messages = [{"role": "tool", "content": "short"}, {"role": "tool", "content": "z" * 2000}]
+ live.drop_stale_pages(messages)
+ self.assertEqual(messages[0]["content"], "short")
+
+ def test_non_tool_messages_are_never_truncated(self):
+ messages = [{"role": "assistant", "content": "a" * 2000}, {"role": "tool", "content": "b" * 2000}]
+ live.drop_stale_pages(messages)
+ self.assertEqual(messages[0]["content"], "a" * 2000)
+
+
+class AllowedNavigationTest(unittest.TestCase):
+ def test_allows_duckduckgo_and_its_subdomains(self):
+ self.assertTrue(live.allowed_navigation("https://duckduckgo.com/?q=oranges"))
+ self.assertTrue(live.allowed_navigation("https://duckduckgo.com/"))
+ self.assertTrue(live.allowed_navigation("https://external-content.duckduckgo.com/iu/?u=1"))
+
+ def test_blocks_other_hosts_including_lookalikes(self):
+ self.assertFalse(live.allowed_navigation("https://example.com/"))
+ self.assertFalse(live.allowed_navigation("https://notduckduckgo.com/"))
+ self.assertFalse(live.allowed_navigation("https://duckduckgo.com.evil.example/"))
+ self.assertFalse(live.allowed_navigation(""))
+
+
+class ContractTest(unittest.TestCase):
+ def test_tools_exclude_file_upload_and_include_save_image(self):
+ self.assertNotIn("browser_file_upload", live.TOOLS)
+ self.assertEqual(live.SAVE_TOOL["function"]["name"], "save_image")
+
+ def test_task_and_system_prompt_scope_navigation_to_duckduckgo(self):
+ self.assertIn("duckduckgo.com", live.TASK)
+ self.assertIn("Only navigate to duckduckgo.com pages", live.SYSTEM)
+
+ def test_image_byte_cap_is_15_mib(self):
+ self.assertEqual(live.MAX_IMAGE_BYTES, 15 * 1024 * 1024)
+
+ def test_cli_accepts_image_dir(self):
+ import inspect
+ source = inspect.getsource(live.main)
+ self.assertIn("--image-dir", source)
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_long_context_probe.py b/tests/test_long_context_probe.py
new file mode 100644
index 0000000..6f43b42
--- /dev/null
+++ b/tests/test_long_context_probe.py
@@ -0,0 +1,175 @@
+"""Tests for the long-context needle probe.
+
+stdlib unittest, not pytest — see tests/test_cache_viz.py for the convention.
+No network: HTTP calls are mocked at urllib.request.urlopen.
+"""
+import importlib.util
+import io
+import json
+import pathlib
+import tempfile
+import unittest
+import urllib.error
+from unittest.mock import patch
+
+ROOT = pathlib.Path(__file__).resolve().parents[1]
+SPEC = importlib.util.spec_from_file_location(
+ "long_context_probe", ROOT / "tests" / "long_context_probe.py")
+probe = importlib.util.module_from_spec(SPEC)
+SPEC.loader.exec_module(probe)
+
+
+class BuildPromptTest(unittest.TestCase):
+ def test_line_count_fills_about_85_percent_of_target(self):
+ # int(0.85 * 4400 / 22.2) = 168 lines, indices 0..167.
+ prompt, _ = probe.build_prompt(4400)
+ self.assertIn(probe.make_line(167), prompt)
+ self.assertNotIn(probe.make_line(168), prompt)
+
+ def test_quotes_the_middle_line(self):
+ prompt, line = probe.build_prompt(4400) # 168 lines -> quote index 84
+ self.assertIn("Quote line 84 exactly", prompt)
+ self.assertEqual(line, probe.make_line(84))
+
+ def test_line_text_appears_verbatim_in_the_prompt(self):
+ prompt, line = probe.build_prompt(2200)
+ self.assertIn(line, prompt)
+
+ def test_small_target_still_returns_at_least_two_lines(self):
+ prompt, line = probe.build_prompt(1)
+ self.assertIn(line, prompt)
+ self.assertTrue(line.startswith("Line "))
+
+ def test_larger_target_yields_more_lines_and_a_higher_quote_number(self):
+ _, small_line = probe.build_prompt(2200)
+ _, big_line = probe.build_prompt(22000)
+ self.assertNotEqual(small_line, big_line)
+
+ def test_lines_are_space_joined_like_the_measured_baseline(self):
+ # The 59,906-prompt-token measurement this heuristic is pinned to
+ # used " ".join(...), not newline-joined lines. Tokenization is not
+ # separator-agnostic, so this must not silently change.
+ prompt, _ = probe.build_prompt(2200)
+ self.assertNotIn("\n", prompt)
+
+
+class TokensPerLineConstantTest(unittest.TestCase):
+ """Guards the exact measurement this whole heuristic is pinned to."""
+
+ def test_tokens_per_line_matches_the_recorded_measurement(self):
+ # 59,906 prompt tokens / 2700 lines on the real Qwen3.5-9B server.
+ self.assertAlmostEqual(probe.TOKENS_PER_LINE, 59906 / 2700, places=1)
+
+ def test_target_fraction_is_85_percent(self):
+ self.assertEqual(probe.TARGET_FRACTION, 0.85)
+
+
+class LineCountPinTest(unittest.TestCase):
+ """Pins the line count for common context sizes so the heuristic cannot
+ silently drift back to overshooting the server's context (as
+ TOKENS_PER_LINE = 22 did before this fix)."""
+
+ def test_32768_context_plans_1254_lines_quoting_line_627(self):
+ prompt, line = probe.build_prompt(32768)
+ self.assertIn(probe.make_line(1253), prompt)
+ self.assertNotIn(probe.make_line(1254), prompt)
+ self.assertEqual(line, probe.make_line(627))
+ self.assertIn("Quote line 627 exactly", prompt)
+
+ def test_65536_context_plans_2509_lines_quoting_line_1254(self):
+ prompt, line = probe.build_prompt(65536)
+ self.assertIn(probe.make_line(2508), prompt)
+ self.assertNotIn(probe.make_line(2509), prompt)
+ self.assertEqual(line, probe.make_line(1254))
+ self.assertIn("Quote line 1254 exactly", prompt)
+
+
+class CheckAnswerTest(unittest.TestCase):
+ def test_exact_quote_passes(self):
+ line = probe.make_line(42)
+ self.assertTrue(probe.check_answer(line, line))
+ self.assertTrue(probe.check_answer(f"Sure, here it is: {line}", line))
+
+ def test_paraphrase_or_wrong_line_fails(self):
+ line = probe.make_line(42)
+ other = probe.make_line(43)
+ self.assertFalse(probe.check_answer(other, line))
+ self.assertFalse(probe.check_answer("I don't know.", line))
+ self.assertFalse(probe.check_answer("", line))
+ self.assertFalse(probe.check_answer(None, line))
+
+ def test_empty_expected_line_never_passes(self):
+ self.assertFalse(probe.check_answer("anything", ""))
+
+
+class RequestBodyTest(unittest.TestCase):
+ def test_temperature_zero_and_thinking_off(self):
+ body = probe.request_body("m", "prompt")
+ self.assertEqual(body["temperature"], 0)
+ self.assertEqual(body["chat_template_kwargs"], {"enable_thinking": False})
+ self.assertFalse(body["stream"])
+
+
+class RunTest(unittest.TestCase):
+ def _output_path(self):
+ return pathlib.Path(tempfile.mkdtemp()) / "long_context.json"
+
+ def test_pass_when_server_quotes_the_line(self):
+ prompt, line = probe.build_prompt(4400)
+ payload = {"choices": [{"message": {"content": f"Here: {line}"}}],
+ "usage": {"prompt_tokens": 4123}}
+
+ def fake_urlopen(request, timeout):
+ return io.BytesIO(json.dumps(payload).encode())
+
+ output = self._output_path()
+ args = type("Args", (), {"base_url": "http://127.0.0.1:1", "model": "m",
+ "target_tokens": 4400, "output": str(output),
+ "key_env": "QCM_UNUSED_KEY", "request_timeout": 5})()
+ with patch.object(probe.urllib.request, "urlopen", side_effect=fake_urlopen):
+ code = probe.run(args)
+ self.assertEqual(code, 0)
+ report = json.loads(output.read_text())
+ self.assertEqual(report["status"], "PASS")
+ self.assertEqual(report["prompt_tokens"], 4123)
+ self.assertEqual(report["target_tokens"], 4400)
+ self.assertIsNone(report["reason"])
+
+ def test_fail_when_answer_misses_the_line(self):
+ payload = {"choices": [{"message": {"content": "I have no idea."}}], "usage": {}}
+
+ def fake_urlopen(request, timeout):
+ return io.BytesIO(json.dumps(payload).encode())
+
+ output = self._output_path()
+ args = type("Args", (), {"base_url": "http://127.0.0.1:1", "model": "m",
+ "target_tokens": 4400, "output": str(output),
+ "key_env": "QCM_UNUSED_KEY", "request_timeout": 5})()
+ with patch.object(probe.urllib.request, "urlopen", side_effect=fake_urlopen):
+ code = probe.run(args)
+ self.assertEqual(code, 1)
+ report = json.loads(output.read_text())
+ self.assertEqual(report["status"], "FAIL")
+ self.assertIn("not found", report["reason"])
+
+ def test_http_400_is_reported_as_fail_with_reason_not_raised(self):
+ def fake_urlopen(request, timeout):
+ raise urllib.error.HTTPError(
+ "http://127.0.0.1:1/v1/chat/completions", 400, "Bad Request",
+ {}, io.BytesIO(b'{"error":"context too small"}'))
+
+ output = self._output_path()
+ args = type("Args", (), {"base_url": "http://127.0.0.1:1", "model": "m",
+ "target_tokens": 4400, "output": str(output),
+ "key_env": "QCM_UNUSED_KEY", "request_timeout": 5})()
+ with patch.object(probe.urllib.request, "urlopen", side_effect=fake_urlopen):
+ code = probe.run(args)
+ self.assertEqual(code, 1)
+ report = json.loads(output.read_text())
+ self.assertEqual(report["status"], "FAIL")
+ self.assertIn("HTTP 400", report["reason"])
+ self.assertIn("context too small", report["reason"])
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_run_suites.py b/tests/test_run_suites.py
new file mode 100644
index 0000000..5b80d83
--- /dev/null
+++ b/tests/test_run_suites.py
@@ -0,0 +1,473 @@
+"""Tests for tests/bench/run_suites.py.
+
+stdlib unittest, not pytest — see tests/test_cache_viz.py for the convention.
+No network: only pure planning/parsing functions and small local subprocesses
+(python -c "...", never a real server) are exercised here.
+"""
+import importlib.util
+import json
+import os
+import pathlib
+import subprocess
+import sys
+import tempfile
+import time
+from types import SimpleNamespace
+import unittest
+
+ROOT = pathlib.Path(__file__).resolve().parents[1]
+SPEC = importlib.util.spec_from_file_location(
+ "run_suites", ROOT / "tests" / "bench" / "run_suites.py")
+run_suites = importlib.util.module_from_spec(SPEC)
+SPEC.loader.exec_module(run_suites)
+
+
+def make_args(**overrides):
+ base = dict(base_url="http://127.0.0.1:1", model="m", key_env="QCM_UNUSED_KEY",
+ ctx=32768, parallel=1, suites=["fixture", "long_context", "live_web"],
+ repetitions=2, vram_cmd=None, gpu=None, run_id=None, job_timeout=None)
+ base.update(overrides)
+ return SimpleNamespace(**base)
+
+
+class CommandBuilderTest(unittest.TestCase):
+ def test_fixture_command_carries_context_and_reasoning_off(self):
+ cmd = run_suites.fixture_command(make_args(), pathlib.Path("/tmp/fixture.json"))
+ self.assertEqual(cmd[0], sys.executable)
+ self.assertTrue(cmd[1].endswith("tests/playwright_agent_bench.py"))
+ self.assertIn("--context", cmd)
+ self.assertEqual(cmd[cmd.index("--context") + 1], "32768")
+ self.assertIn("--reasoning", cmd)
+ self.assertEqual(cmd[cmd.index("--reasoning") + 1], "off")
+ self.assertIn("--repetitions", cmd)
+ self.assertEqual(cmd[cmd.index("--repetitions") + 1], "2")
+
+ def test_long_context_command_uses_ctx_as_target_tokens(self):
+ cmd = run_suites.long_context_command(make_args(), pathlib.Path("/tmp/long_context.json"))
+ self.assertTrue(cmd[1].endswith("tests/long_context_probe.py"))
+ self.assertIn("--target-tokens", cmd)
+ self.assertEqual(cmd[cmd.index("--target-tokens") + 1], "32768")
+ self.assertNotIn("--reasoning", cmd) # the probe has no such flag
+
+ def test_live_web_command_passes_image_dir_and_reasoning_off(self):
+ cmd = run_suites.live_web_command(make_args(), pathlib.Path("/tmp/live_web-1.json"),
+ pathlib.Path("/tmp/live-images"))
+ self.assertTrue(cmd[1].endswith("tests/live_image_agent.py"))
+ self.assertIn("--image-dir", cmd)
+ self.assertEqual(cmd[cmd.index("--image-dir") + 1], "/tmp/live-images")
+ self.assertEqual(cmd[cmd.index("--reasoning") + 1], "off")
+
+ def test_concurrency_command_uses_exact_recorded_flags(self):
+ cmd = run_suites.concurrency_command(make_args(), pathlib.Path("/tmp/concurrency-serial-1.json"))
+ self.assertTrue(cmd[1].endswith("tests/playwright_agent_bench.py"))
+ for flag, value in (("--tasks", "job_application"), ("--repetitions", "1"),
+ ("--task-timeout", "120"), ("--max-tokens", "512"),
+ ("--context", "24576"), ("--reasoning", "off")):
+ self.assertEqual(cmd[cmd.index(flag) + 1], value, flag)
+
+
+class BuildPlanTest(unittest.TestCase):
+ def test_plans_one_batch_each_for_fixture_and_long_context(self):
+ out_dir = pathlib.Path("/tmp/out")
+ batches, errors = run_suites.build_plan(
+ make_args(suites=["fixture", "long_context"]), out_dir)
+ self.assertEqual(errors, {})
+ self.assertEqual(len(batches), 2)
+ self.assertEqual([len(b) for b in batches], [1, 1])
+ self.assertEqual(batches[0][0]["suite"], "fixture")
+ self.assertEqual(batches[0][0]["output"], out_dir / "fixture.json")
+ self.assertEqual(batches[1][0]["suite"], "long_context")
+ self.assertEqual(batches[1][0]["output"], out_dir / "long_context.json")
+
+ def test_live_web_makes_one_batch_per_repetition(self):
+ out_dir = pathlib.Path("/tmp/out")
+ batches, errors = run_suites.build_plan(
+ make_args(suites=["live_web"], repetitions=3), out_dir)
+ self.assertEqual(errors, {})
+ self.assertEqual(len(batches), 3)
+ outputs = [b[0]["output"] for b in batches]
+ self.assertEqual(outputs, [out_dir / "live_web-1.json", out_dir / "live_web-2.json",
+ out_dir / "live_web-3.json"])
+ # All repetitions share one image directory.
+ image_dirs = {b[0]["cmd"][b[0]["cmd"].index("--image-dir") + 1] for b in batches}
+ self.assertEqual(image_dirs, {str(out_dir / "live-images")})
+
+ def test_concurrency_needs_parallel_2_else_errors_and_no_batches(self):
+ out_dir = pathlib.Path("/tmp/out")
+ batches, errors = run_suites.build_plan(
+ make_args(suites=["concurrency"], parallel=1), out_dir)
+ self.assertEqual(batches, [])
+ self.assertIn("concurrency", errors)
+ self.assertIn("--parallel 2", errors["concurrency"])
+
+ def test_concurrency_with_parallel_2_makes_4_serial_then_2_pairs(self):
+ out_dir = pathlib.Path("/tmp/out")
+ batches, errors = run_suites.build_plan(
+ make_args(suites=["concurrency"], parallel=2), out_dir)
+ self.assertEqual(errors, {})
+ sizes = [len(b) for b in batches]
+ self.assertEqual(sizes, [1, 1, 1, 1, 2, 2])
+ serial_outputs = [b[0]["output"].name for b in batches[:4]]
+ self.assertEqual(serial_outputs, [f"concurrency-serial-{n}.json" for n in range(1, 5)])
+ parallel_outputs = sorted(job["output"].name for batch in batches[4:] for job in batch)
+ self.assertEqual(parallel_outputs, [f"concurrency-parallel-{n}.json" for n in range(1, 5)])
+ for job in batches[4]:
+ self.assertEqual(job["suite"], "concurrency")
+
+ def test_suites_run_in_canonical_order_regardless_of_request_order(self):
+ out_dir = pathlib.Path("/tmp/out")
+ batches, _ = run_suites.build_plan(
+ make_args(suites=["live_web", "fixture"], repetitions=1), out_dir)
+ self.assertEqual([b[0]["suite"] for b in batches], ["fixture", "live_web"])
+
+
+class VramSampleParsingTest(unittest.TestCase):
+ def test_parses_a_bare_integer(self):
+ self.assertEqual(run_suites.parse_vram_sample("6269\n"), 6269)
+
+ def test_parses_the_last_line_when_there_is_remote_ssh_banner_noise(self):
+ self.assertEqual(run_suites.parse_vram_sample("Warning: banner\n6269"), 6269)
+
+ def test_returns_none_for_empty_or_non_numeric_output(self):
+ self.assertIsNone(run_suites.parse_vram_sample(""))
+ self.assertIsNone(run_suites.parse_vram_sample(None))
+ self.assertIsNone(run_suites.parse_vram_sample("not a number"))
+
+
+class VramPeakTest(unittest.TestCase):
+ def test_peak_of_samples(self):
+ self.assertEqual(run_suites.vram_peak([100, 6269, 3000]), 6269)
+
+ def test_ignores_failed_samples(self):
+ self.assertEqual(run_suites.vram_peak([100, None, 200]), 200)
+
+ def test_empty_or_all_failed_is_none(self):
+ self.assertIsNone(run_suites.vram_peak([]))
+ self.assertIsNone(run_suites.vram_peak([None, None]))
+
+
+class VramSamplerTest(unittest.TestCase):
+ def test_samples_with_an_injected_runner_and_reports_peak(self):
+ values = iter([100, 200, 150])
+ sampler = run_suites.VramSampler("unused", interval=0.01,
+ runner=lambda: next(values, None))
+ with sampler:
+ # Wait for real samples, not a fixed sleep: a loaded CI host may
+ # not schedule the thread within a few milliseconds.
+ deadline = time.monotonic() + 10
+ while len(sampler.samples) < 3 and time.monotonic() < deadline:
+ time.sleep(0.01)
+ report = sampler.report()
+ self.assertGreaterEqual(len(report["samples_mib"]), 1)
+ self.assertEqual(report["peak_mib"], max(report["samples_mib"]))
+
+ def test_no_cmd_means_no_sampling_and_null_peak(self):
+ sampler = run_suites.VramSampler(None, interval=0.01)
+ with sampler:
+ pass
+ self.assertEqual(sampler.report(), {"samples_mib": [], "peak_mib": None})
+
+
+class BuildMetaTest(unittest.TestCase):
+ def test_builds_fresh_meta_when_nothing_exists_yet(self):
+ args = make_args(suites=["fixture", "long_context"], ctx=32768, parallel=1)
+ meta = run_suites.build_meta(None, args, "abc1234", "RTX 3070 Ti", "42", "2026-09-26T00:00:00+00:00")
+ self.assertEqual(meta, {"model": "m", "ctx": 32768, "parallel": 1, "gpu": "RTX 3070 Ti",
+ "commit": "abc1234", "run_id": "42",
+ "created_utc": "2026-09-26T00:00:00+00:00",
+ "suites": ["fixture", "long_context"],
+ "suite_settings": {
+ "fixture": {"ctx": 32768, "parallel": 1},
+ "long_context": {"ctx": 32768, "parallel": 1}}})
+
+ def test_merging_a_second_invocation_unions_suites_and_keeps_first_created_utc(self):
+ existing = {"model": "m", "ctx": 32768, "parallel": 1, "gpu": "RTX 3070 Ti",
+ "commit": "abc1234", "run_id": "42", "created_utc": "2026-09-26T00:00:00+00:00",
+ "suites": ["fixture", "long_context", "live_web"]}
+ args = make_args(suites=["concurrency"], ctx=49152, parallel=2)
+ meta = run_suites.build_meta(existing, args, "abc1234", "RTX 3070 Ti", "42",
+ "2026-09-26T01:00:00+00:00")
+ self.assertEqual(meta["created_utc"], "2026-09-26T00:00:00+00:00")
+ self.assertEqual(meta["suites"], ["fixture", "long_context", "live_web", "concurrency"])
+ self.assertEqual(meta["ctx"], 49152)
+ self.assertEqual(meta["parallel"], 2)
+
+ def test_suite_settings_keep_each_phase_server(self):
+ """The concurrency phase runs on a different server: its ctx must not
+ overwrite the settings the earlier suites ran with."""
+ first = run_suites.build_meta(None, make_args(suites=["fixture"], ctx=32768, parallel=1),
+ "abc1234", None, "42", "2026-09-26T00:00:00+00:00")
+ second = run_suites.build_meta(first, make_args(suites=["concurrency"], ctx=49152, parallel=2),
+ "abc1234", None, "42", "2026-09-26T01:00:00+00:00")
+ self.assertEqual(second["suite_settings"], {
+ "fixture": {"ctx": 32768, "parallel": 1},
+ "concurrency": {"ctx": 49152, "parallel": 2}})
+
+
+class TimeBudgetTest(unittest.TestCase):
+ """The workflow sizes its step caps and refuses too many repetitions
+ from these numbers, so the restore step always fits in the job."""
+
+ ALL_PHASE1 = ["fixture", "long_context", "live_web"]
+
+ def test_batch_count_matches_build_plan(self):
+ args = make_args(suites=["fixture", "long_context", "live_web", "concurrency"],
+ parallel=2, repetitions=3)
+ batches, _ = run_suites.build_plan(args, pathlib.Path("/tmp/out"))
+ for suite in args.suites:
+ planned = sum(1 for b in batches if b[0]["suite"] == suite)
+ self.assertEqual(run_suites.batch_count(suite, 3), planned, suite)
+
+ def test_step_cap_covers_every_job_timeout(self):
+ reps = 2
+ args = make_args(repetitions=reps)
+ floor = sum(run_suites.batch_count(s, reps) * run_suites.job_timeout_for(s, args)
+ for s in self.ALL_PHASE1)
+ self.assertGreater(run_suites.step_minutes(self.ALL_PHASE1, reps) * 60, floor)
+
+ def test_no_suites_means_no_step_time(self):
+ self.assertEqual(run_suites.step_minutes([], 5), 0)
+
+ def test_max_repetitions_fits_and_one_more_does_not(self):
+ for phase1, phase2 in ((self.ALL_PHASE1, ["concurrency"]), (["fixture"], []),
+ (["live_web"], ["concurrency"])):
+ with self.subTest(phase1=phase1, phase2=phase2):
+ most = run_suites.max_repetitions(phase1, phase2)
+ self.assertGreaterEqual(most, 1)
+ self.assertTrue(run_suites.fits_job(phase1, phase2, most))
+ self.assertFalse(run_suites.fits_job(phase1, phase2, most + 1))
+ total = (run_suites.step_minutes(phase1, most) + run_suites.step_minutes(phase2, most)
+ + run_suites.RESERVED_MINUTES)
+ self.assertLessEqual(total, run_suites.JOB_TIMEOUT_MINUTES)
+
+ def test_default_suites_cap_is_two_repetitions(self):
+ # Documented in docs/benchmark-action.md; update both together.
+ self.assertEqual(run_suites.max_repetitions(self.ALL_PHASE1, ["concurrency"]), 2)
+ self.assertEqual(run_suites.max_repetitions(self.ALL_PHASE1, []), 2)
+
+
+class JobTimeoutTest(unittest.TestCase):
+ """Regression tests for the flat 900s default that killed normal fixture
+ runs (10 cases x repetitions, each up to 300s) mid-way through."""
+
+ def test_fixture_scales_with_cases_and_repetitions(self):
+ t1 = run_suites.job_timeout_for("fixture", make_args(repetitions=1))
+ t3 = run_suites.job_timeout_for("fixture", make_args(repetitions=3))
+ self.assertGreater(t3, t1)
+ self.assertEqual(t1, run_suites.FIXTURE_CASES * 1 * run_suites.DEFAULT_TASK_TIMEOUT
+ + run_suites.STARTUP_MARGIN)
+ self.assertEqual(t3, run_suites.FIXTURE_CASES * 3 * run_suites.DEFAULT_TASK_TIMEOUT
+ + run_suites.STARTUP_MARGIN)
+
+ def test_a_flat_900s_budget_would_still_have_killed_a_3_repetition_fixture_run(self):
+ self.assertGreater(run_suites.job_timeout_for("fixture", make_args(repetitions=3)), 900)
+
+ def test_long_context_live_web_and_concurrency_use_their_own_scripts_defaults(self):
+ self.assertEqual(run_suites.job_timeout_for("long_context", make_args()),
+ run_suites.LONG_CONTEXT_REQUEST_TIMEOUT + run_suites.JOB_STARTUP_MARGIN)
+ self.assertEqual(run_suites.job_timeout_for("live_web", make_args()),
+ run_suites.LIVE_WEB_TASK_TIMEOUT + run_suites.JOB_STARTUP_MARGIN)
+ self.assertEqual(run_suites.job_timeout_for("concurrency", make_args()),
+ run_suites.CONCURRENCY_TASK_TIMEOUT + run_suites.JOB_STARTUP_MARGIN)
+
+ def test_unknown_suite_raises(self):
+ with self.assertRaises(ValueError):
+ run_suites.job_timeout_for("nope", make_args())
+
+
+class RunJobTest(unittest.TestCase):
+ def test_success_when_the_subprocess_writes_valid_json(self):
+ with tempfile.TemporaryDirectory() as tmp:
+ output = pathlib.Path(tmp) / "out.json"
+ job = {"suite": "fixture", "output": output,
+ "cmd": [sys.executable, "-c",
+ f"import json,pathlib; pathlib.Path({str(output)!r}).write_text(json.dumps({{'status': 'PASS'}}))"]}
+ self.assertIsNone(run_suites.run_job(job, timeout=10))
+
+ def test_crash_reported_when_no_report_is_written(self):
+ with tempfile.TemporaryDirectory() as tmp:
+ output = pathlib.Path(tmp) / "out.json"
+ job = {"suite": "fixture", "output": output,
+ "cmd": [sys.executable, "-c", "import sys; sys.exit(1)"]}
+ reason = run_suites.run_job(job, timeout=10)
+ self.assertIsNotNone(reason)
+ self.assertIn("without writing a report", reason)
+
+ def test_crash_reported_when_the_command_cannot_even_start(self):
+ job = {"suite": "fixture", "output": pathlib.Path("/nonexistent/out.json"),
+ "cmd": ["/nonexistent/interpreter", "-c", "pass"]}
+ reason = run_suites.run_job(job, timeout=10)
+ self.assertIsNotNone(reason)
+
+ def test_crash_reported_when_the_report_is_not_valid_json(self):
+ with tempfile.TemporaryDirectory() as tmp:
+ output = pathlib.Path(tmp) / "out.json"
+ job = {"suite": "fixture", "output": output,
+ "cmd": [sys.executable, "-c",
+ f"import pathlib; pathlib.Path({str(output)!r}).write_text('not json')"]}
+ reason = run_suites.run_job(job, timeout=10)
+ self.assertIn("invalid report JSON", reason)
+
+ def test_timeout_without_any_output_is_a_plain_timeout_crash(self):
+ with tempfile.TemporaryDirectory() as tmp:
+ output = pathlib.Path(tmp) / "missing.json"
+ job = {"suite": "fixture", "output": output,
+ "cmd": [sys.executable, "-c", "import time; time.sleep(5)"]}
+ reason = run_suites.run_job(job, timeout=0.3)
+ self.assertIn("timed out", reason)
+ self.assertNotIn("kept", reason)
+ self.assertFalse(output.exists())
+
+ def test_timeout_with_a_valid_partial_report_keeps_it_not_discards_it(self):
+ """The exact bug from the review: a valid partial fixture.json must
+ survive a timeout, not be thrown away as a crash."""
+ with tempfile.TemporaryDirectory() as tmp:
+ output = pathlib.Path(tmp) / "out.json"
+ job = {"suite": "fixture", "output": output,
+ "cmd": [sys.executable, "-c",
+ f"import json,pathlib,time; "
+ f"pathlib.Path({str(output)!r}).write_text(json.dumps({{'results': [1]}})); "
+ f"time.sleep(30)"]}
+ # 3 s: the child must start Python and write before the kill,
+ # even on a slow, loaded CI host.
+ reason = run_suites.run_job(job, timeout=3)
+ self.assertIsNotNone(reason)
+ self.assertIn("timed out", reason)
+ self.assertIn("kept the partial report", reason)
+ self.assertEqual(json.loads(output.read_text()), {"results": [1]})
+
+ def test_timeout_kills_the_whole_process_group_not_just_the_direct_child(self):
+ """Regression test for the orphaned-Chrome/Playwright-children bug:
+ subprocess.run(timeout=...) (or Popen.kill()) only reaches the
+ process it spawned directly. A backgrounded grandchild, started in
+ the same session via start_new_session=True, must die too."""
+ with tempfile.TemporaryDirectory() as tmp:
+ pidfile = pathlib.Path(tmp) / "child.pid"
+ output = pathlib.Path(tmp) / "out.json"
+ job = {"suite": "fixture", "output": output,
+ "cmd": ["bash", "-c", f"sleep 30 & echo $! > {pidfile}; wait"]}
+ # 2 s: bash must have started the grandchild before the kill.
+ reason = run_suites.run_job(job, timeout=2)
+ self.assertIn("timed out", reason)
+ deadline = time.monotonic() + 5
+ pid = None
+ while time.monotonic() < deadline:
+ if pidfile.exists() and pidfile.read_text().strip():
+ pid = int(pidfile.read_text().strip())
+ break
+ time.sleep(0.05)
+ self.assertIsNotNone(pid, "the background grandchild never even started")
+ # Give the process-group kill a moment to land, then confirm the
+ # grandchild (not just the bash process we spawned) is gone.
+ gone = False
+ for _ in range(50):
+ try:
+ os.kill(pid, 0)
+ except ProcessLookupError:
+ gone = True
+ break
+ time.sleep(0.1)
+ self.assertTrue(gone, f"grandchild pid {pid} (sleep 30) was not killed with the group")
+
+
+class RunBatchTest(unittest.TestCase):
+ def test_a_pair_runs_concurrently_and_reports_the_first_crash(self):
+ with tempfile.TemporaryDirectory() as tmp:
+ good_output = pathlib.Path(tmp) / "good.json"
+ bad_output = pathlib.Path(tmp) / "bad.json"
+ good = {"suite": "concurrency", "output": good_output,
+ "cmd": [sys.executable, "-c",
+ f"import json,pathlib; pathlib.Path({str(good_output)!r}).write_text(json.dumps({{'status': 'PASS'}}))"]}
+ bad = {"suite": "concurrency", "output": bad_output,
+ "cmd": [sys.executable, "-c", "import sys; sys.exit(1)"]}
+ suite, reason = run_suites.run_batch([good, bad], timeout=10)
+ self.assertEqual(suite, "concurrency")
+ self.assertIn("without writing a report", reason)
+ self.assertTrue(good_output.exists())
+
+ def test_pair_deadline_starts_when_the_pair_starts_not_at_each_wait(self):
+ """The bug: `for job in batch: process.wait(timeout=timeout)` gives
+ EVERY job in the pair a fresh full `timeout`, so a job checked later
+ effectively gets (time already elapsed + a full fresh timeout)
+ instead of sharing one deadline that starts when the pair starts.
+ Both jobs launch concurrently at t=0; `slow` needs 2.6 s (more than
+ the shared 2.0 s deadline, so the fix must kill it) but less than
+ (fast's 1.0 s + a fresh 2.0 s = 3.0 s), which is exactly the window
+ the bug would have given it. The gaps are whole fractions of a
+ second so a loaded CI host (slow Python start-up) cannot blur them."""
+ with tempfile.TemporaryDirectory() as tmp:
+ fast_output = pathlib.Path(tmp) / "fast.json"
+ slow_output = pathlib.Path(tmp) / "slow.json"
+ fast = {"suite": "concurrency", "output": fast_output,
+ "cmd": [sys.executable, "-c",
+ f"import json,time,pathlib; time.sleep(1.0); "
+ f"pathlib.Path({str(fast_output)!r}).write_text(json.dumps({{'status':'PASS'}}))"]}
+ slow = {"suite": "concurrency", "output": slow_output,
+ "cmd": [sys.executable, "-c",
+ f"import json,time,pathlib; time.sleep(2.6); "
+ f"pathlib.Path({str(slow_output)!r}).write_text(json.dumps({{'status':'PASS'}}))"]}
+ started = time.monotonic()
+ suite, reason = run_suites.run_batch([fast, slow], timeout=2.0)
+ elapsed = time.monotonic() - started
+ self.assertIn("timed out", reason)
+ self.assertTrue(fast_output.exists())
+ self.assertFalse(slow_output.exists(),
+ "slow job got a fresh timeout instead of sharing the pair's deadline")
+ self.assertLess(elapsed, 2.5,
+ "ran past the shared deadline towards slow's own natural finish time")
+
+
+class RunIntegrationTest(unittest.TestCase):
+ """run() end-to-end with the command builders swapped for tiny local
+ scripts — no server, no browser, no network."""
+
+ def test_writes_the_full_contract_file_set_and_exits_0(self):
+ originals = (run_suites.fixture_command, run_suites.long_context_command,
+ run_suites.live_web_command, run_suites.concurrency_command)
+
+ def fake_playwright(args, output, *extra, root=None):
+ return [sys.executable, "-c",
+ f"import json,pathlib; pathlib.Path({str(output)!r}).write_text(json.dumps({{'status':'PASS'}}))"]
+
+ run_suites.fixture_command = fake_playwright
+ run_suites.long_context_command = fake_playwright
+ run_suites.live_web_command = fake_playwright
+ run_suites.concurrency_command = fake_playwright
+ try:
+ with tempfile.TemporaryDirectory() as tmp:
+ out_dir = pathlib.Path(tmp) / "out"
+ args = make_args(suites=["fixture", "long_context", "live_web", "concurrency"],
+ parallel=2, repetitions=2, vram_cmd="echo 1234",
+ run_id="99", job_timeout=10, out_dir=str(out_dir))
+ code = run_suites.run(args)
+ self.assertEqual(code, 0)
+ names = {p.name for p in out_dir.iterdir()}
+ expected = {"meta.json", "fixture.json", "long_context.json",
+ "live_web-1.json", "live_web-2.json",
+ "vram-fixture.json", "vram-long_context.json",
+ "vram-live_web.json", "vram-concurrency.json"}
+ expected |= {f"concurrency-serial-{n}.json" for n in range(1, 5)}
+ expected |= {f"concurrency-parallel-{n}.json" for n in range(1, 5)}
+ self.assertEqual(expected, expected & names)
+ self.assertNotIn("errors.json", names)
+ meta = json.loads((out_dir / "meta.json").read_text())
+ self.assertEqual(meta["run_id"], "99")
+ self.assertEqual(meta["suites"], ["fixture", "long_context", "live_web", "concurrency"])
+ finally:
+ (run_suites.fixture_command, run_suites.long_context_command,
+ run_suites.live_web_command, run_suites.concurrency_command) = originals
+
+ def test_exits_1_and_writes_errors_json_when_concurrency_needs_parallel_2(self):
+ with tempfile.TemporaryDirectory() as tmp:
+ out_dir = pathlib.Path(tmp) / "out"
+ args = make_args(suites=["concurrency"], parallel=1, repetitions=1,
+ vram_cmd=None, job_timeout=10, out_dir=str(out_dir))
+ code = run_suites.run(args)
+ self.assertEqual(code, 1)
+ errors = json.loads((out_dir / "errors.json").read_text())
+ self.assertIn("concurrency", errors)
+ self.assertFalse((out_dir / "vram-concurrency.json").exists())
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_script_permissions.py b/tests/test_script_permissions.py
index edcbab5..5778f40 100644
--- a/tests/test_script_permissions.py
+++ b/tests/test_script_permissions.py
@@ -33,6 +33,10 @@
"claude-bonsai.sh",
"update.sh",
"cache-viz.py",
+ "tests/bench/worker.sh",
+ "tests/live_image_agent.py",
+ "tests/long_context_probe.py",
+ "tests/bench/run_suites.py",
]
diff --git a/tests/test_worker_script.py b/tests/test_worker_script.py
new file mode 100644
index 0000000..d9a27cf
--- /dev/null
+++ b/tests/test_worker_script.py
@@ -0,0 +1,236 @@
+"""tests/bench/worker.sh drives the RTX worker over SSH.
+
+stdlib unittest, not pytest — see tests/test_cache_viz.py for the convention.
+
+No test touches a real worker: a fake `ssh` on PATH logs its argv (one JSON
+list per line) and answers the few remote commands the script sends. The
+answers are steered with FAKE_* environment variables.
+"""
+import json
+import os
+import stat
+import subprocess
+import tempfile
+import textwrap
+import unittest
+
+ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+SCRIPT = os.path.join(ROOT, "tests", "bench", "worker.sh")
+
+FAKE_SSH = textwrap.dedent("""\
+ #!/usr/bin/env python3
+ import json, os, sys
+ with open(os.environ["FAKE_SSH_LOG"], "a") as log:
+ log.write(json.dumps(sys.argv[1:]) + "\\n")
+ with open(os.environ["FAKE_SSH_LOG"]) as log:
+ n_calls = sum(1 for _ in log)
+ # FAKE_SSH_FAIL_FIRST=N: the first N connections fail like a dropped link.
+ if n_calls <= int(os.environ.get("FAKE_SSH_FAIL_FIRST", "0")):
+ print("ssh: connect to host rtx-worker port 22: Connection timed out",
+ file=sys.stderr)
+ sys.exit(255)
+ cmd = sys.argv[-1]
+ if "docker inspect" in cmd:
+ rc = int(os.environ.get("FAKE_TEST_RC", "0"))
+ if rc == 0:
+ print('{"status":"ok"}')
+ sys.exit(rc)
+ if "http_code" in cmd:
+ print(os.environ.get("FAKE_LIVE_CODE", "200"), end="")
+ sys.exit(0)
+ if "nvidia-smi" in cmd:
+ print(os.environ.get("FAKE_VRAM", "6269"))
+ sys.exit(0)
+ if "/health" in cmd:
+ print('{"status":"ok"}', end="")
+ sys.exit(0)
+ sys.exit(0)
+ """)
+
+UP_ENV = {
+ "MODELS_DIR": "/srv/models",
+ "TEMPLATE_FILE": "/srv/templates/qwen3.5.jinja",
+ "MODEL_FILE": "/models/Qwen3.5-9B-GGUF/Qwen3.5-9B-Q4_K_M.gguf",
+ "MODEL_ALIAS": "qwen3.5-9b-q4_k_m",
+}
+
+
+class WorkerScriptTest(unittest.TestCase):
+ def setUp(self):
+ self.tmp = tempfile.TemporaryDirectory()
+ self.addCleanup(self.tmp.cleanup)
+ bindir = os.path.join(self.tmp.name, "bin")
+ os.mkdir(bindir)
+ fake = os.path.join(bindir, "ssh")
+ with open(fake, "w") as f:
+ f.write(FAKE_SSH)
+ os.chmod(fake, os.stat(fake).st_mode | stat.S_IXUSR)
+ self.log = os.path.join(self.tmp.name, "ssh.log")
+ self.env = {
+ "PATH": bindir + os.pathsep + os.environ.get("PATH", ""),
+ "HOME": self.tmp.name,
+ "FAKE_SSH_LOG": self.log,
+ "WORKER_SSH": "bench@rtx-worker",
+ "POLL_INTERVAL": "0",
+ "UP_TIMEOUT": "1",
+ "DOWN_TIMEOUT": "1",
+ "RETRY_SLEEP": "0",
+ }
+
+ def run_script(self, *args, **env):
+ full = dict(self.env)
+ full.update(env)
+ return subprocess.run(["bash", SCRIPT, *args], capture_output=True,
+ text=True, env=full, timeout=60)
+
+ def calls(self):
+ if not os.path.exists(self.log):
+ return []
+ with open(self.log) as f:
+ return [json.loads(line) for line in f]
+
+ def commands(self):
+ return [argv[-1] for argv in self.calls()]
+
+ def index_of(self, needle):
+ for i, cmd in enumerate(self.commands()):
+ if needle in cmd:
+ return i
+ self.fail(f"no remote command contains {needle!r}: {self.commands()}")
+
+ def test_missing_worker_ssh_fails_without_calling_ssh(self):
+ for sub in ("up", "down", "vram", "health"):
+ with self.subTest(sub=sub):
+ out = self.run_script(sub, WORKER_SSH="", **UP_ENV)
+ self.assertNotEqual(out.returncode, 0)
+ self.assertIn("WORKER_SSH", out.stderr)
+ self.assertEqual(self.calls(), [])
+
+ def test_unknown_subcommand_is_rejected(self):
+ out = self.run_script("reboot")
+ self.assertNotEqual(out.returncode, 0)
+ self.assertEqual(self.calls(), [])
+
+ def test_up_stops_live_before_running_test_container(self):
+ out = self.run_script("up", **UP_ENV)
+ self.assertEqual(out.returncode, 0, out.stderr)
+ stop_live = self.index_of("docker stop bonsai-llama-1")
+ run_test = self.index_of("docker run")
+ self.assertLess(stop_live, run_test)
+ # It waits for health after starting, not before.
+ self.assertGreater(self.index_of("docker inspect"), run_test)
+
+ def test_up_reproduces_the_recorded_server_command(self):
+ out = self.run_script("up", CTX="49152", PARALLEL="2", **UP_ENV)
+ self.assertEqual(out.returncode, 0, out.stderr)
+ cmd = self.commands()[self.index_of("docker run")]
+ for part in (
+ "docker run -d --rm --gpus all --name qcm-bench",
+ "-p 127.0.0.1:18080:8080",
+ "-v /srv/models:/models:ro",
+ "-v /srv/templates/qwen3.5.jinja:/template.jinja:ro",
+ "--entrypoint /app/src/llama.cpp-prism/build/bin/llama-server",
+ "qcm-rtx-validation:922be44",
+ "--host 0.0.0.0 --port 8080",
+ "--model /models/Qwen3.5-9B-GGUF/Qwen3.5-9B-Q4_K_M.gguf",
+ "--alias qwen3.5-9b-q4_k_m",
+ "--chat-template-file /template.jinja",
+ "--ctx-size 49152 --n-gpu-layers 99 --flash-attn auto",
+ "--cache-type-k f16 --cache-type-v f16 --reasoning off",
+ "--parallel 2",
+ ):
+ self.assertIn(part, cmd)
+ self.assertNotIn("--jinja", cmd.replace("/template.jinja", ""))
+
+ def test_up_uses_strict_host_key_checking(self):
+ self.run_script("up", **UP_ENV)
+ for argv in self.calls():
+ self.assertIn("StrictHostKeyChecking=yes", argv)
+ self.assertIn("BatchMode=yes", argv)
+ self.assertIn("bench@rtx-worker", argv)
+
+ def test_up_requires_model_settings(self):
+ env = dict(UP_ENV)
+ del env["MODEL_FILE"]
+ out = self.run_script("up", **env)
+ self.assertNotEqual(out.returncode, 0)
+ self.assertIn("MODEL_FILE", out.stderr)
+ self.assertEqual(self.calls(), [])
+
+ def test_up_rejects_non_numeric_ctx(self):
+ out = self.run_script("up", CTX="32768; rm -rf /", **UP_ENV)
+ self.assertNotEqual(out.returncode, 0)
+ self.assertEqual(self.calls(), [])
+
+ def test_up_fails_fast_when_test_container_dies(self):
+ out = self.run_script("up", FAKE_TEST_RC="3", UP_TIMEOUT="30", **UP_ENV)
+ self.assertNotEqual(out.returncode, 0)
+ self.assertIn("exited", out.stderr)
+
+ def test_up_fails_when_test_server_never_healthy(self):
+ out = self.run_script("up", FAKE_TEST_RC="22", **UP_ENV)
+ self.assertNotEqual(out.returncode, 0)
+
+ def test_down_stops_test_then_starts_live(self):
+ out = self.run_script("down")
+ self.assertEqual(out.returncode, 0, out.stderr)
+ stop_test = self.index_of("docker stop qcm-bench")
+ start_live = self.index_of("docker start bonsai-llama-1")
+ self.assertLess(stop_test, start_live)
+ self.assertGreater(self.index_of("http_code"), start_live)
+
+ def test_down_survives_ssh_failures_and_still_starts_live(self):
+ # Calls 1-3 fail: stop test (1), start live (2), stop test (3).
+ # Attempt 2's live start (call 4) succeeds, then health passes.
+ out = self.run_script("down", FAKE_SSH_FAIL_FIRST="3")
+ self.assertEqual(out.returncode, 0, out.stderr)
+ cmds = self.commands()
+ starts = [i for i, c in enumerate(cmds) if "docker start bonsai-llama-1" in c]
+ self.assertEqual(starts, [1, 3])
+ self.assertIn("http_code", cmds[4])
+ self.assertIn("attempt 2/5", out.stderr)
+
+ def test_down_gives_up_after_all_start_attempts(self):
+ out = self.run_script("down", FAKE_SSH_FAIL_FIRST="100", START_ATTEMPTS="3")
+ self.assertEqual(out.returncode, 1)
+ starts = [c for c in self.commands() if "docker start bonsai-llama-1" in c]
+ self.assertEqual(len(starts), 3)
+ self.assertIn("after 3 attempts", out.stderr)
+
+ def test_down_fails_when_live_is_unhealthy(self):
+ out = self.run_script("down", FAKE_LIVE_CODE="401")
+ self.assertEqual(out.returncode, 1)
+ self.assertIn("not healthy", out.stderr)
+
+ def test_down_reads_key_on_worker_and_never_sends_it(self):
+ out = self.run_script("down", LIVE_ENV_FILE="/srv/bonsai/.env",
+ LIVE_HEALTH_URL="http://100.64.0.5:8080/health",
+ BONSAI_API_KEY="local-secret-must-not-leak")
+ self.assertEqual(out.returncode, 0, out.stderr)
+ cmd = self.commands()[self.index_of("http_code")]
+ self.assertIn("/srv/bonsai/.env", cmd)
+ self.assertIn("BONSAI_API_KEY=", cmd)
+ self.assertIn("-H @-", cmd)
+ self.assertIn("http://100.64.0.5:8080/health", cmd)
+ for argv in self.calls():
+ self.assertNotIn("local-secret-must-not-leak", " ".join(argv))
+ self.assertNotIn("local-secret-must-not-leak", out.stdout + out.stderr)
+
+ def test_vram_prints_integer(self):
+ out = self.run_script("vram", FAKE_VRAM=" 6269 ")
+ self.assertEqual(out.returncode, 0, out.stderr)
+ self.assertEqual(out.stdout.strip(), "6269")
+
+ def test_vram_rejects_garbage(self):
+ out = self.run_script("vram", FAKE_VRAM="N/A")
+ self.assertNotEqual(out.returncode, 0)
+
+ def test_health_prints_server_json(self):
+ out = self.run_script("health")
+ self.assertEqual(out.returncode, 0, out.stderr)
+ self.assertEqual(json.loads(out.stdout), {"status": "ok"})
+ self.assertIn("127.0.0.1:18080/health", self.commands()[0])
+
+
+if __name__ == "__main__":
+ unittest.main()