diff --git a/.env.example b/.env.example
new file mode 100644
index 0000000..b53b61c
--- /dev/null
+++ b/.env.example
@@ -0,0 +1,51 @@
+# llm-bench configuration.
+# Copy to .env and edit. Env vars already in your shell take precedence
+# over values here (so you can override per-run without editing the file).
+
+# The OpenAI-compatible chat-completions URL of the model you're testing.
+# Examples:
+# LM Studio (local): http://127.0.0.1:1234/v1/chat/completions
+# Ollama (local): http://127.0.0.1:11434/v1/chat/completions
+# vLLM / llama-server: http://127.0.0.1:8000/v1/chat/completions
+# Ollama Cloud: https://api.ollama-cloud.com/v1/chat/completions
+# OpenAI: https://api.openai.com/v1/chat/completions
+ENDPOINT=http://127.0.0.1:1234/v1/chat/completions
+
+# The model id the endpoint expects in the "model" field of requests.
+# For LM Studio this is the load identifier; for Ollama it's the tag;
+# for vLLM/llama-server it's the served model name; for cloud it's the
+# API model string (e.g. "gpt-4o", "claude-sonnet-5" via a compat layer).
+MODEL=local
+
+# Bearer token, sent as "Authorization: Bearer <key>" if non-empty.
+# Leave blank for local endpoints (LM Studio, local Ollama, local vLLM).
+# Set for cloud endpoints or any endpoint that requires auth.
+API_KEY=
+
+# Per-turn generation cap (max_tokens in the request body).
+# Coding suite default is 8000; agentic benches default to 16000.
+MAX_TOKENS=8000
+
+# Sampling temperature. 0 = greedy (deterministic, recommended for
+# coding/tool-use benchmarks). Higher = more creative/noise.
+TEMPERATURE=0
+
+# Per-request wall timeout in seconds. Bump for slow local models.
+HTTP_TIMEOUT=900
+
+# Agentic loop cap — max turns the agent can take before giving up.
+# Only used by agentic-fix, agentic-build, agentic-multitask.
+MAX_TURNS=40
+
+# Suppress reasoning/thinking via an empty <think></think> assistant
+# prefill. Works on Qwen3-family and other models that use the think-tag
+# convention. Set to "1" to enable. Some backends also honor a
+# chat_template_kwargs.enable_thinking field (the no-think instrument
+# tests both paths).
+NOTHINK=
+
+# --- build/test timeouts (agentic benches) ---
+BUILD_TIMEOUT=180
+TEST_TIMEOUT=180
+SHELL_TIMEOUT=60
+SERVER_TIMEOUT=20
\ No newline at end of file
diff --git a/.gitignore b/.gitignore
new file mode 100644
index 0000000..b5945a1
--- /dev/null
+++ b/.gitignore
@@ -0,0 +1,20 @@
+# Python
+__pycache__/
+*.pyc
+*.pyo
+*.egg-info/
+.venv/
+venv/
+
+# Generated by the benches (work dirs + results)
+work/
+bench/work/
+results/
+bench/results/
+
+# Local config — don't commit your endpoint/key
+.env
+
+# OS
+.DS_Store
+Thumbs.db
\ No newline at end of file
diff --git a/README.md b/README.md
new file mode 100644
index 0000000..bd7ae63
--- /dev/null
+++ b/README.md
@@ -0,0 +1,144 @@
+# llm-bench
+
+A local LLM coding-model benchmark suite. Point it at any OpenAI-compatible
+endpoint, pick an instrument, and run. Pure stdlib Python — no packages needed.
+
+## Quick start
+
+```bash
+cp .env.example .env # set ENDPOINT + MODEL for your model
+python bench.py coding-suite validate # prove tasks are well-formed (no model needed)
+python bench.py coding-suite run # run all 24 tasks against the model
+```
+
+Results land in `results/` as JSONL + JSON. Work dirs in `work/` (both gitignored).
+
+## Instruments
+
+```
+python bench.py <instrument> <command> [args...]
+```
+
+| Instrument | Command(s) | What it measures |
+|---|---|---|
+| `coding-suite` | `validate` / `run` / `report` | 24 single-turn bug-fix + tool-calling tasks, real toolchain (Go/Rust/Swift/TS) |
+| `agentic-fix` | `validate` / `run` | Multi-turn tool loop — find a cross-file bug in a small Go codebase |
+| `agentic-build` | `validate` / `run` | Build a Go+SQLite web app end-to-end (4 product gates: build/test/serve/persist) |
+| `agentic-multitask` | `validate <task>` / `run <task>` / `cohort <t1> <t2> ...` | Multi-task gauntlet: Go+HTMX, Rust CLI, Next.js+Prisma |
+| `router` | `validate` / `run` / `report` | Dispatch/router eval (26 ROUTE + 8 ORCH, deterministic scoring, privacy axis) |
+| `needle` | `run [--depths 1000,16000 --needles 3 --repeats 2]` | Needle-in-a-haystack retrieval + speed per depth |
+| `loop-battery` | `run` | CoT runaway probes (5 prompts × N runs, repetition scoring) |
+| `no-think` | `run` | Latency win from suppressing reasoning (think vs no-think A/B) |
+
+### Examples
+
+```bash
+# Coding suite (single-turn, ~5-15 min)
+python bench.py coding-suite run
+
+# Agentic build (multi-turn, ~10-40 min)
+python bench.py agentic-build run
+
+# Full gauntlet — all three real-app tasks, sequential
+python bench.py agentic-multitask cohort go_htmx rust_cli ts_next
+
+# Router/dispatch eval
+python bench.py router run
+
+# Needle retrieval at specific depths
+python bench.py needle run --depths 1000,16000,64000
+
+# Override config per-run without editing .env
+ENDPOINT=http://10.0.0.40:8000/v1/chat/completions \
+MODEL=qwen3.8-27b \
+MAX_TOKENS=16000 \
+python bench.py agentic-build run
+```
+
+## Configuration
+
+Copy `.env.example` to `.env` and edit. Env vars win over `.env`, so you
+can override per-run (see the last example above).
+
+| Var | Default | What |
+|---|---|---|
+| `ENDPOINT` | `http://127.0.0.1:1234/v1/chat/completions` | OpenAI-compat chat URL |
+| `MODEL` | `local` | model id |
+| `API_KEY` | (blank) | bearer token (if endpoint needs auth) |
+| `MAX_TOKENS` | `8000` | per-turn generation cap |
+| `TEMPERATURE` | `0` | sampling temp (0 = greedy, recommended) |
+| `MAX_TURNS` | `40` | agentic loop cap |
+| `HTTP_TIMEOUT` | `900` | per-request wall timeout (s) |
+| `NOTHINK` | (blank) | `1` = suppress reasoning via empty think prefill |
+
+### Compatible endpoints
+
+Anything that speaks OpenAI `/v1/chat/completions`:
+
+- **LM Studio** — `http://127.0.0.1:1234/v1/chat/completions`
+- **Ollama** (local) — `http://127.0.0.1:11434/v1/chat/completions`
+- **vLLM** / **llama-server** — `http://127.0.0.1:8000/v1/chat/completions`
+- **Ollama Cloud** — `https://api.ollama-cloud.com/v1/chat/completions`
+- **OpenAI** — `https://api.openai.com/v1/chat/completions`
+
+Tool-calling tasks use the OpenAI `tools` param. The XML/JSON tool-calling
+tasks use freeform text, so they work even on backends without native
+function-calling.
+
+## Requirements
+
+- Python 3.10+
+- Language toolchains for the tasks you want to run:
+
+| Toolchain | Needed for |
+|---|---|
+| Go 1.22+ | agentic-fix, agentic-build, go_htmx, go bug-fixes |
+| Rust 1.70+ | rust_cli, rust bug-fixes |
+| bun 1.0+ | ts bug-fixes |
+| Node.js 18+ + npm | ts_next |
+| Swift / Xcode | swift bug-fixes (macOS only; skipped gracefully elsewhere) |
+
+No Python packages needed — everything uses the stdlib.
+
+## Adding your own tasks
+
+**Coding suite** — add a dict to `BUGFIX_T1` / `BUGFIX_T2` / `TOOLCALL` in
+`bench/tasks/coding_suite.py`. Bug-fix tasks need: `id`, `lang`, `symptom`,
+`buggy`, `fix`, `test`. `validate` proves buggy fails + fix passes.
+
+**Multitask gauntlet** — drop a `.py` file in `bench/tasks/multitask/`
+exposing `TASK_NAME`, `PLAN`, `STARTER`, `REFERENCE`, `validate(work_dir)`,
+`gates(work_dir)`. Auto-discovered. See `go_htmx.py` for the shape.
+
+**New instrument** — add a module under `bench/tasks/` with a `main()`
+function, register it in `INSTRUMENTS` in `bench.py`. Import from
+`bench.core` (`client`, `runners`, `tools`, `reporting`).
+
+## Project layout
+
+```
+llm-bench/
+├── bench.py # CLI entry point
+├── .env.example # config template
+├── bench/
+│ ├── core/ # client, runners, tools, reporting
+│ └── tasks/ # instruments (one per file)
+│ ├── coding_suite.py
+│ ├── agentic_fix.py
+│ ├── agentic_build.py
+│ ├── agentic_multitask.py
+│ ├── router.py
+│ ├── needle.py
+│ ├── loop_battery.py
+│ ├── no_think.py
+│ └── multitask/ # gauntlet task modules (auto-discovered)
+│ ├── go_htmx.py
+│ ├── rust_cli.py
+│ └── ts_next.py
+├── results/ # generated (gitignored)
+└── work/ # generated (gitignored)
+```
+
+## License
+
+MIT. See LICENSE.
\ No newline at end of file
diff --git a/bench.py b/bench.py
new file mode 100755
index 0000000..1f34ebd
--- /dev/null
+++ b/bench.py
@@ -0,0 +1,115 @@
+#!/usr/bin/env python3
+"""
+llm-bench — a local LLM coding-model benchmark suite.
+
+Unified CLI entry point. Point it at any OpenAI-compatible endpoint via .env
+(or env vars), pick an instrument, and run. Instruments live under
+bench/tasks/ as importable modules.
+
+Usage:
+ python bench.py <instrument> <command> [args...]
+
+ python bench.py coding-suite validate
+ python bench.py coding-suite run
+ python bench.py agentic-fix validate
+ python bench.py agentic-fix run
+ python bench.py agentic-build validate
+ python bench.py agentic-build run
+ python bench.py agentic-multitask validate go_htmx
+ python bench.py agentic-multitask run go_htmx
+ python bench.py agentic-multitask cohort go_htmx rust_cli ts_next
+ python bench.py router validate
+ python bench.py router run
+ python bench.py needle run --depths 1000,16000
+ python bench.py loop-battery run
+ python bench.py no-think run
+
+Config (via .env in the repo root, or env vars):
+ ENDPOINT OpenAI chat-completions URL
+ MODEL model id
+ API_KEY bearer token (if your endpoint needs auth)
+ MAX_TOKENS per-turn generation cap
+ TEMPERATURE sampling temperature
+ MAX_TURNS agentic loop cap (for multi-turn instruments)
+ HTTP_TIMEOUT per-request wall timeout
+ NOTHINK "1" = suppress reasoning via empty think prefill
+
+See .env.example for all options.
+"""
+import os
+import sys
+
+
+def _load_dotenv():
+ """Load .env from the script's directory (or cwd) into os.environ.
+
+ Only sets vars that aren't already in the environment (env vars win
+ over .env, so you can override per-run without editing the file).
+ """
+ here = os.path.dirname(os.path.abspath(__file__))
+ for candidate in (os.path.join(here, ".env"), os.path.join(os.getcwd(), ".env")):
+ if not os.path.isfile(candidate):
+ continue
+ with open(candidate) as f:
+ for line in f:
+ line = line.strip()
+ if not line or line.startswith("#"):
+ continue
+ if "=" not in line:
+ continue
+ key, _, val = line.partition("=")
+ key = key.strip()
+ val = val.strip().strip('"').strip("'")
+ if key and key not in os.environ:
+ os.environ[key] = val
+ break
+
+
+# Instrument name -> module path (under bench/tasks/)
+INSTRUMENTS = {
+ "coding-suite": "bench.tasks.coding_suite",
+ "agentic-fix": "bench.tasks.agentic_fix",
+ "agentic-build": "bench.tasks.agentic_build",
+ "agentic-multitask": "bench.tasks.agentic_multitask",
+ "router": "bench.tasks.router",
+ "needle": "bench.tasks.needle",
+ "loop-battery": "bench.tasks.loop_battery",
+ "no-think": "bench.tasks.no_think",
+}
+
+
+def main():
+ _load_dotenv()
+ if len(sys.argv) < 2:
+ print(__doc__)
+ print("\nAvailable instruments:")
+ for name in sorted(INSTRUMENTS):
+ print(f" {name}")
+ sys.exit(1)
+
+ instrument = sys.argv[1]
+ if instrument not in INSTRUMENTS:
+ print(f"ERROR: unknown instrument {instrument!r}")
+ print(f"Available: {', '.join(sorted(INSTRUMENTS))}")
+ sys.exit(1)
+
+ # Import the instrument module and pass the remaining args to it.
+ # The script's directory must be on sys.path so `bench` package resolves.
+ here = os.path.dirname(os.path.abspath(__file__))
+ if here not in sys.path:
+ sys.path.insert(0, here)
+
+ mod = __import__(INSTRUMENTS[instrument], fromlist=["__main__"])
+ # Strip the instrument name from argv so the module sees its own args.
+ sys.argv = [INSTRUMENTS[instrument]] + sys.argv[2:]
+ # Each instrument module has a main() that parses sys.argv for its
+ # subcommand (validate / run / report / etc).
+ if hasattr(mod, "main"):
+ mod.main()
+ else:
+ print(f"ERROR: instrument {instrument} has no main()")
+ sys.exit(1)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/bench/__init__.py b/bench/__init__.py
new file mode 100644
index 0000000..0b54fd5
--- /dev/null
+++ b/bench/__init__.py
@@ -0,0 +1 @@
+"""Top-level bench package."""
\ No newline at end of file
diff --git a/bench/core/__init__.py b/bench/core/__init__.py
new file mode 100644
index 0000000..cf553d6
--- /dev/null
+++ b/bench/core/__init__.py
@@ -0,0 +1,7 @@
+"""Package init — exposes the core modules."""
+from . import client
+from . import runners
+from . import tools
+from . import reporting
+
+__all__ = ["client", "runners", "tools", "reporting"]
\ No newline at end of file
diff --git a/bench/core/client.py b/bench/core/client.py
new file mode 100644
index 0000000..b468dc3
--- /dev/null
+++ b/bench/core/client.py
@@ -0,0 +1,105 @@
+"""OpenAI-compatible chat-completions client used by every instrument.
+
+One thin wrapper around urllib + json. No deps. Every instrument imports
+`call_model` from here so endpoint/auth/model config lives in exactly one
+place (loaded from .env at import time, overridable via env vars).
+
+Env vars (all optional — sensible defaults):
+ ENDPOINT chat-completions URL (default http://127.0.0.1:1234/v1/chat/completions)
+ MODEL model id (default "local")
+ API_KEY bearer token, sent if set (default none)
+ MAX_TOKENS per-turn generation cap (default 8000)
+ TEMPERATURE sampling temp (default 0.0)
+ HTTP_TIMEOUT per-request wall timeout (default 900s)
+ NOTHINK "1" = append empty <think/> (default off)
+"""
+import json
+import os
+import time
+import urllib.request
+import urllib.error
+
+ENDPOINT = os.environ.get(
+ "ENDPOINT", "http://127.0.0.1:1234/v1/chat/completions")
+MODEL = os.environ.get("MODEL", "local")
+API_KEY = os.environ.get("API_KEY", "")
+MAX_TOKENS = int(os.environ.get("MAX_TOKENS", "8000"))
+TEMPERATURE = float(os.environ.get("TEMPERATURE", "0"))
+HTTP_TIMEOUT = int(os.environ.get("HTTP_TIMEOUT", "900"))
+NOTHINK = os.environ.get("NOTHINK", "") == "1"
+
+
+def call_model(messages, tools=None, *, max_tokens=None, temperature=None,
+ model=None, endpoint=None, api_key=None, http_timeout=None):
+ """Send a chat-completions request. Returns (resp_dict, elapsed_seconds).
+
+ All kwargs default to the module-level config (env-driven). Pass kwargs
+ only when an instrument needs to override per-call (e.g. router_bench
+ wants NOTHINK but coding-suite does not).
+ """
+ msgs = list(messages)
+ if NOTHINK:
+ msgs = msgs + [{"role": "assistant", "content": "<think></think>"}]
+ body = {
+ "model": model or MODEL,
+ "messages": msgs,
+ "max_tokens": max_tokens if max_tokens is not None else MAX_TOKENS,
+ "temperature": temperature if temperature is not None else TEMPERATURE,
+ }
+ if tools:
+ body["tools"] = tools
+ data = json.dumps(body).encode()
+ headers = {"Content-Type": "application/json"}
+ key = api_key if api_key is not None else API_KEY
+ if key:
+ headers["Authorization"] = "Bearer " + key
+ req = urllib.request.Request(
+ endpoint or ENDPOINT, data=data, headers=headers)
+ t0 = time.time()
+ timeout = http_timeout or HTTP_TIMEOUT
+ with urllib.request.urlopen(req, timeout=timeout) as r:
+ resp = json.load(r)
+ return resp, time.time() - t0
+
+
+def content(resp):
+ """Extract the assistant content string from a chat-completions response."""
+ return resp["choices"][0]["message"].get("content") or ""
+
+
+def tool_calls(resp):
+ """Extract the tool_calls list (empty if none)."""
+ return resp["choices"][0]["message"].get("tool_calls") or []
+
+
+def reasoning(resp):
+ """Extract reasoning_content (thinking) if the backend exposes it."""
+ msg = resp["choices"][0]["message"]
+ return msg.get("reasoning_content") or ""
+
+
+def finish_reason(resp):
+ return resp["choices"][0].get("finish_reason")
+
+
+def usage(resp):
+ """Return a normalized usage dict: prompt/completion/reasoning tokens."""
+ u = resp.get("usage") or {}
+ return {
+ "prompt_tokens": u.get("prompt_tokens") or 0,
+ "completion_tokens": u.get("completion_tokens") or 0,
+ "reasoning_tokens":
+ (u.get("completion_tokens_details") or {}).get("reasoning_tokens") or 0,
+ }
+
+
+def timings(resp):
+ """Return prefill/decode tok/s if the backend reports them (llama-server).
+
+ OpenAI-style responses don't have these; they'll be None.
+ """
+ t = resp.get("timings") or {}
+ return {
+ "prefill_ps": t.get("prompt_per_second"),
+ "decode_ps": t.get("predicted_per_second"),
+ }
\ No newline at end of file
diff --git a/bench/core/reporting.py b/bench/core/reporting.py
new file mode 100644
index 0000000..88786b9
--- /dev/null
+++ b/bench/core/reporting.py
@@ -0,0 +1,62 @@
+"""Shared reporting helpers — JSONL append, summary tables, result loading."""
+import json
+import os
+import re
+
+HERE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
+RESULTS_DIR = os.path.join(HERE, "results")
+os.makedirs(RESULTS_DIR, exist_ok=True)
+
+
+def safe_label(s):
+ """Sanitize a model name into a filesystem-safe label."""
+ return re.sub(r"[^A-Za-z0-9._-]", "_", s)
+
+
+def jsonl_path(label, prefix=""):
+ """Return the JSONL output path for a given label/prefix."""
+ name = f"{prefix}_{safe_label(label)}.jsonl" if prefix else f"{safe_label(label)}.jsonl"
+ return os.path.join(RESULTS_DIR, name)
+
+
+def append_jsonl(path, record):
+ with open(path, "a") as f:
+ f.write(json.dumps(record) + "\n")
+
+
+def load_jsonl(path):
+ if not os.path.exists(path):
+ return []
+ return [json.loads(l) for l in open(path) if l.strip()]
+
+
+def write_json(record, label, prefix=""):
+ path = os.path.join(RESULTS_DIR, f"{prefix}_{safe_label(label)}.json"
+ if prefix else f"{safe_label(label)}.json")
+ with open(path, "w") as f:
+ json.dump(record, f, indent=2)
+ return path
+
+
+def coding_suite_summary(results):
+ """Print + return the coding-suite scoreboard."""
+ npass = sum(1 for r in results if r.get("passed"))
+ print("\n========================= SUMMARY =========================")
+ print("Overall: %d/%d passed (%.0f%%)\n" %
+ (npass, len(results), 100 * npass / max(len(results), 1)))
+ cats = {}
+ for r in results:
+ cats.setdefault(r["cat"], []).append(r)
+ for cat, rs in cats.items():
+ p = sum(1 for r in rs if r.get("passed"))
+ print(" %-9s %d/%d" % (cat, p, len(rs)))
+ print("\n %-16s %-5s %-6s %-7s %-9s %s" %
+ ("task", "pass", "lat_s", "tot_tok", "reason_tok", "note"))
+ for r in results:
+ print(" %-16s %-5s %-6s %-7s %-9s %s" % (
+ r["id"], "Y" if r.get("passed") else "n",
+ r.get("latency_s", "-"), r.get("total_tokens", "-"),
+ r.get("reasoning_tokens", "-"),
+ ("" if r.get("passed") else (r.get("detail", "") or "")[:80])))
+ print()
+ return npass == len(results)
\ No newline at end of file
diff --git a/bench/core/runners.py b/bench/core/runners.py
new file mode 100644
index 0000000..2521c3c
--- /dev/null
+++ b/bench/core/runners.py
@@ -0,0 +1,135 @@
+"""Language runners — compile + execute model output against hidden tests.
+
+Each runner takes a task dict (id, lang, symptom, buggy, fix, test) and
+model-generated code, writes it to a fresh temp dir, runs the real toolchain,
+and returns (passed: bool, detail: str).
+
+Supported languages: Go, Rust, TypeScript (via bun), Python, Swift (if xcrun
+available). The coding-suite tasks reference these by their runner key.
+
+Adding a new language: add a `run_<lang>(task, code)` function here and
+register it in RUNNERS.
+"""
+import os
+import re
+import subprocess
+
+WORK = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "work")
+RUN_TIMEOUT = 120 # seconds per compile+run
+
+
+def _freshdir(task_id, work=WORK):
+ d = os.path.join(work, task_id)
+ subprocess.run(["rm", "-rf", d])
+ os.makedirs(d, exist_ok=True)
+ return d
+
+
+def _sh(cmd, cwd, env=None):
+ p = subprocess.run(cmd, cwd=cwd, capture_output=True, text=True,
+ timeout=RUN_TIMEOUT, env=env)
+ return p.returncode, (p.stdout + "\n" + p.stderr).strip()
+
+
+def run_go(task, code):
+ d = _freshdir(task["id"])
+ code = re.sub(r'^\s*package\s+\w+', 'package solution', code, count=1, flags=re.M)
+ if not re.search(r'^\s*package\s+', code, flags=re.M):
+ code = "package solution\n\n" + code
+ open(os.path.join(d, "solution.go"), "w").write(code)
+ open(os.path.join(d, "solution_test.go"), "w").write(task["test"])
+ open(os.path.join(d, "go.mod"), "w").write("module bench\n\ngo 1.26\n")
+ try:
+ rc, out = _sh(["go", "test", "./..."], d)
+ except subprocess.TimeoutExpired:
+ return False, "TIMEOUT"
+ return rc == 0, out[-1500:]
+
+
+def run_rust(task, code):
+ d = _freshdir(task["id"])
+ src = code + "\n\n" + task["test"]
+ open(os.path.join(d, "solution.rs"), "w").write(src)
+ try:
+ rc, out = _sh(["rustc", "--edition", "2021", "--test",
+ "solution.rs", "-o", "tb"], d)
+ if rc != 0:
+ return False, "COMPILE:\n" + out[-1500:]
+ rc, out = _sh([os.path.join(d, "tb")], d)
+ except subprocess.TimeoutExpired:
+ return False, "TIMEOUT"
+ return rc == 0, out[-1500:]
+
+
+def run_ts(task, code):
+ d = _freshdir(task["id"])
+ open(os.path.join(d, "solution.ts"), "w").write(code)
+ open(os.path.join(d, "runtests.ts"), "w").write(task["test"])
+ try:
+ rc, out = _sh(["bun", "run", "runtests.ts"], d)
+ except subprocess.TimeoutExpired:
+ return False, "TIMEOUT"
+ return rc == 0, out[-1500:]
+
+
+def run_python(task, code):
+ d = _freshdir(task["id"])
+ open(os.path.join(d, "solution.py"), "w").write(code)
+ open(os.path.join(d, "test_solution.py"), "w").write(task["test"])
+ try:
+ rc, out = _sh([os.sys.executable, "test_solution.py"], d)
+ except subprocess.TimeoutExpired:
+ return False, "TIMEOUT"
+ return rc == 0, out[-1500:]
+
+
+_SWIFT_SDK = None
+
+def _swift_env():
+ global _SWIFT_SDK
+ env = {k: v for k, v in os.environ.items() if k != "SDKROOT"}
+ if _SWIFT_SDK is None:
+ try:
+ p = subprocess.run(["xcrun", "--sdk", "macosx", "--show-sdk-path"],
+ capture_output=True, text=True, env=env)
+ _SWIFT_SDK = p.stdout.strip()
+ except FileNotFoundError:
+ _SWIFT_SDK = ""
+ return env, _SWIFT_SDK
+
+
+def run_swift(task, code):
+ d = _freshdir(task["id"])
+ open(os.path.join(d, "solution.swift"), "w").write(code)
+ open(os.path.join(d, "main.swift"), "w").write(task["test"])
+ env, sdk = _swift_env()
+ if not sdk:
+ return False, "SKIP: xcrun/Swift not available on this platform"
+ try:
+ rc, out = _sh(["xcrun", "swiftc", "solution.swift", "main.swift",
+ "-sdk", sdk, "-o", "bin"], d, env=env)
+ if rc != 0:
+ return False, "COMPILE:\n" + out[-1500:]
+ rc, out = _sh([os.path.join(d, "bin")], d, env=env)
+ except subprocess.TimeoutExpired:
+ return False, "TIMEOUT"
+ return rc == 0, out[-1500:]
+
+
+RUNNERS = {
+ "go": run_go,
+ "rust": run_rust,
+ "ts": run_ts,
+ "python": run_python,
+ "swift": run_swift,
+}
+
+
+def extract_code(content):
+ """Return the most plausible code block from a chat content string."""
+ if not content:
+ return ""
+ blocks = re.findall(r"```[a-zA-Z0-9_+#.-]*\n(.*?)```", content, re.DOTALL)
+ if blocks:
+ return max(blocks, key=len).strip()
+ return content.strip()
\ No newline at end of file
diff --git a/bench/core/tools.py b/bench/core/tools.py
new file mode 100644
index 0000000..503a2a6
--- /dev/null
+++ b/bench/core/tools.py
@@ -0,0 +1,205 @@
+"""File/exec tool implementations for agentic build benches.
+
+These are the tools the agent can call during a multi-turn build loop:
+list_files, read_file, write_file, edit_file, search, run_build, run_test,
+run_shell. They operate on a work dir set via `set_work_dir()`.
+
+The tool SCHEMA (OpenAI function-calling format) is in TOOLS; the
+implementations are in TOOL_IMPLS. Instruments that want a different tool
+set (e.g. agentic_bench_v1 uses read-only exploration + submit_fix) define
+their own and don't import these.
+"""
+import os
+import re
+import subprocess
+import shutil
+
+BUILD_TIMEOUT = int(os.environ.get("BUILD_TIMEOUT", "180"))
+TEST_TIMEOUT = int(os.environ.get("TEST_TIMEOUT", "180"))
+SHELL_TIMEOUT = int(os.environ.get("SHELL_TIMEOUT", "60"))
+
+_WORK = None
+
+def set_work_dir(d):
+ global _WORK
+ _WORK = d
+
+def get_work_dir():
+ return _WORK
+
+
+def materialize(files, dest=None):
+ """Write a dict of {relative_path: content} into a work dir (cleared first)."""
+ d = dest or _WORK
+ shutil.rmtree(d, ignore_errors=True)
+ os.makedirs(d, exist_ok=True)
+ for path, content in files.items():
+ fp = os.path.join(d, path)
+ os.makedirs(os.path.dirname(fp), exist_ok=True)
+ with open(fp, "w") as f:
+ f.write(content)
+ return d
+
+
+def _run(cmd, timeout):
+ try:
+ p = subprocess.run(cmd, cwd=_WORK, shell=True, capture_output=True,
+ text=True, timeout=timeout)
+ except subprocess.TimeoutExpired:
+ return f"TIMEOUT after {timeout}s"
+ out = (p.stdout + "\n" + p.stderr).strip()
+ return f"exit={p.returncode}\n{out}"
+
+
+def _list_files(args):
+ out = []
+ for root, dirs, files in os.walk(_WORK):
+ dirs[:] = sorted(d for d in dirs if d not in {".git", "node_modules",
+ ".next", "target", "dist"})
+ for f in sorted(files):
+ full = os.path.join(root, f)
+ rel = os.path.relpath(full, _WORK)
+ out.append(rel)
+ return "\n".join(out)
+
+
+def _read_file(args):
+ p = args.get("path", "")
+ fp = os.path.join(_WORK, p)
+ if not os.path.isfile(fp):
+ return f"ERROR: no such file {p!r}. Use list_files."
+ with open(fp) as f:
+ return f.read()
+
+
+def _write_file(args):
+ p = args.get("path", "")
+ content = args.get("content", "")
+ fp = os.path.join(_WORK, p)
+ os.makedirs(os.path.dirname(fp), exist_ok=True)
+ with open(fp, "w") as f:
+ f.write(content)
+ return f"wrote {p} ({len(content)} bytes)"
+
+
+def _edit_file(args):
+ p = args.get("path", "")
+ old = args.get("old_str", "")
+ new = args.get("new_str", "")
+ fp = os.path.join(_WORK, p)
+ if not os.path.isfile(fp):
+ return f"ERROR: no such file {p!r}."
+ with open(fp) as f:
+ content = f.read()
+ count = content.count(old)
+ if count == 0:
+ return f"ERROR: old_str not found in {p}."
+ if count > 1:
+ return f"ERROR: old_str appears {count} times in {p}; make it unique."
+ new_content = content.replace(old, new, 1)
+ with open(fp, "w") as f:
+ f.write(new_content)
+ return f"edited {p} ({len(old)} -> {len(new)} chars)"
+
+
+def _search(args):
+ pat = args.get("pattern", "")
+ try:
+ rx = re.compile(pat)
+ except re.error as e:
+ return f"ERROR: bad regex: {e}"
+ hits = []
+ for root, dirs, files in os.walk(_WORK):
+ dirs[:] = sorted(d for d in dirs if d not in {".git", "node_modules",
+ ".next", "target", "dist"})
+ for f in sorted(files):
+ full = os.path.join(root, f)
+ rel = os.path.relpath(full, _WORK)
+ with open(full, errors="replace") as fh:
+ for i, line in enumerate(fh, 1):
+ if rx.search(line):
+ hits.append(f"{rel}:{i}: {line.strip()}")
+ return "\n".join(hits) if hits else "(no matches)"
+
+
+def _run_build(args):
+ return _run("go build ./...", BUILD_TIMEOUT)
+
+
+def _run_test(args):
+ return _run("go test ./...", TEST_TIMEOUT)
+
+
+def _run_shell(args):
+ cmd = args.get("cmd", "")
+ if not cmd:
+ return "ERROR: empty cmd"
+ return _run(cmd, SHELL_TIMEOUT)
+
+
+TOOL_IMPLS = {
+ "list_files": _list_files,
+ "read_file": _read_file,
+ "write_file": _write_file,
+ "edit_file": _edit_file,
+ "search": _search,
+ "run_build": _run_build,
+ "run_test": _run_test,
+ "run_shell": _run_shell,
+}
+
+TOOLS = [
+ {"type": "function", "function": {
+ "name": "list_files",
+ "description": "List all file paths in the work dir. Returns one path per line.",
+ "parameters": {"type": "object", "properties": {}},
+ }},
+ {"type": "function", "function": {
+ "name": "read_file",
+ "description": "Return the full contents of one file in the work dir.",
+ "parameters": {"type": "object", "properties": {
+ "path": {"type": "string", "description": "Project-relative file path"},
+ }, "required": ["path"]},
+ }},
+ {"type": "function", "function": {
+ "name": "write_file",
+ "description": "Create or overwrite a file in the work dir with the given contents.",
+ "parameters": {"type": "object", "properties": {
+ "path": {"type": "string"},
+ "content": {"type": "string", "description": "Full file contents to write"},
+ }, "required": ["path", "content"]},
+ }},
+ {"type": "function", "function": {
+ "name": "edit_file",
+ "description": "Apply a patch-style edit: replace the first occurrence of old_str with new_str. Fails if old_str is not found or appears more than once.",
+ "parameters": {"type": "object", "properties": {
+ "path": {"type": "string"},
+ "old_str": {"type": "string", "description": "The exact text to find (must be unique)"},
+ "new_str": {"type": "string", "description": "The replacement text"},
+ }, "required": ["path", "old_str", "new_str"]},
+ }},
+ {"type": "function", "function": {
+ "name": "search",
+ "description": "Regex-search the whole work dir; returns matching path:line: text.",
+ "parameters": {"type": "object", "properties": {
+ "pattern": {"type": "string"},
+ }, "required": ["pattern"]},
+ }},
+ {"type": "function", "function": {
+ "name": "run_build",
+ "description": f"Run `go build ./...` in the work dir. Returns exit code + output. Timeout {BUILD_TIMEOUT}s.",
+ "parameters": {"type": "object", "properties": {}},
+ }},
+ {"type": "function", "function": {
+ "name": "run_test",
+ "description": f"Run `go test ./...` in the work dir. Returns exit code + output. Timeout {TEST_TIMEOUT}s.",
+ "parameters": {"type": "object", "properties": {}},
+ }},
+ {"type": "function", "function": {
+ "name": "run_shell",
+ "description": f"Run an arbitrary shell command in the work dir (cwd = work dir). Use for `go get`, `go mod tidy`, npm, cargo, etc. Returns exit code + output. Timeout {SHELL_TIMEOUT}s.",
+ "parameters": {"type": "object", "properties": {
+ "cmd": {"type": "string", "description": "The shell command to run (sh -c)"},
+ }, "required": ["cmd"]},
+ }},
+]
\ No newline at end of file
diff --git a/bench/tasks/__init__.py b/bench/tasks/__init__.py
new file mode 100644
index 0000000..ba9dd56
--- /dev/null
+++ b/bench/tasks/__init__.py
@@ -0,0 +1,6 @@
+"""Task instruments for the benchmark framework.
+
+Each module is an instrument (coding-suite, agentic-fix, agentic-build,
+agentic-multitask, router, needle, loop-battery, no-think) with a
+validate() and run() function, importable by the bench.py CLI.
+"""
\ No newline at end of file
diff --git a/bench/tasks/agentic_build.py b/bench/tasks/agentic_build.py
new file mode 100644
index 0000000..c0f8405
--- /dev/null
+++ b/bench/tasks/agentic_build.py
@@ -0,0 +1,572 @@
+"""Agentic build instrument — "build a real small app" end-to-end.
+
+Given a plan and a thin starter scaffold, the agent uses tools to write ~8
+files, get `go build` green, get `go test` green, and produce a binary that
+starts, serves an HTML form, and persists a POSTed note to SQLite. The four
+"real product" gates are run by the harness, not the agent.
+
+This is a Go + SQLite + html/template app — a realistic small self-hosted
+web tool in miniature. It tests the actual workload: multi-turn tool-use,
+file creation, dependency management, iterative build/test cycles.
+
+Usage (via the bench.py CLI):
+ python bench.py agentic-build validate
+ python bench.py agentic-build run
+
+Env (via .env): ENDPOINT, MODEL, API_KEY, MAX_TOKENS, MAX_TURNS.
+"""
+import json
+import os
+import re
+import sys
+import time
+import shutil
+import subprocess
+import socket
+import signal
+import urllib.request
+import urllib.parse
+import urllib.error
+
+from bench.core import client
+from bench.core import tools
+from bench.core.reporting import write_json
+
+HERE = os.path.dirname(os.path.abspath(__file__))
+WORK = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "work", "agentic_build")
+MAX_TOKENS = int(os.environ.get("MAX_TOKENS", "16000"))
+MAX_TURNS = int(os.environ.get("MAX_TURNS", "40"))
+BUILD_TIMEOUT = int(os.environ.get("BUILD_TIMEOUT", "180"))
+TEST_TIMEOUT = int(os.environ.get("TEST_TIMEOUT", "180"))
+SERVER_TIMEOUT = int(os.environ.get("SERVER_TIMEOUT", "20"))
+
+PLAN = """\
+# Task — Build a Go web binary with a SQLite-backed notes CRUD
+
+## Goal
+`go build ./...` and `go test ./...` run green in the work dir, and the
+resulting binary starts on port 8080, serves an HTML form at `/` that lists
+existing notes and lets a visitor create a new one, and persists created
+notes to an embedded SQLite file `notes.db`.
+
+## Prerequisites
+A starter scaffold is already in the work dir:
+ - `go.mod` (module `notes`, go 1.26, no deps yet)
+ - `cmd/server/main.go` (empty stub: `package main; func main(){}`)
+You start from this scaffold. Do not delete existing files; build on them.
+
+## Spec — the app to build
+A single Go binary that serves a one-page HTML UI over HTTP. One `cmd/`,
+modular `internal/` packages, server-rendered `html/template`, embedded
+assets, embedded SQLite via the pure-Go driver `modernc.org/sqlite`.
+
+### Files to produce (layout)
+ cmd/server/main.go boot: config -> db -> migrate -> handlers -> ListenAndServe + graceful shutdown
+ internal/config/config.go Config{Port, DBPath}; LoadFromEnv() with sane defaults (port 8080, db "notes.db"). MUST read env vars NOTES_PORT (port) and NOTES_DB (db path) — the scoring harness sets these.
+ internal/db/db.go Open(dsn): open modernc sqlite, PRAGMA WAL + busy_timeout + foreign_keys ON; Migrate() via embedded .sql
+ internal/db/migrations/00001_init.sql CREATE TABLE notes (id INTEGER PRIMARY KEY, title TEXT NOT NULL, body TEXT NOT NULL, created_at DATETIME DEFAULT CURRENT_TIMESTAMP);
+ internal/models/note.go Note struct + Create(db, title, body) + List(db) + Delete(db, id); plain database/sql, no ORM
+ internal/handlers/handlers.go Handlers struct; routes via stdlib net/http ServeMux: GET / (list+form), POST / (create), POST /{id}/delete (delete)
+ internal/handlers/handlers_test.go httptest: POST a note, GET /, assert the note title appears in the response body
+ web/templates/index.html the one page: a form (title + body + submit) above a list of notes (each with a delete button); rendered server-side by html/template
+
+### Constraints
+ - Use `modernc.org/sqlite` (pure-Go, CGO_ENABLED=0). Import path is `_ "modernc.org/sqlite"`.
+ - Embed the migration SQL and the templates via `//go:embed`.
+ - Stdlib `net/http` ServeMux only — no Gin, no chi.
+ - No JS framework. Plain HTML form posts back to the server.
+
+## Steps
+1. Add the dependency: `go get modernc.org/sqlite`. Run `go mod tidy`.
+2. Write the files. Natural order: config -> db -> models -> handlers -> templates -> main.
+3. Build (`go build ./...`). Fix any compile errors.
+4. Test (`go test ./...`). The handlers_test.go must pass.
+5. Optionally run the binary briefly to confirm it serves.
+
+## Acceptance (harness-verified)
+1. `go build ./...` exits 0.
+2. `go test ./...` exits 0.
+3. The built binary starts on port 8080 and `GET /` returns HTTP 200 with a `<form` element.
+4. `POST /` with form fields title+body creates a row; a subsequent `GET /` shows that title; the SQLite file on disk contains the row.
+
+## Out of scope
+Auth, sessions, CSRF, styling beyond minimal readable HTML, HTMX, pagination, any JS.
+
+## Commit
+You do not commit; the harness owns the work dir.
+"""
+
+STARTER = {
+ "go.mod": "module notes\n\ngo 1.26\n",
+ "cmd/server/main.go": "package main\n\nfunc main() {}\n",
+}
+
+REFERENCE = {
+ "go.mod": "module notes\n\ngo 1.26\n\nrequire modernc.org/sqlite v1.50.0\n",
+ "cmd/server/main.go": '''package main
+
+import (
+ "context"
+ "fmt"
+ "log"
+ "net/http"
+ "os"
+ "os/signal"
+ "syscall"
+ "time"
+
+ "notes/internal/config"
+ "notes/internal/db"
+ "notes/internal/handlers"
+ "notes/internal/models"
+)
+
+func main() {
+ cfg := config.LoadFromEnv()
+ database, err := db.Open(cfg.DBPath)
+ if err != nil { log.Fatalf("db open: %v", err) }
+ defer database.Close()
+ if err := database.Migrate(); err != nil { log.Fatalf("migrate: %v", err) }
+ notes := &models.Store{DB: database.DB}
+ h := handlers.New(notes)
+ srv := &http.Server{Addr: fmt.Sprintf(":%d", cfg.Port), Handler: h.Routes()}
+ go func() {
+ log.Printf("listening on :%d", cfg.Port)
+ if err := srv.ListenAndServe(); err != nil && err != http.ErrServerClosed {
+ log.Fatalf("serve: %v", err)
+ }
+ }()
+ stop := make(chan os.Signal, 1)
+ signal.Notify(stop, os.Interrupt, syscall.SIGTERM)
+ <-stop
+ log.Println("shutting down")
+ ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
+ defer cancel()
+ _ = srv.Shutdown(ctx)
+}
+''',
+ "internal/config/config.go": '''package config
+
+import ("os"; "strconv")
+
+type Config struct { Port int; DBPath string }
+
+func LoadFromEnv() Config {
+ cfg := Config{Port: 8080, DBPath: "notes.db"}
+ if p := os.Getenv("NOTES_PORT"); p != "" {
+ if v, err := strconv.Atoi(p); err == nil { cfg.Port = v }
+ }
+ if d := os.Getenv("NOTES_DB"); d != "" { cfg.DBPath = d }
+ return cfg
+}
+''',
+ "internal/db/db.go": '''package db
+
+import ("database/sql"; "embed"; "fmt"; _ "modernc.org/sqlite")
+
+//go:embed migrations/*.sql
+var migrationsFS embed.FS
+
+type Database struct { *sql.DB }
+
+func Open(dsn string) (*Database, error) {
+ db, err := sql.Open("sqlite", dsn)
+ if err != nil { return nil, fmt.Errorf("open sqlite: %w", err) }
+ if _, err := db.Exec(`PRAGMA journal_mode=WAL; PRAGMA busy_timeout=5000; PRAGMA foreign_keys=ON;`); err != nil {
+ db.Close(); return nil, fmt.Errorf("pragmas: %w", err)
+ }
+ return &Database{DB: db}, nil
+}
+
+func (d *Database) Migrate() error {
+ if _, err := d.Exec(`CREATE TABLE IF NOT EXISTS schema_migrations (name TEXT PRIMARY KEY, applied_at DATETIME DEFAULT CURRENT_TIMESTAMP)`); err != nil {
+ return fmt.Errorf("create schema_migrations: %w", err)
+ }
+ entries, err := migrationsFS.ReadDir("migrations")
+ if err != nil { return fmt.Errorf("read migrations dir: %w", err) }
+ for _, e := range entries {
+ if e.IsDir() { continue }
+ name := e.Name()
+ var applied int
+ if err := d.QueryRow(`SELECT count(*) FROM schema_migrations WHERE name = ?`, name).Scan(&applied); err != nil {
+ return fmt.Errorf("check %s: %w", name, err)
+ }
+ if applied > 0 { continue }
+ b, err := migrationsFS.ReadFile("migrations/" + name)
+ if err != nil { return fmt.Errorf("read %s: %w", name, err) }
+ if _, err := d.Exec(string(b)); err != nil { return fmt.Errorf("apply %s: %w", name, err) }
+ if _, err := d.Exec(`INSERT INTO schema_migrations (name) VALUES (?)`, name); err != nil {
+ return fmt.Errorf("record %s: %w", name, err)
+ }
+ }
+ return nil
+}
+''',
+ "internal/db/migrations/00001_init.sql": '''CREATE TABLE IF NOT EXISTS notes (id INTEGER PRIMARY KEY AUTOINCREMENT, title TEXT NOT NULL, body TEXT NOT NULL, created_at DATETIME DEFAULT CURRENT_TIMESTAMP);''',
+ "internal/models/note.go": '''package models
+
+import ("database/sql"; "time")
+
+type Note struct { ID int64; Title string; Body string; CreatedAt time.Time }
+type Store struct { DB *sql.DB }
+
+func (s *Store) Create(title, body string) (*Note, error) {
+ res, err := s.DB.Exec(`INSERT INTO notes (title, body) VALUES (?, ?)`, title, body)
+ if err != nil { return nil, err }
+ id, err := res.LastInsertId()
+ if err != nil { return nil, err }
+ return &Note{ID: id, Title: title, Body: body, CreatedAt: time.Now()}, nil
+}
+
+func (s *Store) List() ([]Note, error) {
+ rows, err := s.DB.Query(`SELECT id, title, body, created_at FROM notes ORDER BY id DESC`)
+ if err != nil { return nil, err }
+ defer rows.Close()
+ var out []Note
+ for rows.Next() {
+ var n Note
+ if err := rows.Scan(&n.ID, &n.Title, &n.Body, &n.CreatedAt); err != nil { return nil, err }
+ out = append(out, n)
+ }
+ return out, rows.Err()
+}
+
+func (s *Store) Delete(id int64) (bool, error) {
+ res, err := s.DB.Exec(`DELETE FROM notes WHERE id = ?`, id)
+ if err != nil { return false, err }
+ n, err := res.RowsAffected()
+ if err != nil { return false, err }
+ return n > 0, nil
+}
+''',
+ "internal/handlers/handlers.go": '''package handlers
+
+import ("embed"; "html/template"; "log"; "net/http"; "strconv"; "notes/internal/models")
+
+//go:embed templates/*.html
+var templatesFS embed.FS
+
+type Handlers struct { notes *models.Store; tmpl *template.Template }
+
+func New(notes *models.Store) *Handlers {
+ tmpl := template.Must(template.ParseFS(templatesFS, "templates/index.html"))
+ return &Handlers{notes: notes, tmpl: tmpl}
+}
+
+func (h *Handlers) Routes() http.Handler {
+ mux := http.NewServeMux()
+ mux.HandleFunc("GET /", h.index)
+ mux.HandleFunc("POST /", h.create)
+ mux.HandleFunc("POST /{id}/delete", h.delete)
+ return mux
+}
+
+func (h *Handlers) index(w http.ResponseWriter, r *http.Request) {
+ notes, err := h.notes.List()
+ if err != nil { log.Printf("list: %v", err); http.Error(w, "internal error", http.StatusInternalServerError); return }
+ w.Header().Set("Content-Type", "text/html; charset=utf-8")
+ if err := h.tmpl.Execute(w, notes); err != nil { log.Printf("render: %v", err) }
+}
+
+func (h *Handlers) create(w http.ResponseWriter, r *http.Request) {
+ title := r.FormValue("title"); body := r.FormValue("body")
+ if title == "" || body == "" { http.Error(w, "title and body are required", http.StatusBadRequest); return }
+ if _, err := h.notes.Create(title, body); err != nil { log.Printf("create: %v", err); http.Error(w, "internal error", http.StatusInternalServerError); return }
+ http.Redirect(w, r, "/", http.StatusSeeOther)
+}
+
+func (h *Handlers) delete(w http.ResponseWriter, r *http.Request) {
+ idStr := r.PathValue("id"); id, err := strconv.ParseInt(idStr, 10, 64)
+ if err != nil { http.Error(w, "bad id", http.StatusBadRequest); return }
+ if _, err := h.notes.Delete(id); err != nil { log.Printf("delete: %v", err); http.Error(w, "internal error", http.StatusInternalServerError); return }
+ http.Redirect(w, r, "/", http.StatusSeeOther)
+}
+''',
+ "internal/handlers/templates/index.html": '''<!DOCTYPE html><html lang="en"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width, initial-scale=1"><title>Notes</title><style>body{font-family:system-ui,sans-serif;max-width:40em;margin:2rem auto;padding:0 1rem}form{margin:1rem 0;padding:1rem;border:1px solid #ccc;border-radius:6px}label{display:block;margin:0.5rem 0}input[type=text],textarea{width:100%;box-sizing:border-box;padding:0.4rem}.note{margin:0.75rem 0;padding:0.75rem;border-bottom:1px solid #eee}form.delete{display:inline}form.delete button{background:none;border:none;color:#c00;cursor:pointer}</style></head><body><h1>Notes</h1><form method="POST" action="/"><label>Title <input type="text" name="title" required></label><label>Body <textarea name="body" rows="4" required></textarea></label><button type="submit">Add note</button></form>{{range .}}<div class="note"><h3>{{.Title}}</h3><p>{{.Body}}</p><form class="delete" method="POST" action="/{{.ID}}/delete"><button type="submit">delete</button></form></div>{{else}}<p>No notes yet.</p>{{end}}</body></html>''',
+ "internal/handlers/handlers_test.go": '''package handlers
+
+import ("database/sql"; "io"; "net/http"; "net/http/httptest"; "net/url"; "strings"; "testing"; "notes/internal/models"; _ "modernc.org/sqlite")
+
+func newTestStore(t *testing.T) (*models.Store, func()) {
+ t.Helper()
+ db, err := sql.Open("sqlite", ":memory:")
+ if err != nil { t.Fatalf("open: %v", err) }
+ _, err = db.Exec(`CREATE TABLE notes (id INTEGER PRIMARY KEY AUTOINCREMENT, title TEXT NOT NULL, body TEXT NOT NULL, created_at DATETIME DEFAULT CURRENT_TIMESTAMP)`)
+ if err != nil { db.Close(); t.Fatalf("create table: %v", err) }
+ return &models.Store{DB: db}, func() { db.Close() }
+}
+
+func TestCreateThenList(t *testing.T) {
+ store, cleanup := newTestStore(t); defer cleanup()
+ h := New(store)
+ srv := httptest.NewServer(h.Routes()); defer srv.Close()
+ form := url.Values{"title": {"My Note"}, "body": {"First body"}}
+ resp, err := http.PostForm(srv.URL+"/", form)
+ if err != nil { t.Fatalf("post: %v", err) }
+ resp.Body.Close()
+ if resp.StatusCode != http.StatusOK { t.Fatalf("post status = %d, want 200", resp.StatusCode) }
+ resp2, err := http.Get(srv.URL + "/")
+ if err != nil { t.Fatalf("get: %v", err) }
+ defer resp2.Body.Close()
+ body, _ := io.ReadAll(resp2.Body)
+ if !strings.Contains(string(body), "My Note") { t.Fatalf("response does not contain the created note title; got:\\n%s", body) }
+}
+''',
+ "internal/handlers/doc.go": "package handlers",
+ "internal/models/doc.go": "package models",
+ "internal/config/doc.go": "package config",
+ "internal/db/doc.go": "package db",
+}
+
+# ----------------------------------------------------------------- scoring gates
+def _has_real_app(work_dir):
+ main_go = os.path.join(work_dir, "cmd", "server", "main.go")
+ if not os.path.isfile(main_go): return False
+ with open(main_go) as f: src = f.read()
+ if "ListenAndServe" not in src and "http.Server" not in src: return False
+ internal = os.path.join(work_dir, "internal")
+ if not os.path.isdir(internal): return False
+ pkg_dirs = [d for d in os.listdir(internal) if os.path.isdir(os.path.join(internal, d))]
+ return len(pkg_dirs) >= 3
+
+def _run(cmd, work_dir, timeout):
+ try:
+ p = subprocess.run(cmd, cwd=work_dir, shell=True, capture_output=True, text=True, timeout=timeout)
+ except subprocess.TimeoutExpired:
+ return f"TIMEOUT after {timeout}s"
+ return f"exit={p.returncode}\n{(p.stdout + p.stderr).strip()}"
+
+def _free_port():
+ s = socket.socket(socket.AF_INET, socket.SOCK_STREAM); s.bind(("", 0))
+ p = s.getsockname()[1]; s.close(); return p
+
+def gates(work_dir):
+ g = {}
+ if not _has_real_app(work_dir):
+ return {"build": False, "test": False, "serves": False, "persists": False}
+ b = _run("go build ./...", work_dir, BUILD_TIMEOUT)
+ g["build"] = b.startswith("exit=0")
+ t = _run("go test ./...", work_dir, TEST_TIMEOUT)
+ g["test"] = t.startswith("exit=0")
+ binpath = os.path.abspath(os.path.join(work_dir, "server"))
+ r = _run(f"go build -o {binpath} ./cmd/server", work_dir, BUILD_TIMEOUT)
+ if not r.startswith("exit=0") or not os.path.isfile(binpath):
+ g["serves"] = False; g["persists"] = False; return g
+ port = _free_port()
+ dbpath = os.path.abspath(os.path.join(work_dir, f"bench_{port}.db"))
+ for ext in ("", "-wal", "-shm"):
+ if os.path.exists(dbpath + ext): os.remove(dbpath + ext)
+ env = dict(os.environ, NOTES_PORT=str(port), NOTES_DB=dbpath)
+ try:
+ proc = subprocess.Popen([binpath], cwd=work_dir, env=env, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
+ except Exception:
+ g["serves"] = False; g["persists"] = False; return g
+ try:
+ for _ in range(40):
+ try:
+ with socket.create_connection(("127.0.0.1", port), 0.25): break
+ except OSError: time.sleep(0.25)
+ else:
+ g["serves"] = False; g["persists"] = False; return g
+ base = f"http://127.0.0.1:{port}"
+ try:
+ with urllib.request.urlopen(base + "/", timeout=SERVER_TIMEOUT) as resp:
+ if resp.status != 200:
+ g["serves"] = False; g["persists"] = False; return g
+ html = resp.read().decode(errors="replace")
+ except Exception:
+ g["serves"] = False; g["persists"] = False; return g
+ g["serves"] = "<form" in html
+ if not g["serves"]:
+ g["persists"] = False; return g
+ data = urllib.parse.urlencode({"title": "BenchNote", "body": "created by the gate"}).encode()
+ try:
+ req = urllib.request.Request(base + "/", data=data, method="POST")
+ with urllib.request.urlopen(req, timeout=SERVER_TIMEOUT) as resp: _ = resp.read()
+ except Exception:
+ g["persists"] = False; return g
+ try:
+ with urllib.request.urlopen(base + "/", timeout=SERVER_TIMEOUT) as resp:
+ after = resp.read().decode(errors="replace")
+ except Exception:
+ g["persists"] = False; return g
+ g["persists"] = "BenchNote" in after
+ if g["persists"] and os.path.exists(dbpath):
+ try:
+ import sqlite3
+ conn = sqlite3.connect(dbpath)
+ cur = conn.execute("SELECT count(*) FROM notes WHERE title = 'BenchNote'")
+ g["persists"] = cur.fetchone()[0] > 0
+ conn.close()
+ except Exception: pass
+ return g
+ finally:
+ proc.send_signal(signal.SIGTERM)
+ try: proc.wait(timeout=5)
+ except subprocess.TimeoutExpired: proc.kill()
+ for ext in ("", "-wal", "-shm"):
+ p = dbpath + ext
+ if os.path.exists(p):
+ try: os.remove(p)
+ except: pass
+
+# ----------------------------------------------------------------- phases
+def validate():
+ print("=== VALIDATION: reference app builds, tests, and serves ===\n")
+ tools.set_work_dir(WORK)
+ tools.materialize(REFERENCE, dest=WORK)
+ r = _run("go mod tidy", WORK, BUILD_TIMEOUT)
+ print("go mod tidy:", r[:200])
+ g = gates(WORK)
+ print(f"gates: {g}")
+ ok = all(g.values())
+ print(f"\n-> {'OK' if ok else 'BAD'}: reference app is {'buildable + working' if ok else 'BROKEN'}")
+ return ok
+
+SYSTEM = (
+ "You are a coding agent working in a Go project work dir. You are given "
+ "a task plan below. Execute it step by step using the tools. You can "
+ "write files, edit files, run builds and tests, and run shell commands "
+ "(e.g. `go get`, `go mod tidy`) in the work dir. Keep going until the "
+ "build is green, the tests pass, and the binary would work. Use the "
+ "tools to verify your own progress — don't guess; compile and test.\n\n"
+ "When you are confident the app is complete and the tests pass, say "
+ "`DONE` in your final message (no tool call) and stop.\n\n"
+ "TASK PLAN:\n" + PLAN
+)
+
+def run():
+ tools.set_work_dir(WORK)
+ tools.materialize(STARTER, dest=WORK)
+ messages = [
+ {"role": "system", "content": SYSTEM},
+ {"role": "user", "content":
+ "The work dir has the starter scaffold (go.mod + a stub main.go). "
+ "Execute the task plan. Begin by adding the dependency and writing "
+ "the files. Use the tools to build and test as you go."},
+ ]
+ turns_log = []
+ peak_prompt = total_completion = total_reasoning = 0
+ finish_reasons = []
+ t0 = time.time()
+ turns = 0
+ done_said = False
+ no_tool_streak = 0
+ for turn in range(1, MAX_TURNS + 1):
+ turns = turn
+ try:
+ resp, dt = client.call_model(messages, tools=tools.TOOLS, max_tokens=MAX_TOKENS)
+ except Exception as e:
+ print(f" turn {turn}: REQUEST ERROR {e!r}")
+ turns_log.append({"turn": turn, "error": repr(e)})
+ break
+ u = client.usage(resp)
+ peak_prompt = max(peak_prompt, u["prompt_tokens"])
+ total_completion += u["completion_tokens"]
+ total_reasoning += u["reasoning_tokens"]
+ choice = resp["choices"][0]
+ fr = choice.get("finish_reason")
+ finish_reasons.append(fr)
+ msg = choice["message"]
+ tcs = msg.get("tool_calls") or []
+ content = msg.get("content") or ""
+ tlog = {"turn": turn, "dt": round(dt, 1), "prompt_tok": u["prompt_tokens"],
+ "completion_tok": u["completion_tokens"], "reasoning_tok": u["reasoning_tokens"],
+ "finish_reason": fr, "n_tool_calls": len(tcs)}
+ turns_log.append(tlog)
+ tc_summary = ",".join((tc["function"]["name"] for tc in tcs)) or "(no tool call)"
+ print(f" turn {turn:2d}: {dt:5.1f}s ptok={u['prompt_tokens']:6d} "
+ f"ctok={u['completion_tokens']:5d} rtok={u['reasoning_tokens']:5d} "
+ f"fr={fr:8s} tools=[{tc_summary}]")
+ if not tcs and "DONE" in content.upper():
+ done_said = True
+ turns_log[-1]["done"] = True
+ break
+ if not tcs:
+ no_tool_streak += 1
+ if no_tool_streak >= 2:
+ print(" -> giving up after 2 no-tool turns")
+ break
+ turns_log[-1]["no_tool_streak"] = no_tool_streak
+ messages.append({"role": "assistant", "content": content})
+ messages.append({"role": "user", "content":
+ "You must either call a tool to keep working, or say DONE if "
+ "you believe the app is complete and tests pass."})
+ continue
+ no_tool_streak = 0
+ a_msg = {"role": "assistant", "content": content}
+ if tcs: a_msg["tool_calls"] = tcs
+ messages.append(a_msg)
+ for tc in tcs:
+ fn = tc["function"]
+ name = fn["name"]
+ try:
+ args = json.loads(fn["arguments"] or "{}")
+ except json.JSONDecodeError as e:
+ args = {}
+ result = (f"ERROR: your tool_call arguments were not valid "
+ f"JSON ({e}). Please re-emit the {name} tool call "
+ f"with valid JSON arguments.")
+ else:
+ impl = tools.TOOL_IMPLS.get(name)
+ if impl is None:
+ result = f"ERROR: unknown tool {name!r}"
+ else:
+ try:
+ result = impl(args)
+ except Exception as e:
+ result = f"ERROR: {name} raised {e!r}"
+ if len(result) > 6000:
+ result = result[:6000] + f"\n...[truncated, {len(result)} total chars]"
+ messages.append({"role": "tool", "tool_call_id": tc.get("id", ""), "content": result})
+ tlog.setdefault("tool_results", []).append({
+ "tool": name, "result_len": len(result),
+ "result_head": result[:120].replace("\n", " ")})
+ wall = time.time() - t0
+ g = gates(WORK)
+ fr_counter = {}
+ for f in finish_reasons:
+ if f: fr_counter[f] = fr_counter.get(f, 0) + 1
+ rec = {
+ "model": client.MODEL, "max_tokens_per_turn": MAX_TOKENS,
+ "max_turns": MAX_TURNS, "turns_used": turns, "done_said": done_said,
+ "wall_s": round(wall, 1), "peak_prompt_tokens": peak_prompt,
+ "total_completion_tokens": total_completion,
+ "total_reasoning_tokens": total_reasoning,
+ "finish_reasons": fr_counter, "checkpoints": g,
+ "all_pass": all(g.values()), "turns_log": turns_log,
+ }
+ _report(rec)
+
+def _report(rec):
+ print("\n===================== AGENTIC BUILD RESULT =====================")
+ print(f"model : {rec['model']}")
+ print(f"turns used : {rec['turns_used']}/{rec['max_turns']} (done_said={rec['done_said']})")
+ print(f"wall clock : {rec['wall_s']}s ({rec['wall_s']/60:.1f} min)")
+ print(f"peak prompt tok : {rec['peak_prompt_tokens']}")
+ print(f"total completion : {rec['total_completion_tokens']}")
+ print(f"total reasoning : {rec['total_reasoning_tokens']}")
+ print(f"finish reasons : {rec['finish_reasons']}")
+ print()
+ print(f"--- checkpoints ---")
+ for k, v in rec["checkpoints"].items():
+ print(f" {k:10s}: {'PASS' if v else 'fail'}")
+ print(f" => {'ALL PASS — real product works' if rec['all_pass'] else 'INCOMPLETE'}")
+ write_json(rec, rec["model"], prefix="agentic-build")
+ print()
+
+
+
+def main():
+ mode = sys.argv[1] if len(sys.argv) > 1 else 'validate'
+ if mode == 'validate':
+ sys.exit(0 if validate() else 1)
+ elif mode == 'run':
+ run()
+ else:
+ print(f'usage: python bench.py agentic-build [validate|run]')
+ sys.exit(1)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/bench/tasks/agentic_fix.py b/bench/tasks/agentic_fix.py
new file mode 100644
index 0000000..0583922
--- /dev/null
+++ b/bench/tasks/agentic_fix.py
@@ -0,0 +1,488 @@
+"""Agentic fix instrument — multi-turn tool loop over a tiny in-memory codebase.
+
+The model gets a failing Go test and a read-only toolbox (list_files / read_file
+/ search / submit_fix), must explore the code, reason ACROSS files, and submit
+a fix. Scored by compile+run against the hidden test with the real Go toolchain.
+
+Why this shape:
+ - Long context: the fix requires tracing Set() in store.go against isExpired()
+ in ttl.go; a model that only reads one file cannot find it.
+ - Agentic: real tool-call loop (assistant tool_calls -> tool results -> repeat),
+ the exact pattern an agent panel drives — not a one-shot prompt.
+ - Coherence-under-turns: MAX_TURNS caps runaway loops; we record turns used
+ and whether the agent converged, which is where reasoning-loop behavior shows.
+
+Usage (via the bench.py CLI):
+ python bench.py agentic-fix validate
+ python bench.py agentic-fix run
+
+Env (via .env): ENDPOINT, MODEL, API_KEY, MAX_TOKENS, MAX_TURNS, HTTP_TIMEOUT.
+"""
+import json
+import os
+import re
+import sys
+import time
+import shutil
+import subprocess
+
+from bench.core import client
+from bench.core.reporting import write_json
+
+HERE = os.path.dirname(os.path.abspath(__file__))
+WORK = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "work", "agentic_fix")
+MAX_TOKENS = int(os.environ.get("MAX_TOKENS", "4000"))
+MAX_TURNS = int(os.environ.get("MAX_TURNS", "24"))
+RUN_TIMEOUT = 120
+
+# ----------------------------------------------------------------- the codebase
+# A small in-memory TTL cache split across files. The bug is cross-file:
+# store.go Set() stores expireAt = now instead of now.Add(ttl), so every
+# entry is born expired. Finding it requires reading BOTH store.go (writes
+# expireAt) and ttl.go (isExpired compares expireAt to now).
+PROJECT = {
+"go.mod": "module kvstore\n\ngo 1.26\n",
+"doc.go": '''// Package kvstore is a small concurrency-safe in-memory key/value cache with
+// per-entry TTL expiry and a background sweeper that evicts expired entries.
+package kvstore
+''',
+"store.go": '''package kvstore
+
+import (
+ "sync"
+ "time"
+)
+
+type entry struct {
+ value any
+ expireAt time.Time
+ created time.Time
+}
+
+type Store struct {
+ mu sync.RWMutex
+ items map[string]*entry
+ stats *Stats
+ opts options
+ stopCh chan struct{}
+ closed bool
+}
+
+func New(optFns ...Option) *Store {
+ o := defaultOptions()
+ for _, fn := range optFns {
+ fn(&o)
+ }
+ s := &Store{
+ items: make(map[string]*entry),
+ stats: &Stats{},
+ opts: o,
+ stopCh: make(chan struct{}),
+ }
+ go s.sweepLoop()
+ return s
+}
+
+func (s *Store) Set(key string, value any, ttl time.Duration) {
+ s.mu.Lock()
+ defer s.mu.Unlock()
+ now := time.Now()
+ e := &entry{value: value, created: now}
+ if ttl > 0 {
+ e.expireAt = now // BUG: forgets to add ttl -> entry is born expired.
+ }
+ s.items[key] = e
+}
+
+func (s *Store) Get(key string) (any, bool) {
+ s.mu.Lock()
+ defer s.mu.Unlock()
+ e, ok := s.items[key]
+ if !ok {
+ s.stats.recordMiss()
+ return nil, false
+ }
+ if isExpired(e, time.Now()) {
+ delete(s.items, key)
+ s.stats.recordMiss()
+ return nil, false
+ }
+ s.stats.recordHit()
+ return e.value, true
+}
+
+func (s *Store) Delete(key string) bool {
+ s.mu.Lock()
+ defer s.mu.Unlock()
+ _, ok := s.items[key]
+ delete(s.items, key)
+ return ok
+}
+
+func (s *Store) Len() int {
+ s.mu.RLock()
+ defer s.mu.RUnlock()
+ return len(s.items)
+}
+
+func (s *Store) Keys() []string {
+ s.mu.RLock()
+ defer s.mu.RUnlock()
+ ks := make([]string, 0, len(s.items))
+ for k := range s.items {
+ ks = append(ks, k)
+ }
+ return ks
+}
+
+func (s *Store) Close() {
+ s.mu.Lock()
+ defer s.mu.Unlock()
+ if !s.closed {
+ close(s.stopCh)
+ s.closed = true
+ }
+}
+''',
+"ttl.go": '''package kvstore
+
+import "time"
+
+func isExpired(e *entry, now time.Time) bool {
+ if e.expireAt.IsZero() {
+ return false
+ }
+ return now.After(e.expireAt)
+}
+
+func remaining(e *entry, now time.Time) time.Duration {
+ if e.expireAt.IsZero() {
+ return 0
+ }
+ d := e.expireAt.Sub(now)
+ if d < 0 {
+ return 0
+ }
+ return d
+}
+''',
+"sweeper.go": '''package kvstore
+
+import "time"
+
+func (s *Store) sweepLoop() {
+ t := time.NewTicker(s.opts.sweepInterval)
+ defer t.Stop()
+ for {
+ select {
+ case <-s.stopCh:
+ return
+ case <-t.C:
+ s.sweepOnce()
+ }
+ }
+}
+
+func (s *Store) sweepOnce() {
+ now := time.Now()
+ s.mu.Lock()
+ defer s.mu.Unlock()
+ for k, e := range s.items {
+ if isExpired(e, now) {
+ delete(s.items, k)
+ s.stats.recordEviction()
+ }
+ }
+}
+''',
+"stats.go": '''package kvstore
+
+import "sync/atomic"
+
+type Stats struct {
+ hits atomic.Int64
+ misses atomic.Int64
+ evictions atomic.Int64
+}
+
+func (s *Stats) recordHit() { s.hits.Add(1) }
+func (s *Stats) recordMiss() { s.misses.Add(1) }
+func (s *Stats) recordEviction() { s.evictions.Add(1) }
+
+func (s *Stats) Snapshot() (hits, misses, evictions int64) {
+ return s.hits.Load(), s.misses.Load(), s.evictions.Load()
+}
+
+func (s *Store) Stats() *Stats { return s.stats }
+''',
+"options.go": '''package kvstore
+
+import "time"
+
+type options struct {
+ sweepInterval time.Duration
+}
+
+type Option func(*options)
+
+func defaultOptions() options {
+ return options{sweepInterval: 30 * time.Second}
+}
+
+func WithSweepInterval(d time.Duration) Option {
+ return func(o *options) {
+ if d > 0 {
+ o.sweepInterval = d
+ }
+ }
+}
+''',
+"store_test.go": '''package kvstore
+
+import (
+ "testing"
+ "time"
+)
+
+func TestSetThenGetWithinTTL(t *testing.T) {
+ s := New(WithSweepInterval(time.Hour))
+ defer s.Close()
+ s.Set("k", "v", time.Hour)
+ got, ok := s.Get("k")
+ if !ok {
+ t.Fatalf("Get right after Set: want hit, got miss")
+ }
+ if got != "v" {
+ t.Fatalf("Get value = %v, want v", got)
+ }
+}
+
+func TestExpiresAfterTTL(t *testing.T) {
+ s := New(WithSweepInterval(time.Hour))
+ defer s.Close()
+ s.Set("k", "v", 10*time.Millisecond)
+ time.Sleep(25 * time.Millisecond)
+ if _, ok := s.Get("k"); ok {
+ t.Fatalf("Get after TTL: want miss, got hit")
+ }
+}
+
+func TestZeroTTLNeverExpires(t *testing.T) {
+ s := New(WithSweepInterval(time.Hour))
+ defer s.Close()
+ s.Set("k", "v", 0)
+ time.Sleep(15 * time.Millisecond)
+ if _, ok := s.Get("k"); !ok {
+ t.Fatalf("zero-ttl entry: want hit, got miss")
+ }
+}
+''',
+}
+
+FIX = {
+"store.go": PROJECT["store.go"].replace(
+ "\t\te.expireAt = now // BUG: forgets to add ttl -> entry is born expired.",
+ "\t\te.expireAt = now.Add(ttl)",
+),
+}
+BUG_FILES = ["store.go"]
+
+# ----------------------------------------------------------------- go toolchain
+def _materialize(files):
+ shutil.rmtree(WORK, ignore_errors=True)
+ os.makedirs(WORK, exist_ok=True)
+ for path, content in files.items():
+ fp = os.path.join(WORK, path)
+ os.makedirs(os.path.dirname(fp), exist_ok=True)
+ with open(fp, "w") as f:
+ f.write(content)
+ return WORK
+
+def _go_test(d):
+ try:
+ p = subprocess.run(["go", "test", "./..."], cwd=d,
+ capture_output=True, text=True, timeout=RUN_TIMEOUT)
+ except subprocess.TimeoutExpired:
+ return False, "TIMEOUT"
+ return p.returncode == 0, (p.stdout + "\n" + p.stderr).strip()
+
+def _apply(base, patch):
+ merged = dict(base)
+ merged.update(patch)
+ return merged
+
+# ----------------------------------------------------------------- tools
+TOOLS = [
+ {"type": "function", "function": {
+ "name": "list_files", "description": "List all file paths in the project.",
+ "parameters": {"type": "object", "properties": {}}}},
+ {"type": "function", "function": {
+ "name": "read_file", "description": "Return the full contents of one file.",
+ "parameters": {"type": "object", "properties": {
+ "path": {"type": "string", "description": "Project-relative file path"}},
+ "required": ["path"]}}},
+ {"type": "function", "function": {
+ "name": "search", "description": "Regex-search the whole project; returns matching path:line: text.",
+ "parameters": {"type": "object", "properties": {"pattern": {"type": "string"}},
+ "required": ["pattern"]}},
+ },
+ {"type": "function", "function": {
+ "name": "submit_fix", "description": "Submit the complete corrected contents of ONE file to fix the failing test. Call this exactly once when you have the fix.",
+ "parameters": {"type": "object", "properties": {
+ "path": {"type": "string"},
+ "content": {"type": "string", "description": "Full new file contents"}},
+ "required": ["path", "content"]}},
+ },
+]
+
+def _tool_result(name, args):
+ if name == "list_files":
+ return "\n".join(sorted(PROJECT.keys()))
+ if name == "read_file":
+ path = args.get("path", "")
+ return PROJECT.get(path, f"ERROR: no such file {path!r}. Use list_files.")
+ if name == "search":
+ pat = args.get("pattern", "")
+ try:
+ rx = re.compile(pat)
+ except re.error as e:
+ return f"ERROR: bad regex: {e}"
+ hits = []
+ for path, content in sorted(PROJECT.items()):
+ for i, line in enumerate(content.splitlines(), 1):
+ if rx.search(line):
+ hits.append(f"{path}:{i}: {line.strip()}")
+ return "\n".join(hits) if hits else "(no matches)"
+ return f"ERROR: unknown tool {name}"
+
+SYSTEM = (
+ "You are a coding agent working in a Go project. A test is failing. Use the "
+ "tools to explore the codebase, find the root cause (it may span multiple "
+ "files), and fix it. When you are confident, call submit_fix with the COMPLETE "
+ "corrected contents of the single file that needs changing. Keep every other "
+ "file untouched and preserve all names and signatures. Do not submit until you "
+ "have read enough to be sure."
+)
+
+def validate():
+ print("=== VALIDATION: buggy fails, reference fix passes ===\n")
+ d = _materialize(PROJECT)
+ bug_pass, bug_out = _go_test(d)
+ d = _materialize(_apply(PROJECT, FIX))
+ fix_pass, fix_out = _go_test(d)
+ ok = (not bug_pass) and fix_pass
+ print(f"buggy_fails={not bug_pass} fix_passes={fix_pass} -> {'OK' if ok else 'BAD'}")
+ if bug_pass:
+ print(" !! buggy project unexpectedly PASSED")
+ if not fix_pass:
+ print(" !! reference fix FAILED:\n" + fix_out[-1200:])
+ return ok
+
+def run():
+ d = _materialize(PROJECT)
+ _, fail_out = _go_test(d)
+ messages = [
+ {"role": "system", "content": SYSTEM},
+ {"role": "user", "content":
+ "`go test ./...` fails in this project:\n\n```\n" + fail_out[-1500:] +
+ "\n```\n\nInvestigate with the tools and submit a fix."},
+ ]
+ reads, searches, submitted = [], 0, None
+ t0 = time.time()
+ peak_prompt_tokens = 0
+ turns = 0
+ for turn in range(1, MAX_TURNS + 1):
+ turns = turn
+ try:
+ resp, dt = client.call_model(messages, tools=TOOLS, max_tokens=MAX_TOKENS)
+ except Exception as e:
+ print(f" turn {turn}: REQUEST ERROR {e!r}")
+ submitted = ("__error__", "")
+ break
+ u = client.usage(resp)
+ peak_prompt_tokens = max(peak_prompt_tokens, u["prompt_tokens"])
+ msg = resp["choices"][0]["message"]
+ tcs = msg.get("tool_calls") or []
+ if not tcs:
+ content = (msg.get("content") or "")[:200]
+ print(f" turn {turn}: no tool call (content={content!r})")
+ messages.append({"role": "assistant", "content": msg.get("content") or ""})
+ messages.append({"role": "user", "content":
+ "You must either call a tool to keep investigating or call "
+ "submit_fix. Do not answer in prose."})
+ continue
+ messages.append({"role": "assistant", "content": msg.get("content") or "",
+ "tool_calls": tcs})
+ done = False
+ for tc in tcs:
+ fn = tc["function"]
+ name = fn["name"]
+ try:
+ args = json.loads(fn["arguments"] or "{}")
+ except Exception:
+ args = {}
+ if name == "submit_fix":
+ submitted = (args.get("path", ""), args.get("content", ""))
+ messages.append({"role": "tool", "tool_call_id": tc.get("id", ""),
+ "content": "fix received"})
+ done = True
+ print(f" turn {turn}: submit_fix({args.get('path','?')}) "
+ f"[{len(reads)} reads, {searches} searches]")
+ break
+ if name == "read_file":
+ reads.append(args.get("path", ""))
+ elif name == "search":
+ searches += 1
+ result = _tool_result(name, args)
+ messages.append({"role": "tool", "tool_call_id": tc.get("id", ""),
+ "content": result})
+ if done:
+ break
+ wall = time.time() - t0
+ rec = {"model": client.MODEL, "turns": turns, "reads": reads, "searches": searches,
+ "peak_prompt_tokens": peak_prompt_tokens, "wall_s": round(wall, 1)}
+ if not submitted or submitted[0] == "__error__":
+ rec.update({"passed": False, "reason": "no fix submitted (turn cap or error)"})
+ else:
+ path, content = submitted
+ rec["submitted_path"] = path
+ if path not in BUG_FILES:
+ rec["note"] = f"submitted {path}, expected one of {BUG_FILES}"
+ merged = _apply(PROJECT, {path: content})
+ d = _materialize(merged)
+ passed, out = _go_test(d)
+ rec.update({"passed": bool(passed), "test_output": out[-800:]})
+ _report(rec)
+ return rec
+
+def _report(rec):
+ print("\n===================== AGENTIC FIX RESULT =====================")
+ print(f"model : {rec['model']}")
+ print(f"passed : {'YES' if rec.get('passed') else 'no'}")
+ print(f"turns used : {rec['turns']}/{MAX_TURNS}")
+ print(f"files read : {len(rec['reads'])} {rec['reads']}")
+ print(f"searches : {rec['searches']}")
+ print(f"peak prompt tok : {rec['peak_prompt_tokens']}")
+ print(f"wall clock : {rec['wall_s']}s")
+ if rec.get("note"):
+ print(f"note : {rec['note']}")
+ if not rec.get("passed") and rec.get("test_output"):
+ print("test output:\n" + rec["test_output"])
+ write_json(rec, rec["model"], prefix="agentic-fix")
+ print()
+
+
+
+def main():
+ mode = sys.argv[1] if len(sys.argv) > 1 else 'validate'
+ if mode == 'validate':
+ sys.exit(0 if validate() else 1)
+ elif mode == 'run':
+ run()
+ else:
+ print(f'usage: python bench.py agentic-fix [validate|run]')
+ sys.exit(1)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/bench/tasks/agentic_multitask.py b/bench/tasks/agentic_multitask.py
new file mode 100644
index 0000000..9f61b88
--- /dev/null
+++ b/bench/tasks/agentic_multitask.py
@@ -0,0 +1,269 @@
+"""Agentic multi-task instrument — the v2-hard gauntlet.
+
+Runs multiple build-a-real-app tasks (each in a different language/framework)
+against one model, sequentially. Each task is a self-contained module under
+`bench/tasks/multitask/` exposing TASK_NAME, PLAN, STARTER, REFERENCE,
+validate(), gates(), TOOLS, TOOL_IMPLS. The harness materializes the
+starter, runs a multi-turn tool loop, then scores via the task's gates().
+
+Shipped tasks (the gauntlet):
+ go_htmx — Go + HTMX partials + embedded SQLite + recursive CTE
+ rust_cli — Rust + reqwest + quick-xml + async + Digest auth
+ ts_next — Next.js + Prisma + Zod validation + API route
+
+These are realistic small-app shapes that test the real agent workload:
+multi-file creation, dependency management, iterative build/test cycles,
+and cross-file reasoning. Add your own by dropping a task module into
+`bench/tasks/multitask/` — it auto-discovers.
+
+Usage (via the bench.py CLI):
+ python bench.py agentic-multitask validate go_htmx
+ python bench.py agentic-multitask run go_htmx
+ python bench.py agentic-multitask cohort go_htmx rust_cli ts_next
+
+Env (via .env): ENDPOINT, MODEL, API_KEY, MAX_TOKENS, MAX_TURNS, TEMPERATURE.
+"""
+import importlib.util
+import json
+import os
+import re
+import sys
+import time
+
+from bench.core import client
+from bench.core import tools
+from bench.core.reporting import write_json, safe_label
+
+HERE = os.path.dirname(os.path.abspath(__file__))
+MULTITASK_DIR = os.path.join(HERE, "multitask")
+WORK_BASE = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "work")
+
+MAX_TOKENS = int(os.environ.get("MAX_TOKENS", "16000"))
+MAX_TURNS = int(os.environ.get("MAX_TURNS", "40"))
+TEMPERATURE = float(os.environ.get("TEMPERATURE", "0"))
+
+
+def load_task(name):
+ path = os.path.join(MULTITASK_DIR, f"{name}.py")
+ if not os.path.isfile(path):
+ print(f"ERROR: no such task module: {path}", file=sys.stderr)
+ sys.exit(2)
+ spec = importlib.util.spec_from_file_location(f"task_{name}", path)
+ m = importlib.util.module_from_spec(spec)
+ spec.loader.exec_module(m)
+ return {
+ "TASK_NAME": m.TASK_NAME,
+ "PLAN": m.PLAN,
+ "STARTER": m.STARTER,
+ "REFERENCE": m.REFERENCE,
+ "validate": m.validate,
+ "gates": m.gates,
+ "TOOLS": getattr(m, "TOOLS", tools.TOOLS),
+ "TOOL_IMPLS": getattr(m, "TOOL_IMPLS", tools.TOOL_IMPLS),
+ }
+
+
+def available_tasks():
+ if not os.path.isdir(MULTITASK_DIR):
+ return []
+ return sorted(f[:-3] for f in os.listdir(MULTITASK_DIR)
+ if f.endswith(".py") and not f.startswith("_"))
+
+
+def _materialize(files, dest):
+ import shutil
+ shutil.rmtree(dest, ignore_errors=True)
+ os.makedirs(dest, exist_ok=True)
+ for path, content in files.items():
+ fp = os.path.join(dest, path)
+ os.makedirs(os.path.dirname(fp), exist_ok=True)
+ with open(fp, "w") as f:
+ f.write(content)
+ return dest
+
+
+def run_solo(task, work_dir):
+ tools.set_work_dir(work_dir)
+ task_tools = task.get("TOOLS", tools.TOOLS)
+ task_tool_impls = task.get("TOOL_IMPLS", tools.TOOL_IMPLS)
+ _materialize(task["STARTER"], dest=work_dir)
+ messages = [
+ {"role": "system", "content":
+ "You are a coding agent working in a project work dir. You are given "
+ "a task plan below. Execute it step by step using the tools. You can "
+ "write files, edit files, run builds and tests, and run shell commands "
+ "in the work dir. Keep going until the build is green, the tests pass, "
+ "and the app/binary works. Use the tools to verify your own progress — "
+ "don't guess; compile and test.\n\n"
+ "When you are confident the task is complete and tests pass, say "
+ "`DONE` in your final message (no tool call) and stop.\n\n"
+ "TASK PLAN:\n" + task["PLAN"]},
+ {"role": "user", "content":
+ "The work dir has the starter scaffold. Execute the task plan. Begin "
+ "by adding any dependencies and writing the files. Use the tools to "
+ "build and test as you go."},
+ ]
+ turns_log = []
+ peak_prompt = total_completion = total_reasoning = 0
+ finish_reasons = []
+ t0 = time.time()
+ turns = 0
+ done_said = False
+ no_tool_streak = 0
+ for turn in range(1, MAX_TURNS + 1):
+ turns = turn
+ body = {"model": client.MODEL, "messages": messages, "max_tokens": MAX_TOKENS,
+ "temperature": TEMPERATURE, "tools": task_tools}
+ try:
+ resp, dt = client.call_model(messages, tools=task_tools, max_tokens=MAX_TOKENS,
+ temperature=TEMPERATURE)
+ except Exception as e:
+ print(f" turn {turn}: REQUEST ERROR {e!r}")
+ turns_log.append({"turn": turn, "error": repr(e)})
+ break
+ u = client.usage(resp)
+ peak_prompt = max(peak_prompt, u["prompt_tokens"])
+ total_completion += u["completion_tokens"]
+ total_reasoning += u["reasoning_tokens"]
+ choice = resp["choices"][0]
+ fr = choice.get("finish_reason")
+ finish_reasons.append(fr)
+ msg = choice["message"]
+ tcs = msg.get("tool_calls") or []
+ content = msg.get("content") or ""
+ tlog = {"turn": turn, "dt": round(dt, 1), "prompt_tok": u["prompt_tokens"],
+ "completion_tok": u["completion_tokens"], "reasoning_tok": u["reasoning_tokens"],
+ "finish_reason": fr, "n_tool_calls": len(tcs)}
+ turns_log.append(tlog)
+ tc_summary = ",".join((tc["function"]["name"] for tc in tcs)) or "(no tool call)"
+ print(f" turn {turn:2d}: {dt:5.1f}s ptok={u['prompt_tokens']:6d} ctok={u['completion_tokens']:5d} "
+ f"rtok={u['reasoning_tokens']:5d} fr={fr:8s} tools=[{tc_summary}]")
+ if not tcs and "DONE" in content.upper():
+ done_said = True
+ tlog["done"] = True
+ break
+ if not tcs:
+ no_tool_streak += 1
+ if no_tool_streak >= 2:
+ print(" -> giving up after 2 no-tool turns")
+ break
+ messages.append({"role": "assistant", "content": content})
+ messages.append({"role": "user", "content":
+ "You must either call a tool to keep working, or say DONE if "
+ "you believe the task is complete and tests pass."})
+ continue
+ no_tool_streak = 0
+ a_msg = {"role": "assistant", "content": content}
+ if tcs: a_msg["tool_calls"] = tcs
+ messages.append(a_msg)
+ for tc in tcs:
+ fn = tc["function"]
+ name = fn["name"]
+ try:
+ args = json.loads(fn["arguments"] or "{}")
+ except json.JSONDecodeError as e:
+ args = {}
+ result = f"ERROR: invalid JSON args ({e}); re-emit the {name} call."
+ else:
+ impl = task_tool_impls.get(name)
+ result = impl(args) if impl else f"ERROR: unknown tool {name!r}"
+ if len(result) > 6000:
+ result = result[:6000] + f"\n...[truncated, {len(result)} total]"
+ messages.append({"role": "tool", "tool_call_id": tc.get("id", ""), "content": result})
+ tlog.setdefault("tool_results", []).append({
+ "tool": name, "result_head": result[:120].replace("\n", " ")})
+ wall = time.time() - t0
+ gate_results = task["gates"](work_dir)
+ fr_counter = {}
+ for f in finish_reasons:
+ if f: fr_counter[f] = fr_counter.get(f, 0) + 1
+ rec = {
+ "architecture": "solo",
+ "task": task["TASK_NAME"],
+ "model": client.MODEL,
+ "max_tokens_per_turn": MAX_TOKENS,
+ "max_turns": MAX_TURNS,
+ "turns_used": turns,
+ "done_said": done_said,
+ "wall_s": round(wall, 1),
+ "peak_prompt_tokens": peak_prompt,
+ "total_completion_tokens": total_completion,
+ "total_reasoning_tokens": total_reasoning,
+ "finish_reasons": fr_counter,
+ "checkpoints": gate_results,
+ "all_pass": all(gate_results.values()),
+ "turns_log": turns_log,
+ }
+ return rec
+
+
+def _report(rec):
+ tag = safe_label(f"{rec['task']}__{rec['model']}")
+ print("\n===================== V2-HARD RESULT =====================")
+ print(f"task : {rec['task']}")
+ print(f"model : {rec['model']}")
+ print(f"turns used : {rec['turns_used']}/{rec['max_turns']}")
+ print(f"done_said : {rec['done_said']}")
+ print(f"wall clock : {rec['wall_s']}s ({rec['wall_s']/60:.1f} min)")
+ print(f"peak prompt tok : {rec['peak_prompt_tokens']}")
+ print(f"total completion : {rec['total_completion_tokens']}")
+ print(f"total reasoning : {rec['total_reasoning_tokens']}")
+ print(f"finish reasons : {rec['finish_reasons']}")
+ print()
+ print("--- gates ---")
+ for k, v in rec["checkpoints"].items():
+ print(f" {k:16s}: {'PASS' if v else 'fail'}")
+ print(f" => {'ALL PASS' if rec['all_pass'] else 'INCOMPLETE'}")
+ write_json(rec, tag, prefix="v2hard")
+ print()
+
+
+def validate(task_name):
+ task = load_task(task_name)
+ work_dir = os.path.join(WORK_BASE, f"multitask_validate_{task_name}")
+ print(f"=== VALIDATION: {task_name} reference app ===")
+ task["validate"](work_dir)
+
+
+def run(task_name):
+ task = load_task(task_name)
+ work_dir = os.path.join(WORK_BASE, f"multitask_solo_{task_name}")
+ rec = run_solo(task, work_dir)
+ _report(rec)
+
+
+def cohort(*task_names):
+ results = []
+ for name in task_names:
+ print(f"\n{'='*60}\n=== COHORT: {name} ===\n{'='*60}")
+ task = load_task(name)
+ work_dir = os.path.join(WORK_BASE, f"multitask_cohort_{name}")
+ rec = run_solo(task, work_dir)
+ _report(rec)
+ results.append(rec)
+ print(f"\n{'='*60}\n=== COHORT SUMMARY ===\n{'='*60}")
+ print(f"{'task':16s} {'gates':>6s} {'wall':>8s} {'turns':>6s} {'comp tok':>9s} {'rsn tok':>9s}")
+ for r in results:
+ g = 'PASS' if r['all_pass'] else 'fail'
+ print(f"{r['task']:16s} {g:>6s} {r['wall_s']/60:>7.1f}m {r['turns_used']:>6} {r['total_completion_tokens']:>9,} {r['total_reasoning_tokens']:>9,}")
+
+
+def main():
+ if len(sys.argv) < 2:
+ print(f"usage: python bench.py agentic-multitask [validate|run|cohort] <task>...")
+ print(f"available tasks: {', '.join(available_tasks())}")
+ sys.exit(1)
+ cmd = sys.argv[1]
+ if cmd == "validate":
+ validate(sys.argv[2])
+ elif cmd == "run":
+ run(sys.argv[2])
+ elif cmd == "cohort":
+ cohort(*sys.argv[2:])
+ else:
+ print(f"unknown command: {cmd}")
+ sys.exit(1)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/bench/tasks/coding_suite.py b/bench/tasks/coding_suite.py
new file mode 100644
index 0000000..812c02c
--- /dev/null
+++ b/bench/tasks/coding_suite.py
@@ -0,0 +1,421 @@
+"""Coding-suite instrument — single-turn bug-fix + tool-calling tasks.
+
+Two tiers of difficulty. Each bug-fix answer is compiled and run against
+hidden tests with the real toolchain (Go/Rust/Swift/TS/Python). Tool-calling
+answers are parse-validated. The suite is pre-validated (buggy provably
+fails, reference fix provably passes) by the `validate` command.
+
+Usage (via the bench.py CLI):
+ python bench.py coding-suite validate
+ python bench.py coding-suite run
+ python bench.py coding-suite report
+
+Env (via .env): ENDPOINT, MODEL, API_KEY, MAX_TOKENS, TEMPERATURE.
+"""
+import json
+import os
+import re
+import sys
+
+from bench.core import client
+from bench.core.runners import RUNNERS, extract_code
+from bench.core.reporting import jsonl_path, append_jsonl, write_json, coding_suite_summary
+
+HERE = os.path.dirname(os.path.abspath(__file__))
+
+# ---------------------------------------------------------------- task registry
+# Tier 1 — straightforward bug-fixes
+BUGFIX_T1 = [
+ {
+ "id": "go-average", "lang": "go", "cat": "bugfix",
+ "symptom": "It should return the arithmetic mean as a float, but returns truncated values (e.g. 1 instead of 1.5 for [1,2]).",
+ "buggy": "package solution\n\nfunc Average(xs []int) float64 {\n\tsum := 0\n\tfor _, x := range xs {\n\t\tsum += x\n\t}\n\treturn float64(sum / len(xs))\n}\n",
+ "fix": "package solution\n\nfunc Average(xs []int) float64 {\n\tsum := 0\n\tfor _, x := range xs {\n\t\tsum += x\n\t}\n\treturn float64(sum) / float64(len(xs))\n}\n",
+ "test": 'package solution\n\nimport "testing"\n\nfunc almost(a, b float64) bool { d := a - b; if d < 0 { d = -d }; return d < 1e-9 }\n\nfunc TestAverage(t *testing.T) {\n\tif !almost(Average([]int{1, 2}), 1.5) { t.Fatalf("[1,2]=%v want 1.5", Average([]int{1, 2})) }\n\tif !almost(Average([]int{2, 2, 2}), 2.0) { t.Fatalf("[2,2,2]=%v want 2", Average([]int{2, 2, 2})) }\n\tif !almost(Average([]int{1, 2, 3, 4}), 2.5) { t.Fatalf("[1,2,3,4]=%v want 2.5", Average([]int{1, 2, 3, 4})) }\n}\n',
+ },
+ {
+ "id": "go-nilmap", "lang": "go", "cat": "bugfix",
+ "symptom": "It panics at runtime ('assignment to entry in nil map') instead of grouping the integers by parity.",
+ "buggy": 'package solution\n\nfunc GroupByParity(xs []int) map[string][]int {\n\tvar m map[string][]int\n\tfor _, x := range xs {\n\t\tif x%2 == 0 {\n\t\t\tm["even"] = append(m["even"], x)\n\t\t} else {\n\t\t\tm["odd"] = append(m["odd"], x)\n\t\t}\n\t}\n\treturn m\n}\n',
+ "fix": 'package solution\n\nfunc GroupByParity(xs []int) map[string][]int {\n\tm := make(map[string][]int)\n\tfor _, x := range xs {\n\t\tif x%2 == 0 {\n\t\t\tm["even"] = append(m["even"], x)\n\t\t} else {\n\t\t\tm["odd"] = append(m["odd"], x)\n\t\t}\n\t}\n\treturn m\n}\n',
+ "test": 'package solution\n\nimport (\n\t"reflect"\n\t"testing"\n)\n\nfunc TestGroup(t *testing.T) {\n\tgot := GroupByParity([]int{1, 2, 3, 4})\n\twant := map[string][]int{"odd": {1, 3}, "even": {2, 4}}\n\tif !reflect.DeepEqual(got, want) { t.Fatalf("got %v want %v", got, want) }\n}\n',
+ },
+ {
+ "id": "rust-sumeven", "lang": "rust", "cat": "bugfix",
+ "symptom": "sum_even is supposed to sum the EVEN numbers, but it currently sums the odd ones (e.g. returns 4 instead of 6 for [1,2,3,4]).",
+ "buggy": "pub fn sum_even(v: &[i32]) -> i32 {\n v.iter().filter(|&&x| x % 2 == 1).sum()\n}\n",
+ "fix": "pub fn sum_even(v: &[i32]) -> i32 {\n v.iter().filter(|&&x| x % 2 == 0).sum()\n}\n",
+ "test": "#[cfg(test)]\nmod hidden {\n use super::*;\n #[test]\n fn t() {\n assert_eq!(sum_even(&[1, 2, 3, 4]), 6);\n assert_eq!(sum_even(&[2, 4, 6]), 12);\n assert_eq!(sum_even(&[1, 3, 5]), 0);\n }\n}\n",
+ },
+ {
+ "id": "rust-safediv", "lang": "rust", "cat": "bugfix",
+ "symptom": "safe_div should return none when dividing by zero, but it panics (attempt to divide by zero) instead.",
+ "buggy": "pub fn safe_div(a: i32, b: i32) -> Option<i32> {\n Some(a / b)\n}\n",
+ "fix": "pub fn safe_div(a: i32, b: i32) -> Option<i32> {\n if b == 0 { None } else { Some(a / b) }\n}\n",
+ "test": "#[cfg(test)]\nmod hidden {\n use super::*;\n #[test]\n fn t() {\n assert_eq!(safe_div(10, 2), Some(5));\n assert_eq!(safe_div(7, 0), None);\n assert_eq!(safe_div(-6, 3), Some(-2));\n }\n}\n",
+ },
+ {
+ "id": "swift-parseage", "lang": "swift", "cat": "bugfix",
+ "symptom": "parseAge should return nil for non-numeric input, but it crashes (force-unwrap of nil) on input like \"abc\".",
+ "buggy": "func parseAge(_ s: String) -> Int? {\n return Int(s)!\n}\n",
+ "fix": "func parseAge(_ s: String) -> Int? {\n return Int(s)\n}\n",
+ "test": 'import Foundation\n\nfunc check(_ cond: Bool, _ msg: String) {\n if !cond { FileHandle.standardError.write(("FAIL: " + msg + "\\n").data(using: .utf8)!); exit(1) }\n}\n\ncheck(parseAge("34") == 34, "parseAge(34)")\ncheck(parseAge("abc") == nil, "parseAge(abc) should be nil")\ncheck(parseAge("0") == 0, "parseAge(0)")\nprint("OK")\n',
+ },
+ {
+ "id": "swift-median", "lang": "swift", "cat": "bugfix",
+ "symptom": "median returns the wrong value because it indexes the array without sorting it first (e.g. median([3,1,2]) returns 1 instead of 2).",
+ "buggy": "func median(_ xs: [Double]) -> Double {\n let n = xs.count\n if n % 2 == 1 {\n return xs[n / 2]\n } else {\n return (xs[n / 2 - 1] + xs[n / 2]) / 2\n }\n}\n",
+ "fix": "func median(_ xs: [Double]) -> Double {\n let s = xs.sorted()\n let n = s.count\n if n % 2 == 1 {\n return s[n / 2]\n } else {\n return (s[n / 2 - 1] + s[n / 2]) / 2\n }\n}\n",
+ "test": 'import Foundation\n\nfunc check(_ cond: Bool, _ msg: String) {\n if !cond { FileHandle.standardError.write(("FAIL: " + msg + "\\n").data(using: .utf8)!); exit(1) }\n}\n\ncheck(median([3, 1, 2]) == 2, "median odd")\ncheck(median([4, 1, 3, 2]) == 2.5, "median even")\ncheck(median([10, 2, 8, 4]) == 6, "median even2")\nprint("OK")\n',
+ },
+ {
+ "id": "ts-sort", "lang": "ts", "cat": "bugfix",
+ "symptom": "sortNums sorts numbers lexicographically instead of numerically (e.g. [10,2,1] becomes [1,10,2] instead of [1,2,10]).",
+ "buggy": "export function sortNums(a: number[]): number[] {\n return a.sort();\n}\n",
+ "fix": "export function sortNums(a: number[]): number[] {\n return [...a].sort((x, y) => x - y);\n}\n",
+ "test": 'import { sortNums } from "./solution";\nfunction eq(a: number[], b: number[]) { return a.length === b.length && a.every((v, i) => v === b[i]); }\nlet ok = true;\nif (!eq(sortNums([10, 2, 1]), [1, 2, 10])) { console.error("FAIL sort1"); ok = false; }\nif (!eq(sortNums([3, 30, 1, 2]), [1, 2, 3, 30])) { console.error("FAIL sort2"); ok = false; }\nif (!ok) process.exit(1);\nconsole.log("OK");\n',
+ },
+ {
+ "id": "ts-reduce", "lang": "ts", "cat": "bugfix",
+ "symptom": "total throws a TypeError on an empty array because reduce has no initial value; total([]) should be 0.",
+ "buggy": "export function total(arr: number[]): number {\n return arr.reduce((a, b) => a + b);\n}\n",
+ "fix": "export function total(arr: number[]): number {\n return arr.reduce((a, b) => a + b, 0);\n}\n",
+ "test": 'import { total } from "./solution";\nlet ok = true;\nif (total([1, 2, 3]) !== 6) { console.error("FAIL t1"); ok = false; }\nif (total([]) !== 0) { console.error("FAIL empty"); ok = false; }\nif (total([5]) !== 5) { console.error("FAIL t3"); ok = false; }\nif (!ok) process.exit(1);\nconsole.log("OK");\n',
+ },
+]
+
+# Tier 2 — harder, discriminating bug-fixes (type reasoning, language traps)
+BUGFIX_T2 = [
+ {
+ "id": "go-json-tags", "lang": "go", "cat": "bugfix",
+ "symptom": "ToJSON should produce JSON with exactly the keys \"name\", \"emailAddress\", and \"age\", but the name key comes out capitalized and the age field is missing.",
+ "buggy": 'package solution\n\nimport "encoding/json"\n\ntype User struct {\n\tName string\n\tEmail string `json:"emailAddress"`\n\tage int `json:"age"`\n}\n\nfunc ToJSON(name, email string, age int) (string, error) {\n\tu := User{Name: name, Email: email, age: age}\n\tb, err := json.Marshal(u)\n\treturn string(b), err\n}\n',
+ "fix": 'package solution\n\nimport "encoding/json"\n\ntype User struct {\n\tName string `json:"name"`\n\tEmail string `json:"emailAddress"`\n\tAge int `json:"age"`\n}\n\nfunc ToJSON(name, email string, age int) (string, error) {\n\tu := User{Name: name, Email: email, Age: age}\n\tb, err := json.Marshal(u)\n\treturn string(b), err\n}\n',
+ "test": 'package solution\n\nimport (\n\t"encoding/json"\n\t"testing"\n)\n\nfunc TestToJSON(t *testing.T) {\n\ts, err := ToJSON("Ada", "ada@x.com", 36)\n\tif err != nil { t.Fatal(err) }\n\tvar m map[string]interface{}\n\tif err := json.Unmarshal([]byte(s), &m); err != nil { t.Fatal(err) }\n\tif m["name"] != "Ada" { t.Fatalf("name key wrong; got map %v", m) }\n\tif m["emailAddress"] != "ada@x.com" { t.Fatalf("email key wrong; got %v", m) }\n\tif m["age"] != float64(36) { t.Fatalf("age key wrong/missing; got %v", m) }\n}\n',
+ },
+ {
+ "id": "go-generics", "lang": "go", "cat": "bugfix",
+ "symptom": "MapSlice should return a new slice with f applied to each element, but the result has extra leading zero values (e.g. [0 0 0 2 4 6] instead of [2 4 6]).",
+ "buggy": "package solution\n\nfunc MapSlice[T any, U any](xs []T, f func(T) U) []U {\n\tresult := make([]U, len(xs))\n\tfor _, x := range xs {\n\t\tresult = append(result, f(x))\n\t}\n\treturn result\n}\n",
+ "fix": "package solution\n\nfunc MapSlice[T any, U any](xs []T, f func(T) U) []U {\n\tresult := make([]U, 0, len(xs))\n\tfor _, x := range xs {\n\t\tresult = append(result, f(x))\n\t}\n\treturn result\n}\n",
+ "test": 'package solution\n\nimport (\n\t"reflect"\n\t"testing"\n)\n\nfunc TestMapSlice(t *testing.T) {\n\tgot := MapSlice([]int{1, 2, 3}, func(x int) int { return x * 2 })\n\tif !reflect.DeepEqual(got, []int{2, 4, 6}) { t.Fatalf("got %v want [2 4 6]", got) }\n\tgs := MapSlice([]int{1, 2}, func(x int) string { return string(rune(\'a\' + x)) })\n\tif !reflect.DeepEqual(gs, []string{"b", "c"}) { t.Fatalf("got %v want [b c]", gs) }\n}\n',
+ },
+ {
+ "id": "rust-sumall", "lang": "rust", "cat": "bugfix",
+ "symptom": "sum_all does not compile because the generic type T is unconstrained. It should sum any slice of numbers (integers or floats) and return the total.",
+ "buggy": "pub fn sum_all<T>(items: &[T]) -> T {\n let mut total = 0;\n for x in items {\n total += x;\n }\n total\n}\n",
+ "fix": "pub fn sum_all<T: Copy + std::iter::Sum<T>>(items: &[T]) -> T {\n items.iter().copied().sum()\n}\n",
+ "test": "#[cfg(test)]\nmod hidden {\n use super::*;\n #[test]\n fn t() {\n assert_eq!(sum_all(&[1, 2, 3]), 6);\n assert_eq!(sum_all(&[10, 20]), 30);\n let f: f64 = sum_all(&[1.5, 2.5]);\n assert!((f - 4.0).abs() < 1e-9);\n }\n}\n",
+ },
+ {
+ "id": "rust-dedup", "lang": "rust", "cat": "bugfix",
+ "symptom": "dedup_sorted should sort the vector and remove consecutive duplicates in place, but it panics with an index-out-of-bounds error.",
+ "buggy": "pub fn dedup_sorted(v: &mut Vec<i32>) {\n v.sort();\n for i in 1..v.len() {\n if v[i] == v[i - 1] {\n v.remove(i);\n }\n }\n}\n",
+ "fix": "pub fn dedup_sorted(v: &mut Vec<i32>) {\n v.sort();\n v.dedup();\n}\n",
+ "test": "#[cfg(test)]\nmod hidden {\n use super::*;\n #[test]\n fn t() {\n let mut a = vec![3, 1, 2, 2, 3, 1];\n dedup_sorted(&mut a);\n assert_eq!(a, vec![1, 2, 3]);\n let mut b = vec![5, 5, 5];\n dedup_sorted(&mut b);\n assert_eq!(b, vec![5]);\n }\n}\n",
+ },
+ {
+ "id": "swift-generic-max", "lang": "swift", "cat": "bugfix",
+ "symptom": "maxElement does not compile because the generic type T is not constrained. It should return the largest element, or nil for an empty array.",
+ "buggy": "func maxElement<T>(_ xs: [T]) -> T? {\n guard !xs.isEmpty else { return nil }\n var m = xs[0]\n for x in xs {\n if x > m { m = x }\n }\n return m\n}\n",
+ "fix": "func maxElement<T: Comparable>(_ xs: [T]) -> T? {\n guard !xs.isEmpty else { return nil }\n var m = xs[0]\n for x in xs {\n if x > m { m = x }\n }\n return m\n}\n",
+ "test": 'import Foundation\n\nfunc check(_ cond: Bool, _ msg: String) {\n if !cond { FileHandle.standardError.write(("FAIL: " + msg + "\\n").data(using: .utf8)!); exit(1) }\n}\n\ncheck(maxElement([3, 1, 2]) == 3, "max ints")\ncheck(maxElement([Int]()) == nil, "max empty")\ncheck(maxElement(["b", "a", "c"]) == "c", "max strings")\nprint("OK")\n',
+ },
+ {
+ "id": "swift-mutating", "lang": "swift", "cat": "bugfix",
+ "symptom": "This code does not compile: a struct method modifies a stored property, and the calling code prevents mutation. After fixing, runCounter(5) must return 5.",
+ "buggy": "struct Counter {\n var count = 0\n func increment() {\n count += 1\n }\n}\n\nfunc runCounter(_ times: Int) -> Int {\n let c = Counter()\n for _ in 0..<times { c.increment() }\n return c.count\n}\n",
+ "fix": "struct Counter {\n var count = 0\n mutating func increment() {\n count += 1\n }\n}\n\nfunc runCounter(_ times: Int) -> Int {\n var c = Counter()\n for _ in 0..<times { c.increment() }\n return c.count\n}\n",
+ "test": 'import Foundation\n\nfunc check(_ cond: Bool, _ msg: String) {\n if !cond { FileHandle.standardError.write(("FAIL: " + msg + "\\n").data(using: .utf8)!); exit(1) }\n}\n\ncheck(runCounter(5) == 5, "runCounter(5)")\ncheck(runCounter(0) == 0, "runCounter(0)")\nprint("OK")\n',
+ },
+ {
+ "id": "ts-matrix-alias", "lang": "ts", "cat": "bugfix",
+ "symptom": "makeMatrix should return independent rows, but every row is the SAME array reference (writing to one row changes them all).",
+ "buggy": "export function makeMatrix(rows: number, cols: number): number[][] {\n return new Array(rows).fill(new Array(cols).fill(0));\n}\n",
+ "fix": "export function makeMatrix(rows: number, cols: number): number[][] {\n return Array.from({ length: rows }, () => new Array(cols).fill(0));\n}\n",
+ "test": 'import { makeMatrix } from "./solution";\nconst m = makeMatrix(2, 2);\nm[0][0] = 5;\nif (m[1][0] !== 0) { console.error("FAIL aliasing", m); process.exit(1); }\nif (m.length !== 2 || m[0].length !== 2) { console.error("FAIL shape", m); process.exit(1); }\nconsole.log("OK");\n',
+ },
+ {
+ "id": "ts-closure-var", "lang": "ts", "cat": "bugfix",
+ "symptom": "makeAdders should return functions that return 0, 1, and 2, but every function returns 3 (they all share the same loop variable).",
+ "buggy": "export function makeAdders(): Array<() => number> {\n const fns: Array<() => number> = [];\n for (var i = 0; i < 3; i++) {\n fns.push(() => i);\n }\n return fns;\n}\n",
+ "fix": "export function makeAdders(): Array<() => number> {\n const fns: Array<() => number> = [];\n for (let i = 0; i < 3; i++) {\n fns.push(() => i);\n }\n return fns;\n}\n",
+ "test": 'import { makeAdders } from "./solution";\nconst vals = makeAdders().map((f) => f());\nfunction eq(a: number[], b: number[]) { return a.length === b.length && a.every((v, i) => v === b[i]); }\nif (!eq(vals, [0, 1, 2])) { console.error("FAIL", vals); process.exit(1); }\nconsole.log("OK");\n',
+ },
+]
+
+BUGFIX = BUGFIX_T1 + BUGFIX_T2
+
+# ---------------------------------------------------------------- tool-calling
+WEATHER_TOOL = [{"type": "function", "function": {
+ "name": "get_weather",
+ "description": "Get the current weather for a location.",
+ "parameters": {"type": "object", "properties": {
+ "location": {"type": "string", "description": "City name"},
+ "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}},
+ "required": ["location", "unit"]}}}]
+CALC_TOOL = [{"type": "function", "function": {
+ "name": "calculator",
+ "description": "Evaluate an arithmetic expression.",
+ "parameters": {"type": "object", "properties": {"expression": {"type": "string"}},
+ "required": ["expression"]}}}]
+WEB_TOOL = {"type": "function", "function": {
+ "name": "web_search", "description": "Search the web.",
+ "parameters": {"type": "object", "properties": {"query": {"type": "string"}, "limit": {"type": "integer"}},
+ "required": ["query"]}}}
+THREE_TOOLS = [WEATHER_TOOL[0], CALC_TOOL[0], WEB_TOOL]
+TIMER_TOOL = [{"type": "function", "function": {
+ "name": "set_timer", "description": "Set a countdown timer.",
+ "parameters": {"type": "object", "properties": {"seconds": {"type": "integer", "description": "duration in seconds"}},
+ "required": ["seconds"]}}}]
+
+
+def _v_native(resp):
+ tcs = client.tool_calls(resp)
+ if not tcs:
+ return False, "no tool_calls returned (content=%r)" % client.content(resp)[:120]
+ fn = tcs[0]["function"]
+ if fn["name"] != "get_weather":
+ return False, "wrong tool: %s" % fn["name"]
+ try:
+ args = json.loads(fn["arguments"])
+ except Exception:
+ return False, "args not valid JSON: %r" % fn["arguments"][:120]
+ loc_ok = "paris" in str(args.get("location", "")).lower()
+ unit_ok = str(args.get("unit", "")).lower() == "celsius"
+ return (loc_ok and unit_ok), "args=%s loc_ok=%s unit_ok=%s" % (args, loc_ok, unit_ok)
+
+
+# The XML tool-call format uses angle brackets around a JSON object.
+# We build the regex pattern and system prompt here to avoid the literal
+# string being misinterpreted by tooling.
+_XML_PATTERN = re.compile(r"[\<\u003c]\s*(\{.*?\})\s*[\>\u003e]", re.DOTALL)
+
+def _v_xml(resp):
+ c = client.content(resp)
+ m = _XML_PATTERN.search(c)
+ if not m:
+ return False, "no XML tool-call found: %r" % c[:160]
+ try:
+ obj = json.loads(m.group(1))
+ except Exception:
+ return False, "inner JSON did not parse: %r" % m.group(1)[:160]
+ args = obj.get("arguments", obj.get("parameters", {}))
+ name_ok = obj.get("name") == "web_search"
+ q_ok = "gemma" in str(args.get("query", "")).lower()
+ limit_ok = str(args.get("limit")) == "5"
+ return (name_ok and q_ok and limit_ok), "obj=%s name_ok=%s q_ok=%s limit_ok=%s" % (obj, name_ok, q_ok, limit_ok)
+
+
+def _v_json(resp):
+ c = client.content(resp).strip()
+ m = re.search(r"\{.*\}", c, re.DOTALL)
+ if not m:
+ return False, "no JSON object in output: %r" % c[:160]
+ try:
+ obj = json.loads(m.group(0))
+ except Exception as e:
+ return False, "JSON parse error: %s | %r" % (e, m.group(0)[:160])
+ name_ok = isinstance(obj.get("name"), str) and "maria" in obj["name"].lower()
+ age_ok = obj.get("age") == 34
+ sk = obj.get("skills")
+ skills_ok = isinstance(sk, list) and {"go", "rust"} <= {str(x).lower() for x in sk}
+ return (name_ok and age_ok and skills_ok), "obj=%s name=%s age=%s skills=%s" % (obj, name_ok, age_ok, skills_ok)
+
+
+def _v_notool(resp):
+ tcs = client.tool_calls(resp)
+ if tcs:
+ return False, "incorrectly called a tool: %s" % tcs[0]["function"]["name"]
+ c = client.content(resp).lower()
+ return ("paris" in c), "answered in text, mentions Paris=%s" % ("paris" in c)
+
+
+def _v_multi(resp):
+ tcs = client.tool_calls(resp)
+ locs = []
+ for tc in tcs:
+ try:
+ a = json.loads(tc["function"]["arguments"])
+ except Exception:
+ a = {}
+ locs.append(str(a.get("location", "")).lower())
+ blob = " ".join(locs)
+ ok = len(tcs) >= 2 and "paris" in blob and "tokyo" in blob
+ return ok, "n_calls=%d locs=%s" % (len(tcs), locs)
+
+
+def _v_select(resp):
+ tcs = client.tool_calls(resp)
+ if not tcs:
+ return False, "no tool call; content=%r" % client.content(resp)[:140]
+ fn = tcs[0]["function"]
+ if fn["name"] != "calculator":
+ return False, "picked %s" % fn["name"]
+ try:
+ a = json.loads(fn["arguments"])
+ except Exception:
+ return False, "bad args %r" % fn["arguments"][:120]
+ expr = str(a.get("expression", ""))
+ return ("47" in expr and "89" in expr), "expr=%r" % expr
+
+
+def _v_nested(resp):
+ c = client.content(resp).strip()
+ m = re.search(r"\{.*\}", c, re.DOTALL)
+ if not m:
+ return False, "no json: %r" % c[:140]
+ try:
+ o = json.loads(m.group(0))
+ except Exception as e:
+ return False, "parse err %s" % e
+ svc = o.get("service", {})
+ ok = (isinstance(svc, dict) and str(svc.get("name", "")).lower() == "api"
+ and svc.get("port") == 8080 and o.get("replicas") == 3
+ and "prod" in str(o.get("env", "")).lower())
+ return ok, "obj=%s" % o
+
+
+def _v_derived(resp):
+ tcs = client.tool_calls(resp)
+ if not tcs:
+ return False, "no tool call; content=%r" % client.content(resp)[:140]
+ fn = tcs[0]["function"]
+ if fn["name"] != "set_timer":
+ return False, "picked %s" % fn["name"]
+ try:
+ a = json.loads(fn["arguments"])
+ except Exception:
+ return False, "bad args"
+ return a.get("seconds") == 150, "args=%s" % a
+
+
+# Build the XML tool-call system prompt without a literal angle-bracket tag.
+# The model is told to emit: <JSON object> on one line.
+_XML_SYS = ("You can call tools. To call a tool you MUST output exactly one line "
+ "of the form " + chr(60) + "{\"name\": \"<tool>\", \"arguments\": {...}}" + chr(62) +
+ " and nothing else — no prose, no markdown. Available tool: web_search(query: string, limit: integer).")
+
+
+TOOLCALL = [
+ {"id": "tc-native", "cat": "toolcall", "tools": WEATHER_TOOL, "validator": _v_native,
+ "messages": [{"role": "user", "content": "What is the current weather in Paris? Use celsius."}]},
+ {"id": "tc-xml", "cat": "toolcall", "tools": None, "validator": _v_xml,
+ "messages": [
+ {"role": "system", "content": _XML_SYS},
+ {"role": "user", "content": "Search the web for 'gemma benchmarks' and return at most 5 results."}]},
+ {"id": "tc-json", "cat": "toolcall", "tools": None, "validator": _v_json,
+ "messages": [
+ {"role": "system", "content": "Output ONLY a single JSON object and nothing else. No markdown fences, no commentary."},
+ {"role": "user", "content": "Extract this person into JSON with keys name (string), age (integer), skills (array of strings): 'Maria is 34 years old and knows Go and Rust.'"}]},
+ {"id": "tc-notool", "cat": "toolcall", "tools": CALC_TOOL, "validator": _v_notool,
+ "messages": [{"role": "user", "content": "What is the capital of France? Answer in one word."}]},
+ {"id": "tc-multi", "cat": "toolcall", "tools": WEATHER_TOOL, "validator": _v_multi,
+ "messages": [{"role": "user", "content": "What's the weather in Paris and in Tokyo right now? Use celsius for both."}]},
+ {"id": "tc-select", "cat": "toolcall", "tools": THREE_TOOLS, "validator": _v_select,
+ "messages": [{"role": "user", "content": "Using the tools available to you, compute 47 * 89 and give me the exact product."}]},
+ {"id": "tc-nested-json", "cat": "toolcall", "tools": None, "validator": _v_nested,
+ "messages": [
+ {"role": "system", "content": "Output ONLY a single JSON object, no markdown, no commentary."},
+ {"role": "user", "content": "Produce a deployment config JSON with this shape: {\"service\": {\"name\": string, \"port\": integer}, \"replicas\": integer, \"env\": string}. The service is named 'api', listens on port 8080, runs 3 replicas, in the production environment."}]},
+ {"id": "tc-derived", "cat": "toolcall", "tools": TIMER_TOOL, "validator": _v_derived,
+ "messages": [{"role": "user", "content": "Set a timer for two and a half minutes."}]},
+]
+
+
+def all_tasks():
+ return TOOLCALL + BUGFIX
+
+
+# ---------------------------------------------------------------- phases
+def validate():
+ print("=== VALIDATION: proving each bug-fix task is well-formed ===\n")
+ allok = True
+ for t in BUGFIX:
+ runner = RUNNERS[t["lang"]]
+ bug_pass, bug_out = runner(t, t["buggy"])
+ fix_pass, fix_out = runner(t, t["fix"])
+ valid = (not bug_pass) and fix_pass
+ allok = allok and valid
+ status = "OK " if valid else "BAD"
+ print("[%s] %-16s buggy_fails=%-5s fix_passes=%-5s" %
+ (status, t["id"], (not bug_pass), fix_pass))
+ if not valid:
+ if bug_pass:
+ print(" !! buggy code unexpectedly PASSED tests")
+ if not fix_pass:
+ print(" !! reference fix FAILED:\n " + fix_out.replace("\n", "\n ")[:800])
+ print("\nVALIDATION %s" % ("PASSED — all tasks well-formed" if allok else "FAILED"))
+ return allok
+
+
+def run():
+ tasks = all_tasks()
+ path = jsonl_path(client.MODEL, prefix="coding-suite")
+ open(path, "w").close()
+ results = []
+ print("=== RUN: %d tasks against %s ===\n" % (len(tasks), client.MODEL), flush=True)
+ for i, t in enumerate(tasks, 1):
+ print("[%d/%d] %-16s ..." % (i, len(tasks), t["id"]), end=" ", flush=True)
+ rec = {"id": t["id"], "cat": t["cat"]}
+ try:
+ if t["cat"] == "toolcall":
+ resp, dt = client.call_model(t["messages"], tools=t.get("tools"))
+ passed, detail = t["validator"](resp)
+ else:
+ sysmsg = ("You are an expert %s developer. Fix the bug in the code. "
+ "Respond with ONLY the corrected code in a single fenced code block. "
+ "Keep the same names and signatures. Do NOT add tests, a main function, "
+ "explanations, or extra declarations." % t["lang"].upper())
+ user = ("The following %s code has a bug. %s\n\n```%s\n%s```\n\nReturn the corrected code."
+ % (t["lang"], t["symptom"], t["lang"], t["buggy"]))
+ resp, dt = client.call_model(
+ [{"role": "system", "content": sysmsg},
+ {"role": "user", "content": user}])
+ code = extract_code(client.content(resp))
+ rec["code"] = code
+ passed, detail = RUNNERS[t["lang"]](t, code)
+ u = client.usage(resp)
+ rec.update({
+ "passed": bool(passed),
+ "detail": detail[:1200],
+ "latency_s": round(dt, 1),
+ "finish": client.finish_reason(resp),
+ "total_tokens": u["prompt_tokens"] + u["completion_tokens"],
+ "completion_tokens": u["completion_tokens"],
+ "reasoning_tokens": u["reasoning_tokens"],
+ "lang": t.get("lang", "-"),
+ })
+ print("%s (%.0fs, %s tok)" % ("PASS" if passed else "FAIL", dt, rec.get("total_tokens")), flush=True)
+ except Exception as e:
+ rec.update({"passed": False, "detail": "HARNESS/REQUEST ERROR: %r" % e, "lang": t.get("lang", "-")})
+ print("ERROR: %r" % e, flush=True)
+ results.append(rec)
+ append_jsonl(path, rec)
+ coding_suite_summary(results)
+ write_json(results, client.MODEL, prefix="coding-suite")
+
+
+def report():
+ path = jsonl_path(client.MODEL, prefix="coding-suite")
+ results = [json.loads(l) for l in open(path)]
+ coding_suite_summary(results)
+
+
+
+
+def main():
+ mode = sys.argv[1] if len(sys.argv) > 1 else 'validate'
+ if mode == 'validate':
+ sys.exit(0 if validate() else 1)
+ elif mode == 'run':
+ run()
+ elif mode == 'report':
+ report()
+ else:
+ print(f'usage: python bench.py coding-suite [validate|run|report]')
+ sys.exit(1)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/bench/tasks/loop_battery.py b/bench/tasks/loop_battery.py
new file mode 100644
index 0000000..030482a
--- /dev/null
+++ b/bench/tasks/loop_battery.py
@@ -0,0 +1,128 @@
+"""Loop-provoking battery. Prompts engineered to induce over-rethinking / runaway CoT.
+Runs each probe RUNS times at temp 0.6 (the model's recommended sampler for reasoning).
+Captures: finish_reason (length == ran to cap == runaway), completion_tokens, wall time,
+thinking length, re-derivation markers, and an n-gram repetition score (a high max
+repeated-10gram count == a hard loop, the model literally repeating itself).
+
+Thinking is parsed from `reasoning_content` (vLLM / LM Studio separate channel) or
+falling back to inline <think>...</think> tags in the content. Models that expose no
+thinking channel at all have their answer text scanned instead and flagged so a silent
+0-marker report isn't mistaken for clean reasoning.
+
+Usage (via the bench.py CLI):
+ python bench.py loop-battery run
+
+Env (via .env): ENDPOINT, MODEL, API_KEY, MAX_TOKENS, TEMPERATURE.
+ RUNS number of runs per probe (default 2)
+ NUM_PREDICT per-probe generation cap (default 16000; falls back to MAX_TOKENS)
+"""
+import os
+import re
+import sys
+from collections import Counter
+
+from bench.core import client
+from bench.core.reporting import RESULTS_DIR, safe_label
+
+RUNS = int(os.environ.get("RUNS", "2"))
+NUM_PREDICT = int(os.environ.get("NUM_PREDICT", os.environ.get("MAX_TOKENS", "16000")))
+
+PROBES = [
+ ("self-ref-wordcount",
+ "How many words will be in your final answer to this exact question? Your answer must "
+ "explicitly state that number, and the number must exactly match the number of words you "
+ "actually write. Give one definitive final answer."),
+ ("underdetermined-ages",
+ "Three positive integers multiply to 36. I tell you their sum but you can't determine the "
+ "ages from it; then I tell you the oldest one likes chess. What are the three ages? Give a "
+ "single definitive answer."),
+ ("prove-false",
+ "Prove rigorously that every odd integer greater than 2 is prime. Provide a complete proof."),
+ ("conflicting-opt",
+ "What is the single best general-purpose programming language? I need exactly one definitive "
+ "answer that is optimal across performance, safety, ecosystem, learning curve, and concurrency "
+ "simultaneously. Do not hedge or give conditions."),
+ ("ambiguous-pronoun-math",
+ "A man buys 3 apples and 2 oranges for $7. His friend buys 2 apples and 3 oranges and pays "
+ "the same. Later he says 'I paid a dollar more than him for mine.' What does each fruit cost? "
+ "Give a single definitive numeric answer."),
+]
+
+MARKERS = ["wait", "let me reconsider", "hold on", "actually,", "let me recompute",
+ "re-examine", "let me redo", "hmm", "but wait", "double-check", "recheck",
+ "scratch that", "on second thought", "let me restart", "let me try again"]
+
+
+def split_think(msg, content):
+ rc = msg.get("reasoning_content")
+ if rc:
+ return rc, content
+ m = re.search(r"<think>(.*?)</think>(.*)$", content, re.DOTALL)
+ if m:
+ return m.group(1).strip(), m.group(2).strip()
+ return "", content
+
+
+def rep_score(text):
+ w = text.split()
+ if len(w) < 20:
+ return 0, 1.0
+ grams = Counter(tuple(w[i:i + 10]) for i in range(len(w) - 9))
+ maxrep = max(grams.values())
+ uniq_ratio = len(set(w)) / len(w)
+ return maxrep, round(uniq_ratio, 3)
+
+
+def run():
+ model = client.MODEL
+ print("### LOOP BATTERY vs %s (temp0.6, num_predict=%d, %d runs each) ###" %
+ (model, NUM_PREDICT, RUNS), flush=True)
+ safe = safe_label(model)
+ for label, prompt in PROBES:
+ for n in range(1, RUNS + 1):
+ msgs = [{"role": "user", "content": prompt}]
+ try:
+ resp, dt = client.call_model(
+ msgs, max_tokens=NUM_PREDICT, temperature=0.6)
+ except Exception as e:
+ print("[%s #%d] ERROR %r" % (label, n, e), flush=True)
+ continue
+ ch = resp.get("choices", [{}])[0]
+ msg = ch.get("message", {}) or {}
+ content_str = msg.get("content") or ""
+ think, ans = split_think(msg, content_str)
+ scan = think if think else ans
+ fallback = "" if think else " [no-think-channel, scanned=answer]"
+ done = ch.get("finish_reason")
+ u = client.usage(resp)
+ ct = u["completion_tokens"]
+ low = scan.lower()
+ mk = sum(low.count(m) for m in MARKERS)
+ maxrep, uniq = rep_score(scan)
+ runaway = "RUNAWAY(cap)" if done == "length" else "ok"
+ print("[%-22s #%d] %-12s done=%-7s gen_tok=%-6s wall=%5.1fs "
+ "think_words=%-5d markers=%-3d max10gram=%d uniq=%.2f%s" %
+ (label, n, runaway, done, ct, dt, len(scan.split()),
+ mk, maxrep, uniq, fallback), flush=True)
+ path = os.path.join(
+ RESULTS_DIR, "loop_%s_%s_%d.txt" % (safe, label, n))
+ with open(path, "w") as f:
+ f.write("MODEL:%s\nPROMPT:\n%s\n\n=== THINKING ===\n%s\n\n"
+ "=== ANSWER ===\n%s\n" %
+ (model, prompt, think, ans))
+ print("DONE.", flush=True)
+
+
+
+
+def main():
+ mode = sys.argv[1] if len(sys.argv) > 1 else "run"
+ if mode == "run":
+ run()
+ else:
+ print(f"usage: python bench.py loop-battery [run]")
+ sys.exit(1)
+
+
+if __name__ == "__main__":
+ main()
\ No newline at end of file
diff --git a/bench/tasks/multitask/__init__.py b/bench/tasks/multitask/__init__.py
new file mode 100644
index 0000000..9f976f4
--- /dev/null
+++ b/bench/tasks/multitask/__init__.py
@@ -0,0 +1,12 @@
+"""Multitask task modules — auto-discovered by agentic_multitask.py.
+
+Each .py file (not starting with _) exposes:
+ TASK_NAME str
+ PLAN str (the task plan the agent executes)
+ STARTER dict (starter scaffold files)
+ REFERENCE dict (reference solution — used by validate only)
+ validate(d) function (materialize REFERENCE, run gates, print result)
+ gates(d) function (return dict of gate_name -> bool)
+ TOOLS list (optional, defaults to core tools)
+ TOOL_IMPLS dict (optional, defaults to core tool impls)
+"""
\ No newline at end of file
diff --git a/bench/tasks/multitask/go_htmx.py b/bench/tasks/multitask/go_htmx.py
new file mode 100644
index 0000000..2e7356f
--- /dev/null
+++ b/bench/tasks/multitask/go_htmx.py
@@ -0,0 +1,567 @@
+#!/usr/bin/env python3
+"""
+Task: go_htmx — a Go web app with HTMX partials + multi-table SQLite.
+
+Models a real-world Go web tool in miniature: single Go binary, server-rendered
+html/template + HTMX for partial updates, embedded SQLite (modernc.org/sqlite),
+a multi-table schema with a self-referential location tree (recursive CTE
+read), embedded assets via //go:embed. The "real product" = a binary that
+serves a page listing locations in a tree, lets you add a sub-location via an
+HTMX partial swap (no full page reload), and persists to SQLite.
+"""
+import os, sys, subprocess, time, shutil, signal, socket, urllib.request, urllib.parse, json
+HERE = os.path.dirname(os.path.abspath(__file__))
+from bench.core import tools
+
+TASK_NAME = "go_htmx"
+
+PLAN = """\
+# Task — Build a Go web app with HTMX partials + multi-table SQLite
+
+## Goal
+`go build ./...` and `go test ./...` run green, and the binary starts on port
+8080, serves a locations tree page at `/`, lets you add a sub-location via an
+HTMX partial swap (POST returns just the updated tree HTML fragment, not a full
+page), and persists to embedded SQLite. This is a Go web app with HTMX partials
++ multi-table SQLite: single Go binary, server-rendered html/template + HTMX,
+embedded SQLite, multi-table schema with a self-referential location tree read
+via a recursive CTE.
+
+## Prerequisites
+Starter scaffold in the work dir:
+ - `go.mod` (module `locations`, go 1.26, no deps)
+ - `cmd/server/main.go` (empty stub: `package main; func main(){}`)
+
+## Spec — the app to build
+A single Go binary serving a one-page location-tree manager over HTTP. Uses
+HTMX for partial updates (adding a sub-location swaps just the tree fragment,
+no full page reload). Embedded SQLite via `modernc.org/sqlite`. Embedded
+templates + migrations via `//go:embed`.
+
+### Schema (two tables)
+ locations(
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
+ name TEXT NOT NULL,
+ parent_id INTEGER NULL REFERENCES locations(id) ON DELETE CASCADE,
+ created_at DATETIME DEFAULT CURRENT_TIMESTAMP
+ )
+ -- UNIQUE(name, parent_id) — names unique within a parent (NULL parent = root)
+ CREATE UNIQUE INDEX locations_name_parent ON locations(name, IFNULL(parent_id, 0));
+
+### Files to produce (layout)
+ cmd/server/main.go boot: config -> db -> migrate -> handlers -> ListenAndServe + graceful shutdown
+ internal/config/config.go Config{Port, DBPath}; LoadFromEnv() reads NOTES_PORT (port, int) and NOTES_DB (db path) — MUST use these exact env var names; the scoring harness sets them.
+ internal/db/db.go Open(dsn): modernc sqlite, PRAGMA WAL+busy_timeout+FK on; Migrate() via embedded .sql
+ internal/db/migrations/00001_init.sql the schema above (both tables + the unique index)
+ internal/models/location.go Location struct + Create(db, name, parentID) + Tree(db) — Tree returns a flat list with depth via a recursive CTE: WITH RECURSIVE tree AS (SELECT id, name, parent_id, 0 AS depth FROM locations WHERE parent_id IS NULL UNION ALL SELECT l.id, l.name, l.parent_id, t.depth+1 FROM locations l JOIN tree t ON l.parent_id = t.id) SELECT id, name, parent_id, depth FROM tree ORDER BY depth, name
+ internal/handlers/handlers.go Handlers struct; routes via stdlib net/http ServeMux: GET / (full page: form + tree), POST / (create location: form fields name + parent_id; returns JUST the updated tree fragment as an HTML partial for HTMX swap, Content-Type text/html, status 200 — NOT a redirect, NOT a full page)
+ internal/handlers/handlers_test.go httptest: POST a root location, assert the response body contains the location name indented by depth (one leading space pair per depth level); POST a child, assert it appears nested under the parent
+ web/templates/page.html the full page: <head> pulls in HTMX (a <script src="https://unpkg.com/htmx.org"></script> is fine), a form (name input + hidden parent_id input + submit) with hx-post="/" hx-target="#tree" hx-swap="innerHTML", and a <div id="tree"> containing the rendered tree
+ web/templates/tree.html the tree fragment: renders the locations as a nested <ul> (one <li> per location, indented by depth; children nested in a <ul> inside the parent <li>). This is the partial returned by POST / for the HTMX swap.
+ web/web.go package web; //go:embed templates/*.html + a Parse() helper returning a *template.Template (or two — one for the page, one for the tree fragment)
+
+### Constraints
+ - Use `modernc.org/sqlite` (pure-Go, CGO_ENABLED=0). Import path `_ "modernc.org/sqlite"`.
+ - Embed migrations AND templates via `//go:embed` (templates live in the `web/` package so the embed path is package-local — do NOT use `../` in //go:embed patterns, Go forbids it).
+ - Stdlib `net/http` ServeMux only — no Gin, no chi.
+ - HTMX: the POST / handler returns JUST the tree fragment (Content-Type text/html, status 200), not a redirect and not a full page. The form has `hx-post="/" hx-target="#tree" hx-swap="innerHTML"` so HTMX swaps the tree div's content with the response.
+ - The tree is rendered as a nested <ul>/<li> structure. A root location is a top-level <li>; a child is a <li> inside a <ul> inside its parent's <li>.
+ - Read NOTES_PORT and NOTES_DB env vars (exact names — the harness sets them to run the binary on a free port with a temp db).
+
+## Steps
+1. `go get modernc.org/sqlite && go mod tidy`
+2. Write the files. Natural order: config -> db -> models -> web/templates -> handlers -> main.
+3. Build (`go build ./...`). Fix compile errors.
+4. Test (`go test ./...`). The handlers_test.go must pass.
+5. Optionally run the binary briefly to confirm it serves.
+
+## Acceptance (harness-verified)
+1. `go build ./...` exits 0.
+2. `go test ./...` exits 0.
+3. Binary starts, `GET /` returns 200 with `<div id="tree"` and `htmx` in the body.
+4. `POST /` with form fields name+parent_id returns 200 with the tree fragment (contains the new location name); a subsequent `GET /` shows it in the tree; the SQLite file has the row.
+
+## Out of scope
+Auth, sessions, CSRF, styling beyond minimal readable HTML, delete/edit, drag-drop, any JS beyond the HTMX script tag. This is the smallest real Go web app with HTMX partials + multi-table SQLite — a working location-tree one-pager with HTMX partials.
+
+## Commit
+You do not commit; the harness owns the work dir.
+"""
+
+STARTER = {
+ "go.mod": "module locations\n\ngo 1.26\n",
+ "cmd/server/main.go": "package main\n\nfunc main() {}\n",
+}
+
+REFERENCE = {
+ "go.mod": "module locations\n\ngo 1.26\n\nrequire modernc.org/sqlite v1.50.0\n",
+ "cmd/server/main.go": '''package main
+
+import (
+ "context"
+ "fmt"
+ "log"
+ "net/http"
+ "os"
+ "os/signal"
+ "syscall"
+ "time"
+
+ "locations/internal/config"
+ "locations/internal/db"
+ "locations/internal/handlers"
+ "locations/internal/models"
+)
+
+func main() {
+ cfg := config.LoadFromEnv()
+ database, err := db.Open(cfg.DBPath)
+ if err != nil { log.Fatalf("db open: %v", err) }
+ defer database.Close()
+ if err := database.Migrate(); err != nil { log.Fatalf("migrate: %v", err) }
+ store := &models.Store{DB: database.DB}
+ h := handlers.New(store)
+ srv := &http.Server{Addr: fmt.Sprintf(":%d", cfg.Port), Handler: h.Routes()}
+ go func() {
+ log.Printf("listening on :%d", cfg.Port)
+ if err := srv.ListenAndServe(); err != nil && err != http.ErrServerClosed {
+ log.Fatalf("serve: %v", err)
+ }
+ }()
+ stop := make(chan os.Signal, 1)
+ signal.Notify(stop, os.Interrupt, syscall.SIGTERM)
+ <-stop
+ log.Println("shutting down")
+ ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
+ defer cancel()
+ _ = srv.Shutdown(ctx)
+}
+''',
+ "internal/config/config.go": '''package config
+
+import (
+ "os"
+ "strconv"
+)
+
+type Config struct {
+ Port int
+ DBPath string
+}
+
+func LoadFromEnv() Config {
+ cfg := Config{Port: 8080, DBPath: "locations.db"}
+ if p := os.Getenv("NOTES_PORT"); p != "" {
+ if v, err := strconv.Atoi(p); err == nil { cfg.Port = v }
+ }
+ if d := os.Getenv("NOTES_DB"); d != "" { cfg.DBPath = d }
+ return cfg
+}
+''',
+ "internal/db/db.go": '''package db
+
+import (
+ "database/sql"
+ "embed"
+ "fmt"
+
+ _ "modernc.org/sqlite"
+)
+
+//go:embed migrations/*.sql
+var migrationsFS embed.FS
+
+type Database struct {
+ *sql.DB
+}
+
+func Open(dsn string) (*Database, error) {
+ db, err := sql.Open("sqlite", dsn)
+ if err != nil { return nil, fmt.Errorf("open sqlite: %w", err) }
+ if _, err := db.Exec(`PRAGMA journal_mode=WAL; PRAGMA busy_timeout=5000; PRAGMA foreign_keys=ON;`); err != nil {
+ db.Close(); return nil, fmt.Errorf("pragmas: %w", err)
+ }
+ return &Database{DB: db}, nil
+}
+
+func (d *Database) Migrate() error {
+ if _, err := d.Exec(`CREATE TABLE IF NOT EXISTS schema_migrations (name TEXT PRIMARY KEY, applied_at DATETIME DEFAULT CURRENT_TIMESTAMP)`); err != nil {
+ return fmt.Errorf("create schema_migrations: %w", err)
+ }
+ entries, err := migrationsFS.ReadDir("migrations")
+ if err != nil { return fmt.Errorf("read migrations dir: %w", err) }
+ for _, e := range entries {
+ if e.IsDir() { continue }
+ name := e.Name()
+ var applied int
+ if err := d.QueryRow(`SELECT count(*) FROM schema_migrations WHERE name = ?`, name).Scan(&applied); err != nil {
+ return fmt.Errorf("check %s: %w", name, err)
+ }
+ if applied > 0 { continue }
+ b, err := migrationsFS.ReadFile("migrations/" + name)
+ if err != nil { return fmt.Errorf("read %s: %w", name, err) }
+ if _, err := d.Exec(string(b)); err != nil { return fmt.Errorf("apply %s: %w", name, err) }
+ if _, err := d.Exec(`INSERT INTO schema_migrations (name) VALUES (?)`, name); err != nil {
+ return fmt.Errorf("record %s: %w", name, err)
+ }
+ }
+ return nil
+}
+''',
+ "internal/db/migrations/00001_init.sql": '''CREATE TABLE IF NOT EXISTS locations (
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
+ name TEXT NOT NULL,
+ parent_id INTEGER NULL REFERENCES locations(id) ON DELETE CASCADE,
+ created_at DATETIME DEFAULT CURRENT_TIMESTAMP
+);
+CREATE UNIQUE INDEX IF NOT EXISTS locations_name_parent ON locations(name, IFNULL(parent_id, 0));
+''',
+ "internal/models/location.go": '''package models
+
+import (
+ "database/sql"
+ "fmt"
+)
+
+type Location struct {
+ ID int64
+ Name string
+ ParentID *int64
+ Depth int
+}
+
+type Store struct {
+ DB *sql.DB
+}
+
+func (s *Store) Create(name string, parentID *int64) (*Location, error) {
+ res, err := s.DB.Exec(`INSERT INTO locations (name, parent_id) VALUES (?, ?)`, name, parentID)
+ if err != nil { return nil, fmt.Errorf("insert location: %w", err) }
+ id, err := res.LastInsertId()
+ if err != nil { return nil, err }
+ return &Location{ID: id, Name: name, ParentID: parentID, Depth: 0}, nil
+}
+
+func (s *Store) Tree() ([]Location, error) {
+ rows, err := s.DB.Query(`WITH RECURSIVE tree AS (
+ SELECT id, name, parent_id, 0 AS depth FROM locations WHERE parent_id IS NULL
+ UNION ALL
+ SELECT l.id, l.name, l.parent_id, t.depth+1 FROM locations l JOIN tree t ON l.parent_id = t.id
+ ) SELECT id, name, parent_id, depth FROM tree ORDER BY depth, name`)
+ if err != nil { return nil, fmt.Errorf("tree query: %w", err) }
+ defer rows.Close()
+ var out []Location
+ for rows.Next() {
+ var l Location
+ var pid sql.NullInt64
+ if err := rows.Scan(&l.ID, &l.Name, &pid, &l.Depth); err != nil { return nil, err }
+ if pid.Valid { l.ParentID = &pid.Int64 }
+ out = append(out, l)
+ }
+ return out, rows.Err()
+}
+''',
+ "internal/handlers/handlers.go": '''package handlers
+
+import (
+ "embed"
+ "html/template"
+ "log"
+ "net/http"
+ "strconv"
+ "strings"
+
+ "locations/internal/models"
+)
+
+//go:embed templates/*.html
+var templatesFS embed.FS
+
+type Handlers struct {
+ store *models.Store
+ pageTmpl *template.Template
+ treeTmpl *template.Template
+}
+
+func New(store *models.Store) *Handlers {
+ pageTmpl := template.Must(template.ParseFS(templatesFS, "templates/page.html", "templates/tree.html"))
+ treeTmpl := template.Must(template.ParseFS(templatesFS, "templates/tree.html"))
+ return &Handlers{store: store, pageTmpl: pageTmpl, treeTmpl: treeTmpl}
+}
+
+func (h *Handlers) Routes() http.Handler {
+ mux := http.NewServeMux()
+ mux.HandleFunc("GET /", h.index)
+ mux.HandleFunc("POST /", h.create)
+ return mux
+}
+
+func (h *Handlers) index(w http.ResponseWriter, r *http.Request) {
+ locations, err := h.store.Tree()
+ if err != nil {
+ log.Printf("tree: %v", err)
+ http.Error(w, "internal error", http.StatusInternalServerError)
+ return
+ }
+ w.Header().Set("Content-Type", "text/html; charset=utf-8")
+ if err := h.pageTmpl.Execute(w, locations); err != nil {
+ log.Printf("render page: %v", err)
+ }
+}
+
+func (h *Handlers) create(w http.ResponseWriter, r *http.Request) {
+ name := strings.TrimSpace(r.FormValue("name"))
+ if name == "" {
+ http.Error(w, "name is required", http.StatusBadRequest)
+ return
+ }
+ parentStr := r.FormValue("parent_id")
+ var parentID *int64
+ if parentStr != "" {
+ id, err := strconv.ParseInt(parentStr, 10, 64)
+ if err != nil {
+ http.Error(w, "bad parent_id", http.StatusBadRequest)
+ return
+ }
+ parentID = &id
+ }
+ if _, err := h.store.Create(name, parentID); err != nil {
+ log.Printf("create: %v", err)
+ http.Error(w, "internal error", http.StatusInternalServerError)
+ return
+ }
+ locations, err := h.store.Tree()
+ if err != nil {
+ log.Printf("tree after create: %v", err)
+ http.Error(w, "internal error", http.StatusInternalServerError)
+ return
+ }
+ // Return JUST the tree fragment for the HTMX swap (not a redirect, not a full page).
+ w.Header().Set("Content-Type", "text/html; charset=utf-8")
+ w.WriteHeader(http.StatusOK)
+ if err := h.treeTmpl.ExecuteTemplate(w, "tree", locations); err != nil {
+ log.Printf("render tree: %v", err)
+ }
+}
+''',
+ "internal/handlers/handlers_test.go": '''package handlers
+
+import (
+ "database/sql"
+ "io"
+ "net/http"
+ "net/http/httptest"
+ "net/url"
+ "strconv"
+ "strings"
+ "testing"
+
+ "locations/internal/models"
+
+ _ "modernc.org/sqlite"
+)
+
+func newTestStore(t *testing.T) (*sql.DB, func()) {
+ t.Helper()
+ db, err := sql.Open("sqlite", ":memory:")
+ if err != nil { t.Fatalf("open: %v", err) }
+ _, err = db.Exec(`CREATE TABLE locations (id INTEGER PRIMARY KEY AUTOINCREMENT, name TEXT NOT NULL, parent_id INTEGER NULL REFERENCES locations(id), created_at DATETIME DEFAULT CURRENT_TIMESTAMP)`)
+ if err != nil { db.Close(); t.Fatalf("create table: %v", err) }
+ return db, func() { db.Close() }
+}
+
+func TestCreateRootAndChild(t *testing.T) {
+ db, cleanup := newTestStore(t)
+ defer cleanup()
+ store := &models.Store{DB: db}
+ h := New(store)
+
+ srv := httptest.NewServer(h.Routes())
+ defer srv.Close()
+
+ // POST a root location.
+ resp, err := http.PostForm(srv.URL+"/", url.Values{"name": {"BuildingA"}, "parent_id": {""}})
+ if err != nil { t.Fatalf("post root: %v", err) }
+ resp.Body.Close()
+ if resp.StatusCode != http.StatusOK {
+ t.Fatalf("post root status = %d, want 200", resp.StatusCode)
+ }
+
+ // Get the root's id from the DB.
+ var rootID int64
+ if err := db.QueryRow("SELECT id FROM locations WHERE name = 'BuildingA'").Scan(&rootID); err != nil {
+ t.Fatalf("query root id: %v", err)
+ }
+
+ // POST a child.
+ resp2, err := http.PostForm(srv.URL+"/", url.Values{"name": {"Floor1"}, "parent_id": {strconv.FormatInt(rootID, 10)}})
+ if err != nil { t.Fatalf("post child: %v", err) }
+ defer resp2.Body.Close()
+ if resp2.StatusCode != http.StatusOK {
+ t.Fatalf("post child status = %d, want 200", resp2.StatusCode)
+ }
+ body, _ := io.ReadAll(resp2.Body)
+ if !strings.Contains(string(body), "Floor1") {
+ t.Fatalf("response does not contain child name; got:\\n%s", body)
+ }
+ if !strings.Contains(string(body), "BuildingA") {
+ t.Fatalf("response does not contain root name (tree should include all); got:\\n%s", body)
+ }
+}
+''',
+ "internal/handlers/templates/page.html": '''<!DOCTYPE html>
+<html lang="en">
+<head>
+<meta charset="utf-8">
+<meta name="viewport" content="width=device-width, initial-scale=1">
+<title>Locations</title>
+<script src="https://unpkg.com/htmx.org"></script>
+<style>
+body { font-family: system-ui, sans-serif; max-width: 40em; margin: 2rem auto; padding: 0 1rem; }
+form { margin: 1rem 0; padding: 1rem; border: 1px solid #ccc; border-radius: 6px; }
+label { display: block; margin: 0.5rem 0; }
+input[type=text] { width: 100%; box-sizing: border-box; padding: 0.4rem; }
+ul.tree { list-style: none; padding-left: 1.5em; }
+li.location { margin: 0.25rem 0; }
+</style>
+</head>
+<body>
+<h1>Locations</h1>
+<form hx-post="/" hx-target="#tree" hx-swap="innerHTML">
+ <label>Name <input type="text" name="name" required></label>
+ <input type="hidden" name="parent_id" value="">
+ <button type="submit">Add location</button>
+</form>
+<div id="tree">
+{{template "tree" .}}
+</div>
+</body>
+</html>
+''',
+ "internal/handlers/templates/tree.html": '''{{define "tree"}}<ul class="tree">
+{{range .}}<li class="location" data-depth="{{.Depth}}">{{.Name}}</li>
+{{end}}</ul>{{end}}
+''',
+}
+
+def _materialize(files, dest):
+ shutil.rmtree(dest, ignore_errors=True)
+ os.makedirs(dest, exist_ok=True)
+ for path, content in files.items():
+ fp = os.path.join(dest, path)
+ os.makedirs(os.path.dirname(fp), exist_ok=True)
+ with open(fp, "w") as f: f.write(content)
+ return dest
+
+def _run(cmd, cwd, timeout=180):
+ try:
+ p = subprocess.run(cmd, cwd=cwd, shell=True, capture_output=True, text=True, timeout=timeout)
+ except subprocess.TimeoutExpired:
+ return "TIMEOUT"
+ return f"exit={p.returncode}\n{(p.stdout + p.stderr).strip()}"
+
+def _free_port():
+ s = socket.socket(socket.AF_INET, socket.SOCK_STREAM); s.bind(("", 0))
+ p = s.getsockname()[1]; s.close(); return p
+
+def _has_real_app(work_dir):
+ main_go = os.path.join(work_dir, "cmd", "server", "main.go")
+ if not os.path.isfile(main_go): return False
+ with open(main_go) as f:
+ src = f.read()
+ if "ListenAndServe" not in src and "http.Server" not in src: return False
+ internal = os.path.join(work_dir, "internal")
+ if not os.path.isdir(internal): return False
+ pkg_dirs = [d for d in os.listdir(internal) if os.path.isdir(os.path.join(internal, d))]
+ return len(pkg_dirs) >= 3
+
+def gates(work_dir):
+ """Return a dict of gate_name -> bool."""
+ g = {}
+ if not _has_real_app(work_dir):
+ return {"build": False, "test": False, "serves": False, "persists": False}
+ b = _run("go build ./...", work_dir)
+ g["build"] = b.startswith("exit=0")
+ t = _run("go test ./...", work_dir)
+ g["test"] = t.startswith("exit=0")
+ # serves + persists
+ binpath = os.path.abspath(os.path.join(work_dir, "server"))
+ r = _run(f"go build -o {binpath} ./cmd/server", work_dir)
+ if not r.startswith("exit=0") or not os.path.isfile(binpath):
+ g["serves"] = False; g["persists"] = False; return g
+ port = _free_port()
+ dbpath = os.path.abspath(os.path.join(work_dir, f"bench_{port}.db"))
+ import os as _os
+ for ext in ("", "-wal", "-shm"):
+ if _os.path.exists(dbpath+ext): _os.remove(dbpath+ext)
+ env = dict(_os.environ, NOTES_PORT=str(port), NOTES_DB=dbpath)
+ try:
+ proc = subprocess.Popen([binpath], cwd=work_dir, env=env, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
+ except Exception:
+ g["serves"] = False; g["persists"] = False; return g
+ try:
+ for _ in range(40):
+ try:
+ with socket.create_connection(("127.0.0.1", port), 0.25): break
+ except OSError: time.sleep(0.25)
+ else:
+ g["serves"] = False; g["persists"] = False; return g
+ base = f"http://127.0.0.1:{port}"
+ try:
+ with urllib.request.urlopen(base+"/", timeout=20) as resp:
+ if resp.status != 200: g["serves"]=False; g["persists"]=False; return g
+ html = resp.read().decode(errors="replace")
+ except Exception:
+ g["serves"]=False; g["persists"]=False; return g
+ g["serves"] = ('<div id="tree"' in html) and ('htmx' in html.lower())
+ if not g["serves"]:
+ g["persists"]=False; return g
+ # POST a location (HTMX-style: expects 200 + fragment, not a redirect)
+ data = urllib.parse.urlencode({"name":"BenchLoc","parent_id":""}).encode()
+ try:
+ req = urllib.request.Request(base+"/", data=data, method="POST")
+ with urllib.request.urlopen(req, timeout=20) as resp:
+ post_body = resp.read().decode(errors="replace")
+ except Exception:
+ g["persists"]=False; return g
+ g["persists"] = ("BenchLoc" in post_body)
+ if g["persists"] and os.path.exists(dbpath):
+ try:
+ import sqlite3
+ conn = sqlite3.connect(dbpath)
+ cur = conn.execute("SELECT count(*) FROM locations WHERE name='BenchLoc'")
+ g["persists"] = cur.fetchone()[0] > 0
+ conn.close()
+ except Exception:
+ pass
+ return g
+ finally:
+ proc.send_signal(signal.SIGTERM)
+ try: proc.wait(timeout=5)
+ except subprocess.TimeoutExpired: proc.kill()
+ for ext in ("", "-wal", "-shm"):
+ p = dbpath+ext
+ if os.path.exists(p):
+ try: os.remove(p)
+ except: pass
+
+def validate(work_dir):
+ _materialize(REFERENCE, dest=work_dir)
+ r = _run("go mod tidy", work_dir)
+ print("go mod tidy:", r[:120])
+ b = _run("go build ./...", work_dir)
+ print(f"build: {'OK' if b.startswith('exit=0') else 'FAIL'}\n{b[-300:]}")
+ t = _run("go test ./...", work_dir)
+ print(f"test: {'OK' if t.startswith('exit=0') else 'FAIL'}\n{t[-300:]}")
+ g = gates(work_dir)
+ print(f"gates: {g}")
+ ok = all(g.values())
+ print(f"-> {'OK' if ok else 'BAD'}")
+ return ok
+
+# Allow `python go_htmx.py` to run validate directly
+if __name__ == "__main__":
+ import sys
+ _work_root = os.path.join(HERE, "..", "..", "..", "work")
+ wd = os.path.join(_work_root, "multitask_validate_go_htmx")
+ validate(wd)
\ No newline at end of file
diff --git a/bench/tasks/multitask/rust_cli.py b/bench/tasks/multitask/rust_cli.py
new file mode 100644
index 0000000..e19457a
--- /dev/null
+++ b/bench/tasks/multitask/rust_cli.py
@@ -0,0 +1,343 @@
+#!/usr/bin/env python3
+"""
+Task: rust_cli — a Rust CLI with reqwest + quick-xml + async.
+
+Models a Rust CLI with reqwest + quick-xml + async in miniature: a Rust CLI
+that fetches XML from an HTTP endpoint with Digest auth, parses it with
+serde/quick-xml, and prints a structured result. Async (tokio), reqwest
+(rustls), quick-xml for parsing. The "real product" = a binary that hits a
+mock HTTP server the harness spins up (serves a fixed XML doc behind Digest
+auth), fetches it, parses the device list, and prints one device per line.
+"""
+import os, sys, subprocess, time, shutil, signal, socket, urllib.request, urllib.parse, json, http.server, threading, base64
+
+from bench.core import tools
+
+HERE = os.path.dirname(os.path.abspath(__file__))
+
+TASK_NAME = "rust_cli"
+
+PLAN = """\
+# Task — Build a Rust CLI that fetches + parses XML over HTTP with Digest auth
+
+## Goal
+`cargo build` and `cargo test` run green, and the resulting binary fetches an
+XML device list from a URL (with HTTP Digest auth), parses it with serde +
+quick-xml, and prints one device per line as `id|name|ip`. A Rust CLI with
+reqwest + quick-xml + async: Rust + reqwest (rustls) + quick-XML + async
+(tokio) + a native binary that talks to an HTTP endpoint.
+
+## Prerequisites
+Starter scaffold in the work dir:
+ - `Cargo.toml` (package `device-list`, edition 2021, no deps)
+ - `src/main.rs` (empty stub: `fn main() {}`)
+
+## Spec — the app to build
+A Rust CLI that takes `--url <URL> --user <user> --pass <pass>`, fetches the
+URL with HTTP Digest auth, parses the XML response as a list of `<Device>`
+elements (each with `<ID>`, `<Name>`, `<IP>`), and prints each as
+`id|name|ip` on its own line. Async via tokio; HTTP via reqwest (rustls, no
+default features); XML via quick-xml + serde.
+
+### XML format the server returns (Content-Type: application/xml)
+ <?xml version="1.0" encoding="UTF-8"?>
+ <DeviceList>
+ <Device>
+ <ID>1</ID>
+ <Name>Camera-Front</Name>
+ <IP>10.0.0.50</IP>
+ </Device>
+ <Device>
+ <ID>2</ID>
+ <Name>Camera-Back</Name>
+ <IP>10.0.0.51</IP>
+ </Device>
+ </DeviceList>
+
+### Files to produce
+ Cargo.toml [package] name="device-list", edition="2021"; [dependencies] tokio (rt-multi-thread + macros), reqwest (rustls, no default features, charset), quick-xml (serde), serde (derive), serde_json (optional, for debug). Binary at src/main.rs.
+ src/main.rs arg parsing (manual or std::env::args -- no clap needed), the async fetch (reqwest with Digest auth), XML parse (quick-xml de::from_str into a Vec<Device>), print `id|name|ip` per line. Use `reqwest::Client` with `.digest()` (the `digest_auth` or reqwest's digest support — use the `digest-auth` feature or handle the 401→retry manually). The simplest path: reqwest doesn't have built-in digest, so use the `digest_auth` crate OR handle the two-step (GET → 401 with WWW-Authenticate → compute response → GET with Authorization) manually. For this bench, either is fine — the mock server's Digest challenge is standard RFC 7616.
+ src/parser.rs the XML types: `#[derive(Debug, Deserialize)] struct DeviceList { #[serde(rename = "Device")] devices: Vec<Device> }` and `#[derive(Debug, Deserialize)] struct Device { #[serde(rename = "ID")] id: String, #[serde(rename = "Name")] name: String, #[serde(rename = "IP")] ip: String }`, plus a `parse_devices(xml: &str) -> Result<DeviceList, quick_xml::DeError>` function.
+ src/parser.rs (same file) a #[test] that parses a sample XML string and asserts 2 devices with the right fields.
+ src/main.rs (same file, in a #[cfg(test)] mod) an integration-ish test that spins up an in-process mock HTTP server (use `std::net::TcpListener` + a tiny thread that writes a raw HTTP response — no extra dep needed), fetches from it, and asserts the parsed output. Keep it simple: the mock can return 200 + the XML directly (skip Digest in the test if it complicates things — the bench's scoring gate handles the Digest path against the harness's mock).
+
+### Constraints
+ - Use `reqwest` with `default-features = false, features = ["rustls", "charset"]`. No OpenSSL.
+ - Use `quick-xml` with `serde` for parsing (the `serde` feature on quick-xml).
+ - Use `tokio` with `features = ["rt-multi-thread", "macros"]` for the async runtime.
+ - For Digest auth: the simplest approach is to use the `digest_auth` crate (add it as a dep) to compute the response from the 401's WWW-Authenticate header, then retry with the Authorization header. OR if that's too much, just do Basic auth for the fetch and note Digest as a TODO — the scoring harness's mock accepts Basic too for this bench. **BUT** the plan's intent is Digest; try it.
+ - Arg parsing: manual (std::env::args) is fine — no need for clap.
+ - The binary must exit 0 on success, non-zero on error.
+
+## Steps
+1. Write Cargo.toml with the deps.
+2. Write src/parser.rs (the XML types + parse function + a unit test).
+3. Write src/main.rs (arg parsing, async fetch, parse, print).
+4. `cargo build` — fix compile errors (expect some Rust borrow-checker friction; that's normal).
+5. `cargo test` — the parser unit test + the integration test must pass.
+6. Optionally test the binary against a real URL (the harness will test it against a mock).
+
+## Acceptance (harness-verified)
+1. `cargo build` exits 0.
+2. `cargo test` exits 0.
+3. The binary, pointed at the harness's mock HTTP server (serves the XML above behind Digest auth at a URL the harness provides), fetches it and prints `1|Camera-Front|10.0.0.50` and `2|Camera-Back|10.0.0.51` on separate lines.
+
+## Out of scope
+Tauri (no GUI), keyring (no OS keychain), ONVIF, ISAPI-specific endpoints, batch operations, config files. This is the smallest real Rust CLI with reqwest + quick-xml + async — a working fetch+parse CLI.
+
+## Commit
+You do not commit; the harness owns the work dir.
+"""
+
+STARTER = {
+ "Cargo.toml": '''[package]
+name = "device-list"
+version = "0.1.0"
+edition = "2021"
+
+[dependencies]
+''',
+ "src/main.rs": "fn main() {}\n",
+}
+
+REFERENCE = {
+ "Cargo.toml": '''[package]
+name = "device-list"
+version = "0.1.0"
+edition = "2021"
+
+[dependencies]
+tokio = { version = "1", features = ["rt-multi-thread", "macros"] }
+reqwest = { version = "0.13", default-features = false, features = ["rustls", "charset"] }
+quick-xml = { version = "0.36", features = ["serialize"] }
+serde = { version = "1", features = ["derive"] }
+digest_auth = "0.3"
+''',
+ "src/parser.rs": '''use serde::Deserialize;
+use quick_xml::de::from_str;
+
+#[derive(Debug, Deserialize)]
+pub struct DeviceList {
+ #[serde(rename = "Device")]
+ pub devices: Vec<Device>,
+}
+
+#[derive(Debug, Deserialize)]
+pub struct Device {
+ #[serde(rename = "ID")]
+ pub id: String,
+ #[serde(rename = "Name")]
+ pub name: String,
+ #[serde(rename = "IP")]
+ pub ip: String,
+}
+
+pub fn parse_devices(xml: &str) -> Result<DeviceList, quick_xml::de::DeError> {
+ from_str(xml)
+}
+
+#[cfg(test)]
+mod tests {
+ use super::*;
+ #[test]
+ fn parse_two_devices() {
+ let xml = r#"<?xml version="1.0" encoding="UTF-8"?>
+<DeviceList>
+ <Device><ID>1</ID><Name>Camera-Front</Name><IP>10.0.0.50</IP></Device>
+ <Device><ID>2</ID><Name>Camera-Back</Name><IP>10.0.0.51</IP></Device>
+</DeviceList>"#;
+ let list = parse_devices(xml).unwrap();
+ assert_eq!(list.devices.len(), 2);
+ assert_eq!(list.devices[0].id, "1");
+ assert_eq!(list.devices[0].name, "Camera-Front");
+ assert_eq!(list.devices[0].ip, "10.0.0.50");
+ assert_eq!(list.devices[1].name, "Camera-Back");
+ }
+}
+''',
+ "src/main.rs": '''mod parser;
+
+use std::env;
+use std::process;
+
+fn parse_args() -> (String, String, String) {
+ let mut url = String::new();
+ let mut user = String::new();
+ let mut pass = String::new();
+ let mut args = env::args().skip(1);
+ while let Some(arg) = args.next() {
+ match arg.as_str() {
+ "--url" => { url = args.next().unwrap_or_default(); }
+ "--user" => { user = args.next().unwrap_or_default(); }
+ "--pass" => { pass = args.next().unwrap_or_default(); }
+ _ => {}
+ }
+ }
+ (url, user, pass)
+}
+
+async fn fetch_with_digest(url: &str, user: &str, pass: &str) -> Result<String, Box<dyn std::error::Error>> {
+ let client = reqwest::Client::builder()
+ .build()?;
+ // First request — get the 401 + WWW-Authenticate.
+ let resp = client.get(url).send().await?;
+ if resp.status() == 401 {
+ let www_auth = resp.headers().get("www-authenticate")
+ .and_then(|v| v.to_str().ok())
+ .ok_or("no WWW-Authenticate header")?;
+ // Parse the challenge + compute the digest response.
+ let mut prompt = digest_auth::parse(www_auth)?;
+ let ctx = digest_auth::AuthContext::new(user, pass, url);
+ let auth_header = prompt.respond(&ctx)?;
+ let resp2 = client.get(url)
+ .header("Authorization", auth_header.to_header_string())
+ .send().await?;
+ let body = resp2.text().await?;
+ Ok(body)
+ } else {
+ let body = resp.text().await?;
+ Ok(body)
+ }
+}
+
+#[tokio::main]
+async fn main() {
+ let (url, user, pass) = parse_args();
+ if url.is_empty() {
+ eprintln!("usage: device-list --url <URL> --user <user> --pass <pass>");
+ process::exit(1);
+ }
+ let xml = match fetch_with_digest(&url, &user, &pass).await {
+ Ok(x) => x,
+ Err(e) => { eprintln!("fetch error: {e}"); process::exit(1); }
+ };
+ let list = match parser::parse_devices(&xml) {
+ Ok(l) => l,
+ Err(e) => { eprintln!("parse error: {e}"); process::exit(1); }
+ };
+ for d in list.devices {
+ println!("{}|{}|{}", d.id, d.name, d.ip);
+ }
+}
+''',
+}
+
+def _materialize(files, dest):
+ shutil.rmtree(dest, ignore_errors=True)
+ os.makedirs(dest, exist_ok=True)
+ for path, content in files.items():
+ fp = os.path.join(dest, path)
+ os.makedirs(os.path.dirname(fp), exist_ok=True)
+ with open(fp, "w") as f: f.write(content)
+ return dest
+
+def _run(cmd, cwd, timeout=300):
+ try:
+ p = subprocess.run(cmd, cwd=cwd, shell=True, capture_output=True, text=True, timeout=timeout)
+ except subprocess.TimeoutExpired:
+ return "TIMEOUT"
+ return f"exit={p.returncode}\n{(p.stdout + p.stderr).strip()}"
+
+def _has_real_app(work_dir):
+ main_rs = os.path.join(work_dir, "src", "main.rs")
+ if not os.path.isfile(main_rs): return False
+ with open(main_rs) as f:
+ src = f.read()
+ return "tokio" in src and "reqwest" in src and "parse_devices" in src
+
+class _DigestMockServer:
+ """A tiny HTTP server that serves the XML behind Digest auth."""
+ XML_BODY = (
+ '<?xml version="1.0" encoding="UTF-8"?>\n'
+ '<DeviceList>\n'
+ ' <Device><ID>1</ID><Name>Camera-Front</Name><IP>10.0.0.50</IP></Device>\n'
+ ' <Device><ID>2</ID><Name>Camera-Back</Name><IP>10.0.0.51</IP></Device>\n'
+ '</DeviceList>'
+ )
+ def __init__(self):
+ self.sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
+ self.sock.bind(("127.0.0.1", 0))
+ self.port = self.sock.getsockname()[1]
+ self.sock.listen(4)
+ self.thread = threading.Thread(target=self._serve, daemon=True)
+ self.running = True
+ def start(self): self.thread.start()
+ def stop(self): self.running = False; self.sock.close()
+ def _serve(self):
+ while self.running:
+ try:
+ conn, _ = self.sock.accept()
+ except OSError:
+ break
+ try:
+ data = conn.recv(4096).decode(errors="replace")
+ # If no Authorization header → 401 with Digest challenge.
+ if "Authorization:" not in data and "authorization:" not in data:
+ resp = (
+ "HTTP/1.1 401 Unauthorized\r\n"
+ 'WWW-Authenticate: Digest realm="test", qop="auth", nonce="abc123", opaque=""\r\n'
+ "Content-Length: 0\r\n"
+ "\r\n"
+ )
+ conn.sendall(resp.encode())
+ else:
+ body = self.XML_BODY.encode()
+ resp = (
+ "HTTP/1.1 200 OK\r\n"
+ "Content-Type: application/xml\r\n"
+ f"Content-Length: {len(body)}\r\n"
+ "\r\n"
+ )
+ conn.sendall(resp.encode() + body)
+ except Exception:
+ pass
+ finally:
+ conn.close()
+
+def gates(work_dir):
+ g = {}
+ if not _has_real_app(work_dir):
+ return {"build": False, "test": False, "fetches": False}
+ b = _run("cargo build", work_dir, timeout=600)
+ g["build"] = b.startswith("exit=0")
+ t = _run("cargo test", work_dir, timeout=600)
+ g["test"] = t.startswith("exit=0")
+ # fetches gate: run the binary against the mock server
+ binpath = os.path.join(work_dir, "target", "debug", "device-list")
+ if not os.path.isfile(binpath):
+ g["fetches"] = False; return g
+ srv = _DigestMockServer(); srv.start()
+ try:
+ url = f"http://127.0.0.1:{srv.port}/"
+ try:
+ p = subprocess.run([binpath, "--url", url, "--user", "admin", "--pass", "secret"],
+ capture_output=True, text=True, timeout=30)
+ out = p.stdout.strip()
+ g["fetches"] = ("1|Camera-Front|10.0.0.50" in out and
+ "2|Camera-Back|10.0.0.51" in out)
+ if not g["fetches"]:
+ g["_fetch_detail"] = f"exit={p.returncode} stdout={out[:200]} stderr={p.stderr[:200]}"
+ except Exception as e:
+ g["fetches"] = False
+ g["_fetch_detail"] = repr(e)
+ finally:
+ srv.stop()
+ return g
+
+def validate(work_dir):
+ _materialize(REFERENCE, dest=work_dir)
+ print("cargo build...")
+ b = _run("cargo build", work_dir, timeout=600)
+ print(f"build: {'OK' if b.startswith('exit=0') else 'FAIL'}\n{b[-400:]}")
+ print("cargo test...")
+ t = _run("cargo test", work_dir, timeout=600)
+ print(f"test: {'OK' if t.startswith('exit=0') else 'FAIL'}\n{t[-400:]}")
+ g = gates(work_dir)
+ print(f"gates: {g}")
+ ok = all(v for k, v in g.items() if not k.startswith("_"))
+ print(f"-> {'OK' if ok else 'BAD'}")
+ return ok
+
+if __name__ == "__main__":
+ wd = os.path.join(HERE, "..", "..", "work", "v2hard_validate_rust_cli")
+ validate(wd)
\ No newline at end of file
diff --git a/bench/tasks/multitask/ts_next.py b/bench/tasks/multitask/ts_next.py
new file mode 100644
index 0000000..97ce635
--- /dev/null
+++ b/bench/tasks/multitask/ts_next.py
@@ -0,0 +1,380 @@
+#!/usr/bin/env python3
+"""
+Task: ts_next — a Next.js app with Prisma + API route + Zod validation.
+
+Models a Next.js app with Prisma + API route + Zod validation in miniature:
+a Next.js App Router app with a Prisma schema (2 models), an API route that
+does a CRUD operation with Zod input validation, and a server-rendered page.
+Uses SQLite as the Prisma datasource (so no external Postgres is needed —
+`@prisma/adapter-better-sqlite3` or Prisma's built-in sqlite provider). The
+"real product" = a dev server that responds on `/` (the page) and
+`/api/items` (the API route: GET lists, POST creates with Zod validation,
+invalid input returns 400).
+"""
+import os, sys, subprocess, time, shutil, signal, socket, urllib.request, urllib.parse, json
+
+from bench.core import tools
+
+HERE = os.path.dirname(os.path.abspath(__file__))
+
+TASK_NAME = "ts_next"
+
+PLAN = """\
+# Task — Build a Next.js app with Prisma + API route + Zod validation
+
+## Goal
+`npm install` succeeds, `npx tsc --noEmit` passes (typecheck), `npm test`
+passes (a Vitest unit test of the Zod schema), and the dev server responds:
+`GET /` returns the HTML page, `POST /api/items` with valid JSON creates an
+item (201), `POST /api/items` with invalid JSON returns 400, `GET /api/items`
+returns the list. Mirrors a Next.js app with Prisma + Zod validation: Next.js
+App Router + Prisma + Zod-validated API routes + server-rendered page.
+
+## Prerequisites
+Starter scaffold in the work dir:
+ - `package.json` (name: "items-app", no deps yet)
+ - `tsconfig.json` (basic strict config)
+ - `next.config.ts` (minimal)
+
+## Spec — the app to build
+A Next.js 15+ App Router app with a Prisma-backed CRUD API for "items" (a
+simple name + description entity). One page (`/`) that server-renders the
+item list + a create form. One API route (`POST /api/items` + `GET /api/items`)
+with Zod input validation on POST. Prisma with SQLite (file-based, no external
+DB needed).
+
+### Prisma schema (prisma/schema.prisma)
+ generator client { provider = "prisma-client-js" }
+ datasource db { provider = "sqlite"; url = "file:./dev.db" }
+ model Item {
+ id String @id @default(cuid())
+ name String
+ description String
+ createdAt DateTime @default(now())
+ }
+
+### Files to produce
+ package.json deps: next (15+), react, react-dom; devDeps: typescript, @types/react, @types/node, prisma, @prisma/client, zod, vitest, @vitejs/plugin-react, jsdom. Scripts: dev, build, test, typecheck (tsc --noEmit).
+ tsconfig.json strict, target ES2022, moduleResolution bundler, jsx preserve, lib [dom, dom.iterable, esnext], paths for ~/* -> src/*.
+ next.config.ts minimal (no special config needed).
+ prisma/schema.prisma the schema above.
+ src/lib/prisma.ts export a singleton PrismaClient instance.
+ src/lib/validation.ts export `createItemSchema = z.object({ name: z.string().min(1).max(100), description: z.string().min(1).max(500) })` and `type CreateItemInput = z.infer<typeof createItemSchema>`.
+ src/lib/validation.test.ts a Vitest test: valid input parses, empty name fails, name > 100 chars fails, empty description fails.
+ src/app/layout.tsx root layout (html + body, minimal).
+ src/app/page.tsx server component: fetches items via prisma, renders a list + a client-side create form (the form POSTs to /api/items as JSON; on success, refresh the page — a simple form with onSubmit that fetch + router.refresh() is fine, or even simpler: a plain HTML form that posts to /api/items and redirects back).
+ src/app/api/items/route.ts POST: parse JSON body, validate with createItemSchema, on failure return 400 + errors; on success prisma.item.create + return 201 + the created item. GET: prisma.item.findMany + return 200 + the list.
+
+### Constraints
+ - Use Prisma with the **sqlite** provider (file-based `dev.db` — no external Postgres needed for this bench).
+ - Use `zod` for API input validation — this is the security boundary (the standard pattern for a Next.js app with Prisma + API route + Zod validation).
+ - The API route must return **400** with the Zod errors on invalid input, **201** + the created item on success.
+ - The page (`/`) must server-render the item list (read from Prisma in the server component).
+ - Use Next.js App Router (`src/app/` layout). NOT the pages router.
+ - Run `npx prisma generate` + `npx prisma db push` after writing the schema so the client is generated + the DB is created.
+
+## Steps
+1. Write package.json + tsconfig.json + next.config.ts.
+2. `npm install` (this pulls next, react, prisma, zod, vitest — takes a minute).
+3. Write prisma/schema.prisma. Run `npx prisma generate && npx prisma db push`.
+4. Write src/lib/{prisma.ts, validation.ts, validation.test.ts}.
+5. Write src/app/{layout.tsx, page.tsx, api/items/route.ts}.
+6. `npm run typecheck` (tsc --noEmit) — fix type errors.
+7. `npm test` (vitest) — the validation test must pass.
+8. Optionally start the dev server briefly to confirm it responds.
+
+## Acceptance (harness-verified)
+1. `npm install` exits 0.
+2. `npx tsc --noEmit` exits 0 (typecheck clean).
+3. `npm test` exits 0 (vitest passes).
+4. Dev server starts; `GET /` returns 200 + HTML containing an item list or a form; `POST /api/items` with valid JSON (`{"name":"Test","description":"desc"}`) returns 201; `POST /api/items` with invalid JSON (`{"name":"","description":""}`) returns 400; `GET /api/items` returns 200 + a JSON array.
+
+## Out of scope
+Auth/Auth.js, sessions, CSRF, Postgres, Docker, Tailwind, RBAC, middleware, themes. This is the smallest real Next.js app with Prisma + API route + Zod validation — a working CRUD app with Zod-validated API + server-rendered page.
+
+## Commit
+You do not commit; the harness owns the work dir.
+"""
+
+STARTER = {
+ "package.json": json.dumps({
+ "name": "items-app",
+ "version": "0.1.0",
+ "private": True,
+ "scripts": {"dev": "next dev", "build": "next build", "test": "vitest run", "typecheck": "tsc --noEmit"},
+ "dependencies": {},
+ "devDependencies": {}
+ }, indent=2),
+ "tsconfig.json": json.dumps({
+ "compilerOptions": {
+ "target": "ES2022", "lib": ["dom", "dom.iterable", "esnext"],
+ "allowJs": True, "skipLibCheck": True, "strict": True,
+ "noEmit": True, "esModuleInterop": True, "module": "esnext",
+ "moduleResolution": "bundler", "resolveJsonModule": True,
+ "isolatedModules": True, "jsx": "preserve", "incremental": True,
+ "plugins": [{"name": "next"}],
+ "paths": {"~/*": ["./src/*"]}
+ },
+ "include": ["next-env.d.ts", "**/*.ts", "**/*.tsx", ".next/types/**/*.ts"],
+ "exclude": ["node_modules"]
+ }, indent=2),
+ "next.config.ts": "import type { NextConfig } from 'next'\n\nconst config: NextConfig = {}\n\nexport default config\n",
+}
+
+# The reference is minimal but complete. Prisma + sqlite is the trickiest part
+# to get self-contained — we use the prisma sqlite provider with a file-based db.
+REFERENCE = {
+ "package.json": json.dumps({
+ "name": "items-app",
+ "version": "0.1.0",
+ "private": True,
+ "scripts": {"dev": "next dev", "build": "next build", "start": "next start", "test": "vitest run", "typecheck": "tsc --noEmit"},
+ "dependencies": {
+ "next": "^15",
+ "react": "^19",
+ "react-dom": "^19",
+ "@prisma/client": "^6",
+ "zod": "^3",
+ },
+ "devDependencies": {
+ "typescript": "^5",
+ "@types/react": "^19",
+ "@types/node": "^22",
+ "prisma": "^6",
+ "vitest": "^2",
+ "@vitejs/plugin-react": "^4",
+ "jsdom": "^25",
+ }
+ }, indent=2),
+ "tsconfig.json": STARTER["tsconfig.json"],
+ "next.config.ts": STARTER["next.config.ts"],
+ "vitest.config.ts": '''import { defineConfig } from 'vitest/config'
+export default defineConfig({
+ test: { environment: 'node', include: ['src/**/*.test.ts'] },
+})
+''',
+ "prisma/schema.prisma": '''generator client {
+ provider = "prisma-client-js"
+}
+
+datasource db {
+ provider = "sqlite"
+ url = "file:./dev.db"
+}
+
+model Item {
+ id String @id @default(cuid())
+ name String
+ description String
+ createdAt DateTime @default(now())
+}
+''',
+ "src/lib/prisma.ts": '''import { PrismaClient } from '@prisma/client'
+const globalForPrisma = globalThis as unknown as { prisma: PrismaClient | undefined }
+export const prisma = globalForPrisma.prisma ?? new PrismaClient()
+if (process.env.NODE_ENV !== 'production') globalForPrisma.prisma = prisma
+''',
+ "src/lib/validation.ts": '''import { z } from 'zod'
+
+export const createItemSchema = z.object({
+ name: z.string().min(1).max(100),
+ description: z.string().min(1).max(500),
+})
+
+export type CreateItemInput = z.infer<typeof createItemSchema>
+''',
+ "src/lib/validation.test.ts": '''import { describe, it, expect } from 'vitest'
+import { createItemSchema } from './validation'
+
+describe('createItemSchema', () => {
+ it('accepts valid input', () => {
+ const result = createItemSchema.safeParse({ name: 'Test', description: 'desc' })
+ expect(result.success).toBe(true)
+ })
+ it('rejects empty name', () => {
+ const result = createItemSchema.safeParse({ name: '', description: 'desc' })
+ expect(result.success).toBe(false)
+ })
+ it('rejects name > 100 chars', () => {
+ const result = createItemSchema.safeParse({ name: 'x'.repeat(101), description: 'desc' })
+ expect(result.success).toBe(false)
+ })
+ it('rejects empty description', () => {
+ const result = createItemSchema.safeParse({ name: 'Test', description: '' })
+ expect(result.success).toBe(false)
+ })
+})
+''',
+ "src/app/layout.tsx": '''import type { ReactNode } from 'react'
+
+export default function RootLayout({ children }: { children: ReactNode }) {
+ return (
+ <html lang="en">
+ <body>{children}</body>
+ </html>
+ )
+}
+''',
+ "src/app/page.tsx": '''import { prisma } from '~/lib/prisma'
+
+export default async function Home() {
+ const items = await prisma.item.findMany({ orderBy: { createdAt: 'desc' } })
+ return (
+ <main>
+ <h1>Items</h1>
+ <form action="/api/items" method="POST" id="createForm">
+ <input name="name" placeholder="Name" required />
+ <input name="description" placeholder="Description" required />
+ <button type="submit">Add</button>
+ </form>
+ <ul>
+ {items.map((item) => (
+ <li key={item.id}>{item.name}: {item.description}</li>
+ ))}
+ </ul>
+ </main>
+ )
+}
+''',
+ "src/app/api/items/route.ts": '''import { NextRequest, NextResponse } from 'next/server'
+import { prisma } from '~/lib/prisma'
+import { createItemSchema } from '~/lib/validation'
+
+export async function GET() {
+ const items = await prisma.item.findMany({ orderBy: { createdAt: 'desc' } })
+ return NextResponse.json(items)
+}
+
+export async function POST(request: NextRequest) {
+ let body: unknown
+ try {
+ body = await request.json()
+ } catch {
+ return NextResponse.json({ error: 'invalid JSON' }, { status: 400 })
+ }
+ const parsed = createItemSchema.safeParse(body)
+ if (!parsed.success) {
+ return NextResponse.json({ error: 'validation failed', issues: parsed.error.issues }, { status: 400 })
+ }
+ const item = await prisma.item.create({ data: parsed.data })
+ return NextResponse.json(item, { status: 201 })
+}
+''',
+ "next-env.d.ts": '/// <reference types="next" />\n/// <reference types="next/image-types/global" />\n',
+ ".gitignore": "node_modules\ndev.db*\n.next\n",
+}
+
+def _materialize(files, dest):
+ shutil.rmtree(dest, ignore_errors=True)
+ os.makedirs(dest, exist_ok=True)
+ for path, content in files.items():
+ fp = os.path.join(dest, path)
+ os.makedirs(os.path.dirname(fp), exist_ok=True)
+ with open(fp, "w") as f: f.write(content)
+ return dest
+
+def _run(cmd, cwd, timeout=600, env=None):
+ try:
+ p = subprocess.run(cmd, cwd=cwd, shell=True, capture_output=True, text=True, timeout=timeout, env=env)
+ except subprocess.TimeoutExpired:
+ return "TIMEOUT"
+ return f"exit={p.returncode}\n{(p.stdout + p.stderr).strip()}"
+
+def _free_port():
+ s = socket.socket(socket.AF_INET, socket.SOCK_STREAM); s.bind(("", 0))
+ p = s.getsockname()[1]; s.close(); return p
+
+def _has_real_app(work_dir):
+ if not os.path.isfile(os.path.join(work_dir, "src", "app", "page.tsx")): return False
+ if not os.path.isfile(os.path.join(work_dir, "src", "app", "api", "items", "route.ts")): return False
+ if not os.path.isfile(os.path.join(work_dir, "prisma", "schema.prisma")): return False
+ return True
+
+def gates(work_dir):
+ g = {}
+ if not _has_real_app(work_dir):
+ return {"install": False, "typecheck": False, "test": False, "serves": False}
+ # install
+ g["install"] = _run("npm install", work_dir, timeout=300).startswith("exit=0")
+ if not g["install"]:
+ g["typecheck"] = False; g["test"] = False; g["serves"] = False; return g
+ # prisma generate + db push
+ _run("npx prisma generate", work_dir, timeout=120)
+ _run("npx prisma db push", work_dir, timeout=120)
+ # typecheck
+ g["typecheck"] = _run("npx tsc --noEmit", work_dir, timeout=120).startswith("exit=0")
+ # test
+ g["test"] = _run("npm test", work_dir, timeout=120).startswith("exit=0")
+ # serves: start dev server, test the API + page
+ port = _free_port()
+ env = dict(os.environ, PORT=str(port))
+ try:
+ proc = subprocess.Popen(["npx", "next", "dev", "-p", str(port)], cwd=work_dir,
+ env=env, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
+ except Exception:
+ g["serves"] = False; return g
+ try:
+ # wait for the dev server to be ready (can take 10-30s on first compile)
+ for _ in range(60):
+ try:
+ with urllib.request.urlopen(f"http://127.0.0.1:{port}/", timeout=2) as r:
+ if r.status == 200:
+ page_html = r.read().decode(errors="replace")
+ break
+ except Exception:
+ time.sleep(1)
+ else:
+ g["serves"] = False; return g
+ page_ok = "<h1>Items</h1>" in page_html or "Items" in page_html
+ # POST valid
+ base = f"http://127.0.0.1:{port}"
+ valid_data = json.dumps({"name": "BenchItem", "description": "from harness"}).encode()
+ req = urllib.request.Request(f"{base}/api/items", data=valid_data,
+ headers={"Content-Type": "application/json"}, method="POST")
+ post_ok = False
+ try:
+ with urllib.request.urlopen(req, timeout=15) as r:
+ post_ok = r.status == 201
+ except urllib.error.HTTPError as e:
+ post_ok = False
+ # POST invalid (should 400)
+ invalid_data = json.dumps({"name": "", "description": ""}).encode()
+ req2 = urllib.request.Request(f"{base}/api/items", data=invalid_data,
+ headers={"Content-Type": "application/json"}, method="POST")
+ reject_ok = False
+ try:
+ with urllib.request.urlopen(req2, timeout=15) as r:
+ reject_ok = False # should have been 400
+ except urllib.error.HTTPError as e:
+ reject_ok = e.code == 400
+ # GET list
+ get_ok = False
+ try:
+ with urllib.request.urlopen(f"{base}/api/items", timeout=15) as r:
+ items = json.loads(r.read().decode())
+ get_ok = isinstance(items, list) and any(i.get("name") == "BenchItem" for i in items)
+ except Exception:
+ pass
+ g["serves"] = page_ok and post_ok and reject_ok and get_ok
+ if not g["serves"]:
+ g["_serves_detail"] = f"page={page_ok} post={post_ok} reject400={reject_ok} get={get_ok}"
+ finally:
+ proc.send_signal(signal.SIGTERM)
+ try: proc.wait(timeout=10)
+ except subprocess.TimeoutExpired: proc.kill()
+ return g
+
+def validate(work_dir):
+ _materialize(REFERENCE, dest=work_dir)
+ print("npm install...")
+ g = gates(work_dir)
+ print(f"gates: { {k:v for k,v in g.items() if not k.startswith('_')} }")
+ if not g.get("serves") and "_serves_detail" in g:
+ print(f" serves detail: {g['_serves_detail']}")
+ ok = all(v for k, v in g.items() if not k.startswith("_"))
+ print(f"-> {'OK' if ok else 'BAD'}")
+ return ok
+
+if __name__ == "__main__":
+ wd = os.path.join(HERE, "..", "..", "..", "work", "validate_ts_next")
+ validate(wd)
\ No newline at end of file
diff --git a/bench/tasks/needle.py b/bench/tasks/needle.py
new file mode 100644
index 0000000..8ee5636
--- /dev/null
+++ b/bench/tasks/needle.py
@@ -0,0 +1,256 @@
+#!/usr/bin/env python3
+"""
+needle.py — repeatable needle-in-a-haystack battery for quant/lever A/Bs.
+
+Replaces the ad-hoc one-off probes (EMERALD-7742 @66.5k etc.) with a committed
+instrument. Measures retrieval quality at depth AND speed telemetry per depth,
+so a candidate quant can be compared against the incumbent on one run.
+
+Method (mirrors halogen-flash's published battery + our 2026-09-03 probes):
+ - Filler text is neutral prose paragraphs (seeded, deterministic per depth).
+ - A needle is a single sentence with a random code token: "The maintenance
+ code for the {place} is {CODE}-{NNNN}." Spliced into filler at a target
+ depth, the document continues, then asks for the code. Exact string match
+ on the retrieved code. Temp 0 (greedy) — quant differences only, no
+ sampling noise.
+ - Speed: prefill tok/s (prompt_eval from usage), decode tok/s (eval), wall.
+ - Multiple needles per depth, different insertion positions -> a distribution,
+ not a single anecdote. Also: one NEAR-MISS needle per depth (a second code
+ exists elsewhere in the doc; the model must return the RIGHT one) to catch
+ confabulation.
+
+Usage:
+ python bench.py needle run # full battery
+ python bench.py needle run --depths 1000,16000 # quick subset
+ python bench.py needle run --needles 3 --repeats 2 # smoke config
+
+Env (via .env / bench.core.client): ENDPOINT, MODEL, API_KEY, MAX_TOKENS,
+TEMPERATURE, HTTP_TIMEOUT. Plus:
+ LABEL tag for the JSON/report (default MODEL)
+ N_PREDICT max answer tokens (default 512; reasoning models think first)
+
+Output: results/needle-<label>.json + printed table. Exit 0 if every needle
+retrieved exact (any miss -> exit 1, so CI-style gating can catch it).
+"""
+import json
+import os
+import random
+import sys
+import time
+
+from bench.core import client
+from bench.core.reporting import write_json, RESULTS_DIR
+
+LABEL = os.environ.get("LABEL", client.MODEL)
+N_PREDICT = int(os.environ.get("N_PREDICT", "512"))
+
+DEPTH_GRID = [1000, 4000, 16000, 32000, 64000]
+
+# Deterministic per-depth filler. Neutral office-log prose, no code tokens.
+FILLER_SENTENCES = [
+ "The quarterly inventory review proceeded without any material discrepancies noted by the floor staff.",
+ "Facilities confirmed the heating schedule for the west corridor will revert to the winter profile next week.",
+ "The procurement team circulated updated vendor terms for the spring contract cycle.",
+ "Attendance at the all-hands was recorded at eighty-four percent, which is typical for this month.",
+ "The backup generator passed its monthly load test and the transfer switch responded within specification.",
+ "Reception logged three courier deliveries before noon and routed each to the appropriate department.",
+ "The finance office reminded everyone that expense submissions close on the fifteenth of the month.",
+ "A scheduled patch window for the document management system was announced for Thursday evening.",
+ "The landscaping crew completed the seasonal pruning along the south entrance without incident.",
+ "Warehouse slotting for the new product line was finalized and communicated to the picking team.",
+ "The safety committee published minutes from its monthly walkthrough of the loading dock area.",
+ "IT reported that the printer fleet firmware update completed on all but two devices.",
+]
+
+PLACES = ["north stairwell", "server room", "loading dock", "rooftop hatch", "maintenance office",
+ "freight elevator", "parking garage level two", "electrical closet", "boiler room", "west fire exit"]
+
+CODE_WORDS = ["EMERALD", "COPPER", "GRANITE", "LANTERN", "HARBOR", "MERIDIAN", "SAPPHIRE", "THISTLE"]
+
+
+def _para(rng):
+ return " ".join(rng.choice(FILLER_SENTENCES) for _ in range(4)) + "\n\n"
+
+
+def build_doc(target_tokens, needle, needle_at_frac, rng):
+ """Filler document with `needle` spliced at needle_at_frac of target size."""
+ parts, approx = [], 0
+ while approx < target_tokens:
+ parts.append(_para(rng))
+ approx += 64 # ~64 tok/paragraph (3 sentences avg)
+ doc_parts_before = []
+ tok_before = int(target_tokens * needle_at_frac)
+ idx = max(1, tok_before // 64)
+ head = "".join(parts[:idx])
+ tail = "".join(parts[idx:])
+ return head + needle + "\n\n" + tail
+
+
+def make_needle(rng):
+ place = rng.choice(PLACES)
+ word = rng.choice(CODE_WORDS)
+ num = rng.randint(1000, 9999)
+ code = f"{word}-{num}"
+ sentence = f"Maintenance note: the access code for the {place} is {code} until the end of the quarter."
+ return sentence, code, place
+
+
+def make_decoy(rng):
+ place = rng.choice(PLACES)
+ word = rng.choice(CODE_WORDS)
+ num = rng.randint(1000, 9999)
+ sentence = (f"Maintenance note: the access code for the {place} was {word}-{num} "
+ f"until it was rotated last month.")
+ return sentence
+
+
+def call_model(prompt, n_predict=N_PREDICT, max_retries=1):
+ """Delegate to bench.core.client.call_model and extract needle-relevant telemetry.
+
+ needle_bench needs the `timings` field from llama-server responses for
+ prefill/decode tok/s, so we keep the local parsing of resp.get("timings")
+ and resp.get("usage") on top of the shared client.
+ """
+ last_err = None
+ for attempt in range(max_retries + 1):
+ try:
+ resp, wall = client.call_model(
+ [{"role": "user", "content": prompt}],
+ max_tokens=n_predict,
+ temperature=0.0,
+ http_timeout=1800,
+ )
+ usage = resp.get("usage", {})
+ timings = resp.get("timings", {})
+ ctok = usage.get("completion_tokens", 0)
+ ptok = usage.get("prompt_tokens", 0)
+ # llama-server reports rates top-level in `timings` (prompt_per_second /
+ # predicted_per_second); OpenAI-style usage has no rate fields.
+ pd = timings.get("prompt_per_second")
+ ev = timings.get("predicted_per_second")
+ msg = resp.get("choices", [{}])[0].get("message", {})
+ content = msg.get("content") or ""
+ reasoning = msg.get("reasoning_content") or ""
+ return {"content": content, "reasoning": reasoning, "ptok": ptok, "ctok": ctok,
+ "prefill_ps": pd, "decode_ps": ev, "wall": wall,
+ "finish": resp.get("choices", [{}])[0].get("finish_reason", "")}
+ except Exception as e:
+ last_err = str(e)[:300]
+ time.sleep(2)
+ return {"error": last_err}
+
+
+def run_battery(needles, depths, repeats, decoy_probe=True):
+ results = []
+ for depth in depths:
+ for rep in range(repeats):
+ # ---- normal needles at varying positions
+ for n in range(needles):
+ rng = random.Random(f"{depth}-{rep}-{n}")
+ sentence, code, place = make_needle(rng)
+ pos_frac = (n + 0.5) / needles # spread across the doc
+ doc = build_doc(depth, sentence, pos_frac, rng)
+ prompt = (doc + "\n\nQUESTION: According to the maintenance notes above, "
+ "what is the access code for the " + place +
+ "? Answer with the code exactly as written, nothing else.")
+ r = call_model(prompt)
+ ok = code in r.get("content", "")
+ results.append({"depth": depth, "rep": rep, "needle_idx": n,
+ "code": code, "exact": ok, **{k: r.get(k) for k in
+ ("ptok", "ctok", "prefill_ps", "decode_ps", "wall", "error")}})
+ _print_row(results[-1])
+ # ---- decoy probe: right needle + rotated-out old code in the same doc
+ if decoy_probe:
+ rng = random.Random(f"{depth}-{rep}-decoy")
+ sentence, code, place = make_needle(rng)
+ decoy_sent = make_decoy(rng)
+ rng2 = random.Random(f"{depth}-{rep}-decoy-tail")
+ head = build_doc(int(depth * 0.4), sentence, 0.5, rng)
+ tail = build_doc(int(depth * 0.6), decoy_sent, 0.5, rng2)
+ prompt = (head + tail + "\n\nQUESTION: According to the maintenance notes above, "
+ "what is the CURRENT access code for the " + place +
+ "? Answer with the code exactly as written, nothing else.")
+ r = call_model(prompt)
+ ok = code in r.get("content", "")
+ results.append({"depth": depth, "rep": rep, "needle_idx": "decoy",
+ "code": code, "exact": ok, **{k: r.get(k) for k in
+ ("ptok", "ctok", "prefill_ps", "decode_ps", "wall", "error")}})
+ _print_row(results[-1])
+ return results
+
+
+def _print_row(r):
+ if r.get("error"):
+ print(f" depth={r['depth']:>6} rep={r['rep']} needle={r['needle_idx']!s:>5} ERROR: {r['error']}")
+ return
+ print(f" depth={r['depth']:>6} rep={r['rep']} needle={r['needle_idx']!s:>5} "
+ f"{'PASS' if r['exact'] else 'MISS'} ptok={r.get('ptok', 0):>6} prefill={r.get('prefill_ps') or 0:>7.1f} "
+ f"decode={r.get('decode_ps') or 0:>5.1f} wall={r.get('wall', 0):>6.1f}s")
+
+
+def summarize(results):
+ by_depth = {}
+ for r in results:
+ d = by_depth.setdefault(r["depth"], {"pass": 0, "total": 0, "walls": [], "prefills": [], "decodes": []})
+ d["total"] += 1
+ if r.get("exact"):
+ d["pass"] += 1
+ if not r.get("error"):
+ if r.get("wall") is not None:
+ d["walls"].append(r["wall"])
+ if r.get("prefill_ps"):
+ d["prefills"].append(r["prefill_ps"])
+ if r.get("decode_ps"):
+ d["decodes"].append(r["decode_ps"])
+ print("\n== Summary ==")
+ print(f"{'depth':>8} {'retrieved':>10} {'wall mean':>10} {'prefill t/s':>12} {'decode t/s':>11}")
+ for depth in sorted(by_depth):
+ d = by_depth[depth]
+ walls = d["walls"]
+ prefills = [p for p in d["prefills"] if p]
+ decodes = d["decodes"]
+
+ def mean(xs):
+ return sum(xs) / len(xs) if xs else 0.0
+ print(f"{depth:>8} {d['pass']:>4}/{d['total']:<4} {mean(walls):>9.1f}s {mean(prefills):>11.1f} {mean(decodes):>10.1f}")
+ total_pass = sum(1 for r in results if r.get("exact"))
+ print(f"\nTOTAL: {total_pass}/{len(results)} exact")
+ return total_pass == len(results)
+
+
+def run():
+ args = sys.argv[1:]
+ if not args or args[0] not in ("run",):
+ print(__doc__)
+ sys.exit(2)
+ needles = 3
+ repeats = 1
+ depths = list(DEPTH_GRID)
+ rest = args[1:]
+ if "--needles" in rest:
+ needles = int(rest[rest.index("--needles") + 1])
+ if "--repeats" in rest:
+ repeats = int(rest[rest.index("--repeats") + 1])
+ if "--depths" in rest:
+ depths = [int(x) for x in rest[rest.index("--depths") + 1].split(",")]
+ print(f"needle_bench: model={client.MODEL} endpoint={client.ENDPOINT} label={LABEL}")
+ print(f"depths={depths} needles/depth={needles} repeats={repeats} decoy=on")
+ t0 = time.time()
+ results = run_battery(needles, depths, repeats)
+ all_ok = summarize(results)
+ payload = {"label": LABEL, "model": client.MODEL, "endpoint": client.ENDPOINT,
+ "ts": time.strftime("%Y-%m-%d %H:%M:%S"),
+ "needles_per_depth": needles, "repeats": repeats, "depths": depths, "results": results}
+ out_path = write_json(payload, LABEL, prefix="needle")
+ print(f"wrote {out_path} ({time.time() - t0:.0f}s total)")
+ sys.exit(0 if all_ok else 1)
+
+
+
+
+def main():
+ run()
+
+
+if __name__ == "__main__":
+ main()
\ No newline at end of file
diff --git a/bench/tasks/no_think.py b/bench/tasks/no_think.py
new file mode 100644
index 0000000..444ef73
--- /dev/null
+++ b/bench/tasks/no_think.py
@@ -0,0 +1,87 @@
+"""No-think latency instrument — measures the latency win of suppressing reasoning.
+
+Runs each probe in THINK vs NO_THINK mode against the endpoint and compares
+wall time, completion tokens, and thinking length. The NO_THINK mode uses
+the `chat_template_kwargs.enable_thinking` field (for backends that support
+it) and/or the empty-think assistant prefill trick.
+
+Usage (via the bench.py CLI):
+ python bench.py no-think run
+
+Env (via .env): ENDPOINT, MODEL, API_KEY.
+"""
+import json
+import os
+import re
+import sys
+import time
+
+from bench.core import client
+
+PROBES = [
+ ("triage-classify",
+ "You are a security triage assistant. Classify this finding as one of "
+ "{FALSE_POSITIVE, LOW, MEDIUM, HIGH, CRITICAL} and give a one-sentence reason. "
+ "Finding: CVE in a dev-only test dependency (pytest plugin) not shipped in the "
+ "production image; CVSS 9.8 RCE. Answer with 'SEVERITY: <x>' then the reason."),
+ ("mail-triage",
+ "Extract sender, intent, urgency (low/med/high) from this email as strict JSON with "
+ "keys sender,intent,urgency and nothing else: 'From: billing@acme.com — Subject: "
+ "Invoice #4471 overdue, service suspension in 48h.'"),
+ ("classify-short",
+ "Classify the sentiment of this ticket as POSITIVE/NEUTRAL/NEGATIVE and nothing else: "
+ "'Your update broke our export and support hasn't replied in three days.'"),
+ ("arithmetic",
+ "A backup runs every 15 minutes. How many backups happen in one day? Give just the number."),
+]
+
+
+def _split_think(msg, content):
+ rc = msg.get("reasoning_content")
+ if rc:
+ return rc, content
+ m = re.search(chr(60) + "think" + chr(62) + "(.*?)" + chr(60) + "/think" + chr(62) + "(.*)$", content, re.DOTALL)
+ if m:
+ return m.group(1).strip(), m.group(2).strip()
+ return "", content
+
+
+def run():
+ print("### no-think latency bench vs %s ###" % client.MODEL, flush=True)
+ print("%-16s %-9s %8s %8s %8s %s" % ("probe", "mode", "wall_s", "comp_tok", "think_w", "answer"), flush=True)
+ for name, base in PROBES:
+ for mode in ("think", "no_think"):
+ no_think = (mode == "no_think")
+ msgs = [{"role": "user", "content": base}]
+ if no_think and not client.NOTHINK:
+ msgs = msgs + [{"role": "assistant", "content": chr(60) + "think" + chr(62) + chr(60) + "/think" + chr(62)}]
+ try:
+ resp, dt = client.call_model(
+ msgs, max_tokens=4000,
+ temperature=0.3,
+ )
+ ch = resp["choices"][0]
+ msg = ch.get("message", {})
+ content = (msg.get("content") or "").strip()
+ think, ans = _split_think(msg, content)
+ ct = (resp.get("usage") or {}).get("completion_tokens")
+ ans1 = ans.replace("\n", " ")[:70]
+ print("%-16s %-9s %8.1f %8s %8d %s" % (name, mode, dt, ct, len(think.split()), ans1), flush=True)
+ except Exception as e:
+ print("%-16s %-9s ERROR %r" % (name, mode, e), flush=True)
+ print("DONE.", flush=True)
+
+
+
+
+def main():
+ mode = sys.argv[1] if len(sys.argv) > 1 else "run"
+ if mode == "run":
+ run()
+ else:
+ print(f"usage: python bench.py no-think [run]")
+ sys.exit(1)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/bench/tasks/router.py b/bench/tasks/router.py
new file mode 100644
index 0000000..485773b
--- /dev/null
+++ b/bench/tasks/router.py
@@ -0,0 +1,482 @@
+"""Dispatch / router / orchestrator benchmark for LAIC.
+
+Motivation
+----------
+The idea (2026-08-17): use a *cheap, always-on local* model as the front door for
+all communication, and have IT decide which tier actually does each task — keep
+trivial things local, escalate genuinely hard/large-context/high-stakes work to a
+bigger local model or the cloud. This bench measures how good a given model is at
+BEING that dispatcher. It is deliberately shaped like the coding suite
+(env-driven OpenAI endpoint, self-contained dataset, validate/run/report, JSONL
+out, category breakdown) so it runs against the same LM Studio / vLLM endpoints
+the coding suite uses:
+
+ python bench.py router run
+ python bench.py router report # rebuild summary from results/router_<label>.jsonl
+
+Env (via .env / environment): ENDPOINT, MODEL, API_KEY, MAX_TOKENS,
+TEMPERATURE, HTTP_TIMEOUT, NOTHINK. NOTHINK=1 appends an empty <think/>
+assistant turn (portable no-think trick — the parser already strips
+<think>...</think>, so any echoed prefix is harmless).
+
+Two phases
+----------
+1. ROUTE (single-turn tier selection): given a request + a fixed worker menu, emit
+ one JSON dispatch decision {"target","reason","confidence"}. Scored on exact/
+ acceptable routing, format fidelity, and — the axes that actually matter for a
+ dispatcher — over- vs under-escalation (cost-ordered) and privacy-constraint
+ adherence (some requests must NOT leave the LAN regardless of difficulty).
+2. ORCH (multi-step orchestration): given a COMPOUND request, emit an ordered JSON
+ plan of steps, each tagged with a task `kind` and assigned a `worker`, with
+ `depends_on` edges. Scored on decomposition coverage, per-step assignment
+ sanity (worker capable of that kind), and dependency ordering.
+
+Scoring is deterministic: no model judges another. The dataset is curated so every
+gold target sits inside its own acceptable set and every required orchestration
+dependency is well-formed; `validate` proves that before any model is run.
+"""
+import json
+import os
+import re
+import sys
+
+from bench.core import client
+from bench.core.reporting import jsonl_path, append_jsonl, write_json
+
+
+# ---------------------------------------------------------------- the worker menu
+# cost is a coarse relative $/latency weight used only to score over/under-escalation.
+# "local" workers keep data on the LAN; "cloud" workers send it off-box.
+WORKERS = {
+ "local_small": {"cost": 1, "cloud": False,
+ "desc": "A small fast local model (~4B). Cheap, low latency. Good at: greetings and "
+ "chit-chat, short factual answers, reformatting, simple extraction/classification, "
+ "short summaries, trivial arithmetic."},
+ "local_large": {"cost": 3, "cloud": False,
+ "desc": "A capable local reasoning/coding model (~27-35B on the GPU box). Good at: writing "
+ "and debugging code, multi-step reasoning, medium analysis, and agentic tool-use "
+ "jobs. Slower and pricier than local_small but still on-LAN and free."},
+ "cloud_ollama": {"cost": 6, "cloud": True,
+ "desc": "A mid-tier CLOUD model (Ollama Cloud). More headroom than local_large; use as "
+ "overflow for medium-hard work or when local is saturated. Sends data off-box; "
+ "metered cost."},
+ "cloud_claude": {"cost": 10, "cloud": True,
+ "desc": "A frontier CLOUD model (Claude). Best at: the hardest novel/open-ended reasoning, "
+ "very-long-context synthesis, and high-stakes correctness. Most expensive; sends "
+ "data off-box."},
+ "clarify": {"cost": 0, "cloud": False,
+ "desc": "Not a worker — choose this ONLY when the request is too underspecified to route "
+ "and you must ask the user a clarifying question first."},
+}
+CLOUD = {w for w, m in WORKERS.items() if m["cloud"]}
+def cost(w): return WORKERS.get(w, {}).get("cost", 99)
+
+# ---------------------------------------------------------------- ROUTE dataset
+# gold = the single best (cheapest capable, constraint-respecting) target.
+# ok = the acceptable set (includes gold + genuine ties); anything outside is wrong.
+# local_only=True => request carries sensitive data / an explicit keep-on-LAN rule,
+# so ANY cloud_* target is a hard privacy violation regardless of skill fit.
+ROUTE = [
+ # --- trivial / chit-chat -> local_small ---
+ {"id":"r-greet","cat":"trivial","gold":"local_small","ok":{"local_small"},
+ "request":"Hey! Good morning — how's it going?"},
+ {"id":"r-thanks","cat":"trivial","gold":"local_small","ok":{"local_small"},
+ "request":"Thanks, that's all I needed. Have a good one!"},
+ {"id":"r-reformat","cat":"trivial","gold":"local_small","ok":{"local_small"},
+ "request":"Turn this into a bulleted list: eggs, milk, bread, coffee."},
+ {"id":"r-arith","cat":"trivial","gold":"local_small","ok":{"local_small"},
+ "request":"What's 18% of 250?"},
+ # --- simple factual / extraction / classify -> local_small ---
+ {"id":"r-fact","cat":"simple","gold":"local_small","ok":{"local_small"},
+ "request":"What's the capital of Australia?"},
+ {"id":"r-extract","cat":"simple","gold":"local_small","ok":{"local_small","local_large"},
+ "request":"Pull the name and email out of this line: 'Contact: Dana Lee <dana.lee@corp.io>'."},
+ {"id":"r-classify","cat":"simple","gold":"local_small","ok":{"local_small","local_large"},
+ "request":"Is this review positive or negative? 'Honestly the best purchase I made this year.'"},
+ {"id":"r-shortsum","cat":"simple","gold":"local_small","ok":{"local_small","local_large"},
+ "request":"Give me a one-sentence summary of this paragraph: 'The meeting covered Q3 hiring, the "
+ "office move, and the new expense policy. No decisions were finalized.'"},
+ # --- medium coding / debug -> local_large ---
+ {"id":"r-code-fn","cat":"code","gold":"local_large","ok":{"local_large"},
+ "request":"Write a Go function that returns the median of a []float64, handling the even-length case."},
+ {"id":"r-debug","cat":"code","gold":"local_large","ok":{"local_large"},
+ "request":"This Python raises 'dict changed size during iteration' when I delete keys in the loop. Fix it."},
+ {"id":"r-regex","cat":"code","gold":"local_large","ok":{"local_large","local_small"},
+ "request":"Write a regex that matches an ISO-8601 date like 2026-08-17 and explain each part briefly."},
+ {"id":"r-refactor","cat":"code","gold":"local_large","ok":{"local_large","cloud_ollama"},
+ "request":"Refactor this 60-line Python module into three functions with type hints and docstrings."},
+ # --- multi-step reasoning (medium-hard) -> local_large, cloud_ollama a fair tie ---
+ {"id":"r-reason","cat":"reason","gold":"local_large","ok":{"local_large","cloud_ollama"},
+ "request":"Given a 3-tier pricing table, compute the cheapest plan for a customer using 1.2M "
+ "requests/mo with 40GB egress, and show the arithmetic."},
+ {"id":"r-plan","cat":"reason","gold":"local_large","ok":{"local_large","cloud_ollama"},
+ "request":"Draft a step-by-step migration plan to move a Postgres 14 DB to 18 with minimal downtime."},
+ # --- hard / novel / high-stakes -> cloud_claude ---
+ {"id":"r-hard-algo","cat":"hard","gold":"cloud_claude","ok":{"cloud_claude","cloud_ollama"},
+ "request":"Design and prove the correctness of a lock-free MPMC ring buffer with wraparound, "
+ "including the memory-ordering argument for each atomic."},
+ {"id":"r-hard-novel","cat":"hard","gold":"cloud_claude","ok":{"cloud_claude"},
+ "request":"We're being sued over an ambiguous SLA clause. Analyze the legal exposure, argue both "
+ "sides, and recommend a settlement posture. This goes to our board."},
+ {"id":"r-hard-arch","cat":"hard","gold":"cloud_claude","ok":{"cloud_claude","cloud_ollama"},
+ "request":"Critique the trade-offs of event-sourcing vs CRUD for a fintech ledger that must pass a "
+ "SOC 2 audit, and recommend one with justification."},
+ # --- very-long-context synthesis -> cloud_claude (context, not just difficulty) ---
+ {"id":"r-longdoc","cat":"longdoc","gold":"cloud_claude","ok":{"cloud_claude","cloud_ollama"},
+ "request":"Here are five 40-page vendor contracts. Synthesize every conflicting indemnification "
+ "clause across all of them into one comparison table."},
+ {"id":"r-longcode","cat":"longdoc","gold":"cloud_claude","ok":{"cloud_claude","cloud_ollama"},
+ "request":"Read this entire 12,000-line legacy codebase and produce an architecture overview plus "
+ "the three riskiest coupling points."},
+ # --- overflow / mid-cloud is the intended tier -> cloud_ollama ---
+ {"id":"r-overflow","cat":"overflow","gold":"cloud_ollama","ok":{"cloud_ollama","local_large"},
+ "request":"local_large is currently saturated with a long job. I need a medium-difficulty code "
+ "review done now on a 200-line PR. Where should this go?"},
+ # --- privacy: must stay local regardless of difficulty ---
+ {"id":"r-priv-secret","cat":"privacy","gold":"local_large","ok":{"local_large","local_small"},"local_only":True,
+ "request":"Here is our production database password and connection string. Write a script to rotate "
+ "it. Do NOT send any of this off our network."},
+ {"id":"r-priv-pii","cat":"privacy","gold":"local_small","ok":{"local_small","local_large"},"local_only":True,
+ "request":"This spreadsheet has 500 customers' SSNs and home addresses. Just tell me how many rows "
+ "have a missing ZIP code. Keep it on-prem — compliance rule."},
+ {"id":"r-priv-hard","cat":"privacy","gold":"local_large","ok":{"local_large"},"local_only":True,
+ "request":"Analyze this internal, unreleased financial model (highly confidential, must not leave "
+ "the building) and find the three biggest risks in the assumptions."},
+ # --- ambiguous / underspecified -> clarify ---
+ {"id":"r-amb-vague","cat":"clarify","gold":"clarify","ok":{"clarify"},
+ "request":"Can you help me with the thing from yesterday?"},
+ {"id":"r-amb-it","cat":"clarify","gold":"clarify","ok":{"clarify"},
+ "request":"Fix it."},
+ {"id":"r-amb-empty","cat":"clarify","gold":"clarify","ok":{"clarify"},
+ "request":"?"},
+]
+
+# ---------------------------------------------------------------- ORCH dataset
+# Each compound request must decompose into steps tagged with a `kind` (from KINDS)
+# and assigned a `worker`. Scoring is by KIND, not by exact wording:
+# coverage = every required_kind appears in the plan
+# assign = every produced step whose kind is known is routed to a capable worker
+# ordering = for each (a,b) in deps, some kind-b step depends (directly/transitively)
+# on some kind-a step
+KINDS = {
+ "chat","extract","classify","summarize","math","code","debug","reason",
+ "longdoc","draft_message","web_search",
+}
+# which workers are *capable* of each kind (cheapest-capable-first is the ideal, but
+# any capable worker counts as a correct assignment; escalation cost is scored on ROUTE).
+CAPABLE = {
+ "chat": {"local_small","local_large"},
+ "extract": {"local_small","local_large"},
+ "classify": {"local_small","local_large"},
+ "summarize": {"local_small","local_large","cloud_ollama","cloud_claude"},
+ "math": {"local_small","local_large"},
+ "code": {"local_large","cloud_ollama","cloud_claude"},
+ "debug": {"local_large","cloud_ollama","cloud_claude"},
+ "reason": {"local_large","cloud_ollama","cloud_claude"},
+ "longdoc": {"cloud_ollama","cloud_claude"},
+ "draft_message":{"local_small","local_large"},
+ "web_search": {"local_small","local_large","cloud_ollama"},
+}
+ORCH = [
+ {"id":"o-log-code-msg",
+ "request":"Summarize what went wrong in this 2,000-line error log, then write a Python function to "
+ "parse that log format, then draft a short Slack message telling the team the root cause.",
+ "required_kinds":["summarize","code","draft_message"],
+ "deps":[("summarize","draft_message")]},
+ {"id":"o-extract-analyze-report",
+ "request":"Extract the line items from this invoice, check the arithmetic, and draft an email to "
+ "accounts payable flagging any discrepancy.",
+ "required_kinds":["extract","math","draft_message"],
+ "deps":[("extract","math"),("math","draft_message")]},
+ {"id":"o-research-code",
+ "request":"Look up the current recommended way to do structured logging in Go, then write a small "
+ "logging wrapper using it, then write a unit test for the wrapper.",
+ "required_kinds":["web_search","code"],
+ "deps":[("web_search","code")]},
+ {"id":"o-bigdoc-decide-draft",
+ "request":"Read these three 50-page RFP responses, compare them on price and SLA, and draft a "
+ "one-paragraph recommendation to leadership.",
+ "required_kinds":["longdoc","reason","draft_message"],
+ "deps":[("longdoc","reason"),("reason","draft_message")]},
+ {"id":"o-classify-route",
+ "request":"Classify these 20 support tickets by urgency, summarize the urgent ones, and write a "
+ "message assigning them to on-call.",
+ "required_kinds":["classify","summarize","draft_message"],
+ "deps":[("classify","summarize"),("summarize","draft_message")]},
+ {"id":"o-debug-explain",
+ "request":"Figure out why this Go service deadlocks under load, fix it, and write a short postmortem "
+ "message for the incident channel.",
+ "required_kinds":["debug","draft_message"],
+ "deps":[("debug","draft_message")]},
+ {"id":"o-privacy-chain","local_only":True,
+ "request":"Using this confidential internal salary dataset (must stay on-prem), compute the median "
+ "pay gap by department and draft an internal memo summarizing it.",
+ "required_kinds":["math","draft_message"],
+ "deps":[("math","draft_message")]},
+ {"id":"o-simple-two",
+ "request":"Extract the phone numbers from this contact list and format them as an E.164 list.",
+ "required_kinds":["extract"],
+ "deps":[]},
+]
+
+# ---------------------------------------------------------------- local helpers
+def _first_json(text, want="object"):
+ """Extract the first top-level JSON object ({...}) or array ([...]) from text,
+ tolerating <think>...</think> preambles and markdown fences."""
+ text = re.sub(r"<think>.*?</think>", "", text, flags=re.DOTALL)
+ opener, closer = ("{", "}") if want == "object" else ("[", "]")
+ start = text.find(opener)
+ if start < 0:
+ return None
+ depth, instr, esc = 0, False, False
+ for i in range(start, len(text)):
+ c = text[i]
+ if instr:
+ if esc: esc = False
+ elif c == "\\": esc = True
+ elif c == '"': instr = False
+ continue
+ if c == '"': instr = True
+ elif c == opener: depth += 1
+ elif c == closer:
+ depth -= 1
+ if depth == 0:
+ try:
+ return json.loads(text[start:i+1])
+ except Exception:
+ return None
+ return None
+
+# ---------------------------------------------------------------- ROUTE phase
+MENU_TEXT = "\n".join(" - %s (cost %d, %s): %s" %
+ (w, m["cost"], "CLOUD/off-box" if m["cloud"] else "on-LAN", m["desc"])
+ for w, m in WORKERS.items())
+ROUTE_SYSTEM = (
+ "You are a dispatch router. Every incoming request must be handled by exactly one worker. "
+ "Pick the CHEAPEST worker that can do the job well — keep easy work local, and only escalate "
+ "to a bigger or cloud worker when the task genuinely needs it (hard novel reasoning, very long "
+ "context, or high-stakes correctness). Hard rule: if the request contains sensitive/confidential "
+ "data or says to keep data on-prem/on-LAN, you MUST choose an on-LAN worker (never a CLOUD one), "
+ "even if a cloud worker would be more capable. If the request is too vague to route, choose "
+ "'clarify'.\n\nWorkers:\n" + MENU_TEXT +
+ "\n\nRespond with ONLY a single JSON object and nothing else:\n"
+ '{"target": "<one worker name>", "reason": "<short>", "confidence": <0.0-1.0>}')
+
+def score_route(item, resp):
+ c = client.content(resp)
+ obj = _first_json(c, "object")
+ rec = {"id": item["id"], "cat": item["cat"], "gold": item["gold"]}
+ if not obj or "target" not in obj:
+ rec.update({"format_ok": False, "target": None, "exact": False,
+ "acceptable": False, "detail": ("no JSON target: %r" % c[:140])})
+ return rec
+ target = str(obj.get("target", "")).strip()
+ valid = target in WORKERS
+ ok_set = item["ok"]
+ exact = target == item["gold"]
+ acceptable = target in ok_set
+ local_only = item.get("local_only", False)
+ priv_violation = bool(local_only and target in CLOUD)
+ esc = cost(target) - cost(item["gold"]) if valid and target != "clarify" and item["gold"] != "clarify" else 0
+ rec.update({
+ "format_ok": valid, "target": target, "confidence": obj.get("confidence"),
+ "exact": exact, "acceptable": acceptable and not priv_violation,
+ "priv_violation": priv_violation, "esc_err": esc,
+ "detail": "" if (acceptable and not priv_violation) else
+ ("PRIVACY: sent local-only to cloud" if priv_violation else
+ "routed %s, ok=%s" % (target, sorted(ok_set))),
+ })
+ return rec
+
+# ---------------------------------------------------------------- ORCH phase
+ORCH_SYSTEM = (
+ "You are an orchestrator. Break the user's compound request into an ordered plan of atomic steps. "
+ "For EACH step choose a task `kind` and assign the cheapest capable `worker`, and list which "
+ "earlier steps it depends on. Same routing rules as dispatch: keep cheap work local, escalate only "
+ "when needed, and NEVER assign a CLOUD worker to a step that handles on-prem/confidential data.\n\n"
+ "Valid kinds: " + ", ".join(sorted(KINDS)) + "\n"
+ "Valid workers: " + ", ".join(w for w in WORKERS if w != "clarify") + "\n\n"
+ "Respond with ONLY a JSON array (no prose), each element:\n"
+ '{"step": <int, 1-based>, "task": "<what to do>", "kind": "<one kind>", '
+ '"worker": "<one worker>", "depends_on": [<earlier step ints>]}')
+
+def score_orch(item, resp):
+ c = client.content(resp)
+ plan = _first_json(c, "array")
+ rec = {"id": item["id"], "required_kinds": item["required_kinds"]}
+ if not isinstance(plan, list) or not plan:
+ rec.update({"format_ok": False, "coverage": False, "assign_ok": 0.0,
+ "ordering_ok": False, "passed": False,
+ "detail": "no JSON array plan: %r" % c[:140]})
+ return rec
+ # normalize steps
+ steps = []
+ for s in plan:
+ if not isinstance(s, dict):
+ continue
+ steps.append({
+ "step": s.get("step"),
+ "kind": str(s.get("kind", "")).strip(),
+ "worker": str(s.get("worker", "")).strip(),
+ "depends_on": s.get("depends_on") or [],
+ })
+ kinds_present = {s["kind"] for s in steps}
+ coverage = set(item["required_kinds"]) <= kinds_present
+ # assignment: fraction of known-kind steps routed to a capable worker (+privacy)
+ local_only = item.get("local_only", False)
+ known = [s for s in steps if s["kind"] in KINDS]
+ def assign_good(s):
+ if local_only and s["worker"] in CLOUD:
+ return False
+ return s["worker"] in CAPABLE.get(s["kind"], set())
+ assign_ok = (sum(1 for s in known if assign_good(s)) / len(known)) if known else 0.0
+ priv_violation = bool(local_only and any(s["worker"] in CLOUD for s in steps))
+ # ordering: build step->kind and a reachability check over depends_on
+ by_step = {s["step"]: s for s in steps if isinstance(s["step"], int)}
+ def reaches_kind(start_step, target_kind, seen=None):
+ seen = seen or set()
+ for dep in (by_step.get(start_step, {}).get("depends_on") or []):
+ if not isinstance(dep, int) or dep in seen:
+ continue
+ seen.add(dep)
+ d = by_step.get(dep)
+ if not d:
+ continue
+ if d["kind"] == target_kind or reaches_kind(dep, target_kind, seen):
+ return True
+ return False
+ ordering_ok = True
+ for a_kind, b_kind in item["deps"]:
+ b_steps = [s["step"] for s in steps if s["kind"] == b_kind and isinstance(s["step"], int)]
+ if not b_steps or not any(reaches_kind(bs, a_kind) for bs in b_steps):
+ ordering_ok = False
+ break
+ passed = bool(coverage and assign_ok == 1.0 and ordering_ok and not priv_violation)
+ rec.update({
+ "format_ok": True, "n_steps": len(steps), "kinds": sorted(kinds_present),
+ "coverage": coverage, "assign_ok": round(assign_ok, 2),
+ "ordering_ok": ordering_ok, "priv_violation": priv_violation, "passed": passed,
+ "detail": "" if passed else "cov=%s assign=%.2f order=%s priv=%s" %
+ (coverage, assign_ok, ordering_ok, priv_violation),
+ })
+ return rec
+
+# ---------------------------------------------------------------- phases
+def validate():
+ """Prove the dataset is well-formed before trusting any model score."""
+ ok = True
+ for it in ROUTE:
+ if it["gold"] not in it["ok"]:
+ print("BAD route %s: gold %s not in ok %s" % (it["id"], it["gold"], it["ok"])); ok = False
+ if it["gold"] not in WORKERS:
+ print("BAD route %s: gold %s not a worker" % (it["id"], it["gold"])); ok = False
+ if it.get("local_only") and (it["ok"] & CLOUD):
+ print("BAD route %s: local_only but ok set includes cloud" % it["id"]); ok = False
+ for it in ORCH:
+ for k in it["required_kinds"]:
+ if k not in KINDS:
+ print("BAD orch %s: kind %s unknown" % (it["id"], k)); ok = False
+ for a, b in it["deps"]:
+ if a not in it["required_kinds"] or b not in it["required_kinds"]:
+ print("BAD orch %s: dep (%s,%s) not both required" % (it["id"], a, b)); ok = False
+ print("ROUTE items: %d ORCH items: %d" % (len(ROUTE), len(ORCH)))
+ print("VALIDATION %s" % ("PASSED" if ok else "FAILED"))
+ return ok
+
+def run():
+ path = jsonl_path(client.MODEL, prefix="router")
+ open(path, "w").close()
+ recs = []
+ print("=== ROUTER BENCH vs %s (%d route + %d orch) ===\n" %
+ (client.MODEL, len(ROUTE), len(ORCH)), flush=True)
+ print("-- ROUTE --", flush=True)
+ for it in ROUTE:
+ try:
+ resp, dt = client.call_model([{"role": "system", "content": ROUTE_SYSTEM},
+ {"role": "user", "content": it["request"]}])
+ rec = score_route(it, resp)
+ rec["latency_s"] = round(dt, 1)
+ rec["phase"] = "route"
+ except Exception as e:
+ rec = {"id": it["id"], "cat": it["cat"], "phase": "route", "format_ok": False,
+ "exact": False, "acceptable": False, "detail": "ERROR %r" % e}
+ recs.append(rec)
+ append_jsonl(path, rec)
+ print(" %-14s -> %-12s %s%s" % (it["id"], rec.get("target"),
+ "OK" if rec.get("acceptable") else "MISS",
+ "" if rec.get("acceptable") else " (" + (rec.get("detail","") or "")[:60] + ")"), flush=True)
+ print("\n-- ORCH --", flush=True)
+ for it in ORCH:
+ try:
+ resp, dt = client.call_model([{"role": "system", "content": ORCH_SYSTEM},
+ {"role": "user", "content": it["request"]}])
+ rec = score_orch(it, resp)
+ rec["latency_s"] = round(dt, 1)
+ rec["phase"] = "orch"
+ except Exception as e:
+ rec = {"id": it["id"], "phase": "orch", "format_ok": False, "passed": False,
+ "detail": "ERROR %r" % e}
+ recs.append(rec)
+ append_jsonl(path, rec)
+ print(" %-20s %s %s" % (it["id"], "PASS" if rec.get("passed") else "miss",
+ "" if rec.get("passed") else (rec.get("detail","") or "")[:70]), flush=True)
+ report_from(recs, client.MODEL)
+ write_json(recs, client.MODEL, prefix="router")
+
+def report():
+ path = jsonl_path(client.MODEL, prefix="router")
+ recs = [json.loads(l) for l in open(path)]
+ report_from(recs, client.MODEL)
+
+def report_from(recs, model):
+ route = [r for r in recs if r.get("phase") == "route"]
+ orch = [r for r in recs if r.get("phase") == "orch"]
+ print("\n===================== ROUTER SUMMARY: %s =====================" % model)
+ if route:
+ n = len(route)
+ fmt = sum(1 for r in route if r.get("format_ok"))
+ exact = sum(1 for r in route if r.get("exact"))
+ acc = sum(1 for r in route if r.get("acceptable"))
+ priv = sum(1 for r in route if r.get("priv_violation"))
+ over = sum(1 for r in route if (r.get("esc_err") or 0) > 0)
+ under = sum(1 for r in route if (r.get("esc_err") or 0) < 0)
+ mae = round(sum(abs(r.get("esc_err") or 0) for r in route) / n, 2)
+ lat = round(sum(r.get("latency_s") or 0 for r in route) / n, 1)
+ print("ROUTE acceptable %d/%d (%.0f%%) exact %d/%d format %d/%d" %
+ (acc, n, 100*acc/n, exact, n, fmt, n))
+ print(" over-escalate %d under-escalate %d |esc| mean %.2f privacy-violations %d avg lat %.1fs" %
+ (over, under, mae, priv, lat))
+ # per-category acceptable
+ cats = {}
+ for r in route: cats.setdefault(r["cat"], []).append(r)
+ print(" by-cat: " + " ".join("%s %d/%d" %
+ (c, sum(1 for r in rs if r.get("acceptable")), len(rs)) for c, rs in sorted(cats.items())))
+ if orch:
+ n = len(orch)
+ p = sum(1 for r in orch if r.get("passed"))
+ cov = sum(1 for r in orch if r.get("coverage"))
+ aok = round(sum(r.get("assign_ok") or 0 for r in orch) / n, 2)
+ ord_ok = sum(1 for r in orch if r.get("ordering_ok"))
+ priv = sum(1 for r in orch if r.get("priv_violation"))
+ lat = round(sum(r.get("latency_s") or 0 for r in orch) / n, 1)
+ print("ORCH pass %d/%d (%.0f%%) coverage %d/%d mean-assign %.2f ordering %d/%d privacy-viol %d avg lat %.1fs" %
+ (p, n, 100*p/n, cov, n, aok, ord_ok, n, priv, lat))
+ print("=" * 70)
+
+
+
+def main():
+ mode = sys.argv[1] if len(sys.argv) > 1 else "validate"
+ if mode == "validate":
+ sys.exit(0 if validate() else 1)
+ elif mode == "run":
+ run()
+ elif mode == "report":
+ report()
+ else:
+ print(f"usage: python bench.py router [validate|run|report]")
+ sys.exit(1)
+
+
+if __name__ == "__main__":
+ main()
\ No newline at end of file
initial commit
078d198600f2f78cf98809cc2caf6a679b266e85
cjosie <administrator@josie-c.com> · 2026-09-10T16:25 ·
browse files at this commit