6ad897477d988ba75c2bf78f4acc0949132901e0 / bench.py · 3925 bytes · raw
#!/usr/bin/env python3
"""
llm-bench — a local LLM coding-model benchmark suite.
Unified CLI entry point. Point it at any OpenAI-compatible endpoint via .env
(or env vars), pick an instrument, and run. Instruments live under
bench/tasks/ as importable modules.
Usage:
python bench.py <instrument> <command> [args...]
python bench.py coding-suite validate
python bench.py coding-suite run
python bench.py agentic-fix validate
python bench.py agentic-fix run
python bench.py agentic-build validate
python bench.py agentic-build run
python bench.py agentic-multitask validate go_htmx
python bench.py agentic-multitask run go_htmx
python bench.py agentic-multitask cohort go_htmx rust_cli ts_next
python bench.py router validate
python bench.py router run
python bench.py needle run --depths 1000,16000
python bench.py loop-battery run
python bench.py no-think run
Config (via .env in the repo root, or env vars):
ENDPOINT OpenAI chat-completions URL
MODEL model id
API_KEY bearer token (if your endpoint needs auth)
MAX_TOKENS per-turn generation cap
TEMPERATURE sampling temperature
MAX_TURNS agentic loop cap (for multi-turn instruments)
HTTP_TIMEOUT per-request wall timeout
NOTHINK "1" = suppress reasoning via empty think prefill
See .env.example for all options.
"""
import os
import sys
def _load_dotenv():
"""Load .env from the script's directory (or cwd) into os.environ.
Only sets vars that aren't already in the environment (env vars win
over .env, so you can override per-run without editing the file).
"""
here = os.path.dirname(os.path.abspath(__file__))
for candidate in (os.path.join(here, ".env"), os.path.join(os.getcwd(), ".env")):
if not os.path.isfile(candidate):
continue
with open(candidate) as f:
for line in f:
line = line.strip()
if not line or line.startswith("#"):
continue
if "=" not in line:
continue
key, _, val = line.partition("=")
key = key.strip()
val = val.strip().strip('"').strip("'")
if key and key not in os.environ:
os.environ[key] = val
break
# Instrument name -> module path (under bench/tasks/)
INSTRUMENTS = {
"coding-suite": "bench.tasks.coding_suite",
"agentic-fix": "bench.tasks.agentic_fix",
"agentic-build": "bench.tasks.agentic_build",
"agentic-multitask": "bench.tasks.agentic_multitask",
"router": "bench.tasks.router",
"needle": "bench.tasks.needle",
"loop-battery": "bench.tasks.loop_battery",
"no-think": "bench.tasks.no_think",
}
def main():
_load_dotenv()
if len(sys.argv) < 2:
print(__doc__)
print("\nAvailable instruments:")
for name in sorted(INSTRUMENTS):
print(f" {name}")
sys.exit(1)
instrument = sys.argv[1]
if instrument not in INSTRUMENTS:
print(f"ERROR: unknown instrument {instrument!r}")
print(f"Available: {', '.join(sorted(INSTRUMENTS))}")
sys.exit(1)
# Import the instrument module and pass the remaining args to it.
# The script's directory must be on sys.path so `bench` package resolves.
here = os.path.dirname(os.path.abspath(__file__))
if here not in sys.path:
sys.path.insert(0, here)
mod = __import__(INSTRUMENTS[instrument], fromlist=["__main__"])
# Strip the instrument name from argv so the module sees its own args.
sys.argv = [INSTRUMENTS[instrument]] + sys.argv[2:]
# Each instrument module has a main() that parses sys.argv for its
# subcommand (validate / run / report / etc).
if hasattr(mod, "main"):
mod.main()
else:
print(f"ERROR: instrument {instrument} has no main()")
sys.exit(1)
if __name__ == "__main__":
main()