#!/usr/bin/env python3 """ llm-bench — a local LLM coding-model benchmark suite. Unified CLI entry point. Point it at any OpenAI-compatible endpoint via .env (or env vars), pick an instrument, and run. Instruments live under bench/tasks/ as importable modules. Usage: python bench.py [args...] python bench.py coding-suite validate python bench.py coding-suite run python bench.py agentic-fix validate python bench.py agentic-fix run python bench.py agentic-build validate python bench.py agentic-build run python bench.py agentic-multitask validate go_htmx python bench.py agentic-multitask run go_htmx python bench.py agentic-multitask cohort go_htmx rust_cli ts_next python bench.py router validate python bench.py router run python bench.py needle run --depths 1000,16000 python bench.py loop-battery run python bench.py no-think run Config (via .env in the repo root, or env vars): ENDPOINT OpenAI chat-completions URL MODEL model id API_KEY bearer token (if your endpoint needs auth) MAX_TOKENS per-turn generation cap TEMPERATURE sampling temperature MAX_TURNS agentic loop cap (for multi-turn instruments) HTTP_TIMEOUT per-request wall timeout NOTHINK "1" = suppress reasoning via empty think prefill See .env.example for all options. """ import os import sys def _load_dotenv(): """Load .env from the script's directory (or cwd) into os.environ. Only sets vars that aren't already in the environment (env vars win over .env, so you can override per-run without editing the file). """ here = os.path.dirname(os.path.abspath(__file__)) for candidate in (os.path.join(here, ".env"), os.path.join(os.getcwd(), ".env")): if not os.path.isfile(candidate): continue with open(candidate) as f: for line in f: line = line.strip() if not line or line.startswith("#"): continue if "=" not in line: continue key, _, val = line.partition("=") key = key.strip() val = val.strip().strip('"').strip("'") if key and key not in os.environ: os.environ[key] = val break # Instrument name -> module path (under bench/tasks/) INSTRUMENTS = { "coding-suite": "bench.tasks.coding_suite", "agentic-fix": "bench.tasks.agentic_fix", "agentic-build": "bench.tasks.agentic_build", "agentic-multitask": "bench.tasks.agentic_multitask", "router": "bench.tasks.router", "needle": "bench.tasks.needle", "loop-battery": "bench.tasks.loop_battery", "no-think": "bench.tasks.no_think", } def main(): _load_dotenv() if len(sys.argv) < 2: print(__doc__) print("\nAvailable instruments:") for name in sorted(INSTRUMENTS): print(f" {name}") sys.exit(1) instrument = sys.argv[1] if instrument not in INSTRUMENTS: print(f"ERROR: unknown instrument {instrument!r}") print(f"Available: {', '.join(sorted(INSTRUMENTS))}") sys.exit(1) # Import the instrument module and pass the remaining args to it. # The script's directory must be on sys.path so `bench` package resolves. here = os.path.dirname(os.path.abspath(__file__)) if here not in sys.path: sys.path.insert(0, here) mod = __import__(INSTRUMENTS[instrument], fromlist=["__main__"]) # Strip the instrument name from argv so the module sees its own args. sys.argv = [INSTRUMENTS[instrument]] + sys.argv[2:] # Each instrument module has a main() that parses sys.argv for its # subcommand (validate / run / report / etc). if hasattr(mod, "main"): mod.main() else: print(f"ERROR: instrument {instrument} has no main()") sys.exit(1) if __name__ == "__main__": main()