josie / llm-bench

#!/usr/bin/env python3
"""
llm-bench — a local LLM coding-model benchmark suite.

Unified CLI entry point. Point it at any OpenAI-compatible endpoint via .env
(or env vars), pick an instrument, and run. Instruments live under
bench/tasks/ as importable modules.

Usage:
  python bench.py <instrument> <command> [args...]

  python bench.py coding-suite validate
  python bench.py coding-suite run
  python bench.py agentic-fix validate
  python bench.py agentic-fix run
  python bench.py agentic-build validate
  python bench.py agentic-build run
  python bench.py agentic-multitask validate go_htmx
  python bench.py agentic-multitask run go_htmx
  python bench.py agentic-multitask cohort go_htmx rust_cli ts_next
  python bench.py router validate
  python bench.py router run
  python bench.py needle run --depths 1000,16000
  python bench.py loop-battery run
  python bench.py no-think run

Config (via .env in the repo root, or env vars):
  ENDPOINT   OpenAI chat-completions URL
  MODEL      model id
  API_KEY    bearer token (if your endpoint needs auth)
  MAX_TOKENS per-turn generation cap
  TEMPERATURE sampling temperature
  MAX_TURNS  agentic loop cap (for multi-turn instruments)
  HTTP_TIMEOUT per-request wall timeout
  NOTHINK    "1" = suppress reasoning via empty think prefill

See .env.example for all options.
"""
import os
import sys


def _load_dotenv():
    """Load .env from the script's directory (or cwd) into os.environ.

    Only sets vars that aren't already in the environment (env vars win
    over .env, so you can override per-run without editing the file).
    """
    here = os.path.dirname(os.path.abspath(__file__))
    for candidate in (os.path.join(here, ".env"), os.path.join(os.getcwd(), ".env")):
        if not os.path.isfile(candidate):
            continue
        with open(candidate) as f:
            for line in f:
                line = line.strip()
                if not line or line.startswith("#"):
                    continue
                if "=" not in line:
                    continue
                key, _, val = line.partition("=")
                key = key.strip()
                val = val.strip().strip('"').strip("'")
                if key and key not in os.environ:
                    os.environ[key] = val
        break


# Instrument name -> module path (under bench/tasks/)
INSTRUMENTS = {
    "coding-suite": "bench.tasks.coding_suite",
    "agentic-fix": "bench.tasks.agentic_fix",
    "agentic-build": "bench.tasks.agentic_build",
    "agentic-multitask": "bench.tasks.agentic_multitask",
    "router": "bench.tasks.router",
    "needle": "bench.tasks.needle",
    "loop-battery": "bench.tasks.loop_battery",
    "no-think": "bench.tasks.no_think",
}


def main():
    _load_dotenv()
    if len(sys.argv) < 2:
        print(__doc__)
        print("\nAvailable instruments:")
        for name in sorted(INSTRUMENTS):
            print(f"  {name}")
        sys.exit(1)

    instrument = sys.argv[1]
    if instrument not in INSTRUMENTS:
        print(f"ERROR: unknown instrument {instrument!r}")
        print(f"Available: {', '.join(sorted(INSTRUMENTS))}")
        sys.exit(1)

    # Import the instrument module and pass the remaining args to it.
    # The script's directory must be on sys.path so `bench` package resolves.
    here = os.path.dirname(os.path.abspath(__file__))
    if here not in sys.path:
        sys.path.insert(0, here)

    mod = __import__(INSTRUMENTS[instrument], fromlist=["__main__"])
    # Strip the instrument name from argv so the module sees its own args.
    sys.argv = [INSTRUMENTS[instrument]] + sys.argv[2:]
    # Each instrument module has a main() that parses sys.argv for its
    # subcommand (validate / run / report / etc).
    if hasattr(mod, "main"):
        mod.main()
    else:
        print(f"ERROR: instrument {instrument} has no main()")
        sys.exit(1)


if __name__ == "__main__":
    main()