# llm-bench configuration. # Copy to .env and edit. Env vars already in your shell take precedence # over values here (so you can override per-run without editing the file). # The OpenAI-compatible chat-completions URL of the model you're testing. # Examples: # LM Studio (local): http://127.0.0.1:1234/v1/chat/completions # Ollama (local): http://127.0.0.1:11434/v1/chat/completions # vLLM / llama-server: http://127.0.0.1:8000/v1/chat/completions # Ollama Cloud: https://api.ollama-cloud.com/v1/chat/completions # OpenAI: https://api.openai.com/v1/chat/completions ENDPOINT=http://127.0.0.1:1234/v1/chat/completions # The model id the endpoint expects in the "model" field of requests. # For LM Studio this is the load identifier; for Ollama it's the tag; # for vLLM/llama-server it's the served model name; for cloud it's the # API model string (e.g. "gpt-4o", "claude-sonnet-5" via a compat layer). MODEL=local # Bearer token, sent as "Authorization: Bearer " if non-empty. # Leave blank for local endpoints (LM Studio, local Ollama, local vLLM). # Set for cloud endpoints or any endpoint that requires auth. API_KEY= # Per-turn generation cap (max_tokens in the request body). # Coding suite default is 8000; agentic benches default to 16000. MAX_TOKENS=8000 # Sampling temperature. 0 = greedy (deterministic, recommended for # coding/tool-use benchmarks). Higher = more creative/noise. TEMPERATURE=0 # Per-request wall timeout in seconds. Bump for slow local models. HTTP_TIMEOUT=900 # Agentic loop cap — max turns the agent can take before giving up. # Only used by agentic-fix, agentic-build, agentic-multitask. MAX_TURNS=40 # Suppress reasoning/thinking via an empty assistant # prefill. Works on Qwen3-family and other models that use the think-tag # convention. Set to "1" to enable. Some backends also honor a # chat_template_kwargs.enable_thinking field (the no-think instrument # tests both paths). NOTHINK= # --- build/test timeouts (agentic benches) --- BUILD_TIMEOUT=180 TEST_TIMEOUT=180 SHELL_TIMEOUT=60 SERVER_TIMEOUT=20