"""Loop-provoking battery. Prompts engineered to induce over-rethinking / runaway CoT. Runs each probe RUNS times at temp 0.6 (the model's recommended sampler for reasoning). Captures: finish_reason (length == ran to cap == runaway), completion_tokens, wall time, thinking length, re-derivation markers, and an n-gram repetition score (a high max repeated-10gram count == a hard loop, the model literally repeating itself). Thinking is parsed from `reasoning_content` (vLLM / LM Studio separate channel) or falling back to inline ... tags in the content. Models that expose no thinking channel at all have their answer text scanned instead and flagged so a silent 0-marker report isn't mistaken for clean reasoning. Usage (via the bench.py CLI): python bench.py loop-battery run Env (via .env): ENDPOINT, MODEL, API_KEY, MAX_TOKENS, TEMPERATURE. RUNS number of runs per probe (default 2) NUM_PREDICT per-probe generation cap (default 16000; falls back to MAX_TOKENS) """ import os import re import sys from collections import Counter from bench.core import client from bench.core.reporting import RESULTS_DIR, safe_label RUNS = int(os.environ.get("RUNS", "2")) NUM_PREDICT = int(os.environ.get("NUM_PREDICT", os.environ.get("MAX_TOKENS", "16000"))) PROBES = [ ("self-ref-wordcount", "How many words will be in your final answer to this exact question? Your answer must " "explicitly state that number, and the number must exactly match the number of words you " "actually write. Give one definitive final answer."), ("underdetermined-ages", "Three positive integers multiply to 36. I tell you their sum but you can't determine the " "ages from it; then I tell you the oldest one likes chess. What are the three ages? Give a " "single definitive answer."), ("prove-false", "Prove rigorously that every odd integer greater than 2 is prime. Provide a complete proof."), ("conflicting-opt", "What is the single best general-purpose programming language? I need exactly one definitive " "answer that is optimal across performance, safety, ecosystem, learning curve, and concurrency " "simultaneously. Do not hedge or give conditions."), ("ambiguous-pronoun-math", "A man buys 3 apples and 2 oranges for $7. His friend buys 2 apples and 3 oranges and pays " "the same. Later he says 'I paid a dollar more than him for mine.' What does each fruit cost? " "Give a single definitive numeric answer."), ] MARKERS = ["wait", "let me reconsider", "hold on", "actually,", "let me recompute", "re-examine", "let me redo", "hmm", "but wait", "double-check", "recheck", "scratch that", "on second thought", "let me restart", "let me try again"] def split_think(msg, content): rc = msg.get("reasoning_content") if rc: return rc, content m = re.search(r"(.*?)(.*)$", content, re.DOTALL) if m: return m.group(1).strip(), m.group(2).strip() return "", content def rep_score(text): w = text.split() if len(w) < 20: return 0, 1.0 grams = Counter(tuple(w[i:i + 10]) for i in range(len(w) - 9)) maxrep = max(grams.values()) uniq_ratio = len(set(w)) / len(w) return maxrep, round(uniq_ratio, 3) def run(): model = client.MODEL print("### LOOP BATTERY vs %s (temp0.6, num_predict=%d, %d runs each) ###" % (model, NUM_PREDICT, RUNS), flush=True) safe = safe_label(model) for label, prompt in PROBES: for n in range(1, RUNS + 1): msgs = [{"role": "user", "content": prompt}] try: resp, dt = client.call_model( msgs, max_tokens=NUM_PREDICT, temperature=0.6) except Exception as e: print("[%s #%d] ERROR %r" % (label, n, e), flush=True) continue ch = resp.get("choices", [{}])[0] msg = ch.get("message", {}) or {} content_str = msg.get("content") or "" think, ans = split_think(msg, content_str) scan = think if think else ans fallback = "" if think else " [no-think-channel, scanned=answer]" done = ch.get("finish_reason") u = client.usage(resp) ct = u["completion_tokens"] low = scan.lower() mk = sum(low.count(m) for m in MARKERS) maxrep, uniq = rep_score(scan) runaway = "RUNAWAY(cap)" if done == "length" else "ok" print("[%-22s #%d] %-12s done=%-7s gen_tok=%-6s wall=%5.1fs " "think_words=%-5d markers=%-3d max10gram=%d uniq=%.2f%s" % (label, n, runaway, done, ct, dt, len(scan.split()), mk, maxrep, uniq, fallback), flush=True) path = os.path.join( RESULTS_DIR, "loop_%s_%s_%d.txt" % (safe, label, n)) with open(path, "w") as f: f.write("MODEL:%s\nPROMPT:\n%s\n\n=== THINKING ===\n%s\n\n" "=== ANSWER ===\n%s\n" % (model, prompt, think, ans)) print("DONE.", flush=True) def main(): mode = sys.argv[1] if len(sys.argv) > 1 else "run" if mode == "run": run() else: print(f"usage: python bench.py loop-battery [run]") sys.exit(1) if __name__ == "__main__": main()