"""Loop-provoking battery. Prompts engineered to induce over-rethinking / runaway CoT.
Runs each probe RUNS times at temp 0.6 (the model's recommended sampler for reasoning).
Captures: finish_reason (length == ran to cap == runaway), completion_tokens, wall time,
thinking length, re-derivation markers, and an n-gram repetition score (a high max
repeated-10gram count == a hard loop, the model literally repeating itself).
Thinking is parsed from `reasoning_content` (vLLM / LM Studio separate channel) or
falling back to inline ... tags in the content. Models that expose no
thinking channel at all have their answer text scanned instead and flagged so a silent
0-marker report isn't mistaken for clean reasoning.
Usage (via the bench.py CLI):
python bench.py loop-battery run
Env (via .env): ENDPOINT, MODEL, API_KEY, MAX_TOKENS, TEMPERATURE.
RUNS number of runs per probe (default 2)
NUM_PREDICT per-probe generation cap (default 16000; falls back to MAX_TOKENS)
"""
import os
import re
import sys
from collections import Counter
from bench.core import client
from bench.core.reporting import RESULTS_DIR, safe_label
RUNS = int(os.environ.get("RUNS", "2"))
NUM_PREDICT = int(os.environ.get("NUM_PREDICT", os.environ.get("MAX_TOKENS", "16000")))
PROBES = [
("self-ref-wordcount",
"How many words will be in your final answer to this exact question? Your answer must "
"explicitly state that number, and the number must exactly match the number of words you "
"actually write. Give one definitive final answer."),
("underdetermined-ages",
"Three positive integers multiply to 36. I tell you their sum but you can't determine the "
"ages from it; then I tell you the oldest one likes chess. What are the three ages? Give a "
"single definitive answer."),
("prove-false",
"Prove rigorously that every odd integer greater than 2 is prime. Provide a complete proof."),
("conflicting-opt",
"What is the single best general-purpose programming language? I need exactly one definitive "
"answer that is optimal across performance, safety, ecosystem, learning curve, and concurrency "
"simultaneously. Do not hedge or give conditions."),
("ambiguous-pronoun-math",
"A man buys 3 apples and 2 oranges for $7. His friend buys 2 apples and 3 oranges and pays "
"the same. Later he says 'I paid a dollar more than him for mine.' What does each fruit cost? "
"Give a single definitive numeric answer."),
]
MARKERS = ["wait", "let me reconsider", "hold on", "actually,", "let me recompute",
"re-examine", "let me redo", "hmm", "but wait", "double-check", "recheck",
"scratch that", "on second thought", "let me restart", "let me try again"]
def split_think(msg, content):
rc = msg.get("reasoning_content")
if rc:
return rc, content
m = re.search(r"(.*?)(.*)$", content, re.DOTALL)
if m:
return m.group(1).strip(), m.group(2).strip()
return "", content
def rep_score(text):
w = text.split()
if len(w) < 20:
return 0, 1.0
grams = Counter(tuple(w[i:i + 10]) for i in range(len(w) - 9))
maxrep = max(grams.values())
uniq_ratio = len(set(w)) / len(w)
return maxrep, round(uniq_ratio, 3)
def run():
model = client.MODEL
print("### LOOP BATTERY vs %s (temp0.6, num_predict=%d, %d runs each) ###" %
(model, NUM_PREDICT, RUNS), flush=True)
safe = safe_label(model)
for label, prompt in PROBES:
for n in range(1, RUNS + 1):
msgs = [{"role": "user", "content": prompt}]
try:
resp, dt = client.call_model(
msgs, max_tokens=NUM_PREDICT, temperature=0.6)
except Exception as e:
print("[%s #%d] ERROR %r" % (label, n, e), flush=True)
continue
ch = resp.get("choices", [{}])[0]
msg = ch.get("message", {}) or {}
content_str = msg.get("content") or ""
think, ans = split_think(msg, content_str)
scan = think if think else ans
fallback = "" if think else " [no-think-channel, scanned=answer]"
done = ch.get("finish_reason")
u = client.usage(resp)
ct = u["completion_tokens"]
low = scan.lower()
mk = sum(low.count(m) for m in MARKERS)
maxrep, uniq = rep_score(scan)
runaway = "RUNAWAY(cap)" if done == "length" else "ok"
print("[%-22s #%d] %-12s done=%-7s gen_tok=%-6s wall=%5.1fs "
"think_words=%-5d markers=%-3d max10gram=%d uniq=%.2f%s" %
(label, n, runaway, done, ct, dt, len(scan.split()),
mk, maxrep, uniq, fallback), flush=True)
path = os.path.join(
RESULTS_DIR, "loop_%s_%s_%d.txt" % (safe, label, n))
with open(path, "w") as f:
f.write("MODEL:%s\nPROMPT:\n%s\n\n=== THINKING ===\n%s\n\n"
"=== ANSWER ===\n%s\n" %
(model, prompt, think, ans))
print("DONE.", flush=True)
def main():
mode = sys.argv[1] if len(sys.argv) > 1 else "run"
if mode == "run":
run()
else:
print(f"usage: python bench.py loop-battery [run]")
sys.exit(1)
if __name__ == "__main__":
main()