josie / llm-bench

"""Loop-provoking battery. Prompts engineered to induce over-rethinking / runaway CoT.
Runs each probe RUNS times at temp 0.6 (the model's recommended sampler for reasoning).
Captures: finish_reason (length == ran to cap == runaway), completion_tokens, wall time,
thinking length, re-derivation markers, and an n-gram repetition score (a high max
repeated-10gram count == a hard loop, the model literally repeating itself).

Thinking is parsed from `reasoning_content` (vLLM / LM Studio separate channel) or
falling back to inline <think>...</think> tags in the content. Models that expose no
thinking channel at all have their answer text scanned instead and flagged so a silent
0-marker report isn't mistaken for clean reasoning.

Usage (via the bench.py CLI):
  python bench.py loop-battery run

Env (via .env): ENDPOINT, MODEL, API_KEY, MAX_TOKENS, TEMPERATURE.
  RUNS       number of runs per probe        (default 2)
  NUM_PREDICT  per-probe generation cap       (default 16000; falls back to MAX_TOKENS)
"""
import os
import re
import sys
from collections import Counter

from bench.core import client
from bench.core.reporting import RESULTS_DIR, safe_label

RUNS = int(os.environ.get("RUNS", "2"))
NUM_PREDICT = int(os.environ.get("NUM_PREDICT", os.environ.get("MAX_TOKENS", "16000")))

PROBES = [
    ("self-ref-wordcount",
     "How many words will be in your final answer to this exact question? Your answer must "
     "explicitly state that number, and the number must exactly match the number of words you "
     "actually write. Give one definitive final answer."),
    ("underdetermined-ages",
     "Three positive integers multiply to 36. I tell you their sum but you can't determine the "
     "ages from it; then I tell you the oldest one likes chess. What are the three ages? Give a "
     "single definitive answer."),
    ("prove-false",
     "Prove rigorously that every odd integer greater than 2 is prime. Provide a complete proof."),
    ("conflicting-opt",
     "What is the single best general-purpose programming language? I need exactly one definitive "
     "answer that is optimal across performance, safety, ecosystem, learning curve, and concurrency "
     "simultaneously. Do not hedge or give conditions."),
    ("ambiguous-pronoun-math",
     "A man buys 3 apples and 2 oranges for $7. His friend buys 2 apples and 3 oranges and pays "
     "the same. Later he says 'I paid a dollar more than him for mine.' What does each fruit cost? "
     "Give a single definitive numeric answer."),
]

MARKERS = ["wait", "let me reconsider", "hold on", "actually,", "let me recompute",
           "re-examine", "let me redo", "hmm", "but wait", "double-check", "recheck",
           "scratch that", "on second thought", "let me restart", "let me try again"]


def split_think(msg, content):
    rc = msg.get("reasoning_content")
    if rc:
        return rc, content
    m = re.search(r"<think>(.*?)</think>(.*)$", content, re.DOTALL)
    if m:
        return m.group(1).strip(), m.group(2).strip()
    return "", content


def rep_score(text):
    w = text.split()
    if len(w) < 20:
        return 0, 1.0
    grams = Counter(tuple(w[i:i + 10]) for i in range(len(w) - 9))
    maxrep = max(grams.values())
    uniq_ratio = len(set(w)) / len(w)
    return maxrep, round(uniq_ratio, 3)


def run():
    model = client.MODEL
    print("### LOOP BATTERY vs %s (temp0.6, num_predict=%d, %d runs each) ###" %
          (model, NUM_PREDICT, RUNS), flush=True)
    safe = safe_label(model)
    for label, prompt in PROBES:
        for n in range(1, RUNS + 1):
            msgs = [{"role": "user", "content": prompt}]
            try:
                resp, dt = client.call_model(
                    msgs, max_tokens=NUM_PREDICT, temperature=0.6)
            except Exception as e:
                print("[%s #%d] ERROR %r" % (label, n, e), flush=True)
                continue
            ch = resp.get("choices", [{}])[0]
            msg = ch.get("message", {}) or {}
            content_str = msg.get("content") or ""
            think, ans = split_think(msg, content_str)
            scan = think if think else ans
            fallback = "" if think else " [no-think-channel, scanned=answer]"
            done = ch.get("finish_reason")
            u = client.usage(resp)
            ct = u["completion_tokens"]
            low = scan.lower()
            mk = sum(low.count(m) for m in MARKERS)
            maxrep, uniq = rep_score(scan)
            runaway = "RUNAWAY(cap)" if done == "length" else "ok"
            print("[%-22s #%d] %-12s done=%-7s gen_tok=%-6s wall=%5.1fs "
                  "think_words=%-5d markers=%-3d max10gram=%d uniq=%.2f%s" %
                  (label, n, runaway, done, ct, dt, len(scan.split()),
                   mk, maxrep, uniq, fallback), flush=True)
            path = os.path.join(
                RESULTS_DIR, "loop_%s_%s_%d.txt" % (safe, label, n))
            with open(path, "w") as f:
                f.write("MODEL:%s\nPROMPT:\n%s\n\n=== THINKING ===\n%s\n\n"
                        "=== ANSWER ===\n%s\n" %
                        (model, prompt, think, ans))
    print("DONE.", flush=True)




def main():
    mode = sys.argv[1] if len(sys.argv) > 1 else "run"
    if mode == "run":
        run()
    else:
        print(f"usage: python bench.py loop-battery [run]")
        sys.exit(1)


if __name__ == "__main__":
    main()