josie / llm-bench

"""Dispatch / router / orchestrator benchmark for LAIC.

Motivation
----------
The idea (2026-08-17): use a *cheap, always-on local* model as the front door for
all communication, and have IT decide which tier actually does each task — keep
trivial things local, escalate genuinely hard/large-context/high-stakes work to a
bigger local model or the cloud. This bench measures how good a given model is at
BEING that dispatcher. It is deliberately shaped like the coding suite
(env-driven OpenAI endpoint, self-contained dataset, validate/run/report, JSONL
out, category breakdown) so it runs against the same LM Studio / vLLM endpoints
the coding suite uses:

    python bench.py router run
    python bench.py router report   # rebuild summary from results/router_<label>.jsonl

Env (via .env / environment): ENDPOINT, MODEL, API_KEY, MAX_TOKENS,
TEMPERATURE, HTTP_TIMEOUT, NOTHINK. NOTHINK=1 appends an empty <think/>
assistant turn (portable no-think trick — the parser already strips
<think>...</think>, so any echoed prefix is harmless).

Two phases
----------
1. ROUTE  (single-turn tier selection): given a request + a fixed worker menu, emit
   one JSON dispatch decision {"target","reason","confidence"}. Scored on exact/
   acceptable routing, format fidelity, and — the axes that actually matter for a
   dispatcher — over- vs under-escalation (cost-ordered) and privacy-constraint
   adherence (some requests must NOT leave the LAN regardless of difficulty).
2. ORCH   (multi-step orchestration): given a COMPOUND request, emit an ordered JSON
   plan of steps, each tagged with a task `kind` and assigned a `worker`, with
   `depends_on` edges. Scored on decomposition coverage, per-step assignment
   sanity (worker capable of that kind), and dependency ordering.

Scoring is deterministic: no model judges another. The dataset is curated so every
gold target sits inside its own acceptable set and every required orchestration
dependency is well-formed; `validate` proves that before any model is run.
"""
import json
import os
import re
import sys

from bench.core import client
from bench.core.reporting import jsonl_path, append_jsonl, write_json


# ---------------------------------------------------------------- the worker menu
# cost is a coarse relative $/latency weight used only to score over/under-escalation.
# "local" workers keep data on the LAN; "cloud" workers send it off-box.
WORKERS = {
    "local_small":  {"cost": 1,  "cloud": False,
        "desc": "A small fast local model (~4B). Cheap, low latency. Good at: greetings and "
                "chit-chat, short factual answers, reformatting, simple extraction/classification, "
                "short summaries, trivial arithmetic."},
    "local_large":  {"cost": 3,  "cloud": False,
        "desc": "A capable local reasoning/coding model (~27-35B on the GPU box). Good at: writing "
                "and debugging code, multi-step reasoning, medium analysis, and agentic tool-use "
                "jobs. Slower and pricier than local_small but still on-LAN and free."},
    "cloud_ollama": {"cost": 6,  "cloud": True,
        "desc": "A mid-tier CLOUD model (Ollama Cloud). More headroom than local_large; use as "
                "overflow for medium-hard work or when local is saturated. Sends data off-box; "
                "metered cost."},
    "cloud_claude": {"cost": 10, "cloud": True,
        "desc": "A frontier CLOUD model (Claude). Best at: the hardest novel/open-ended reasoning, "
                "very-long-context synthesis, and high-stakes correctness. Most expensive; sends "
                "data off-box."},
    "clarify":      {"cost": 0,  "cloud": False,
        "desc": "Not a worker — choose this ONLY when the request is too underspecified to route "
                "and you must ask the user a clarifying question first."},
}
CLOUD = {w for w, m in WORKERS.items() if m["cloud"]}
def cost(w): return WORKERS.get(w, {}).get("cost", 99)

# ---------------------------------------------------------------- ROUTE dataset
# gold = the single best (cheapest capable, constraint-respecting) target.
# ok   = the acceptable set (includes gold + genuine ties); anything outside is wrong.
# local_only=True => request carries sensitive data / an explicit keep-on-LAN rule,
#                    so ANY cloud_* target is a hard privacy violation regardless of skill fit.
ROUTE = [
 # --- trivial / chit-chat -> local_small ---
 {"id":"r-greet","cat":"trivial","gold":"local_small","ok":{"local_small"},
  "request":"Hey! Good morning — how's it going?"},
 {"id":"r-thanks","cat":"trivial","gold":"local_small","ok":{"local_small"},
  "request":"Thanks, that's all I needed. Have a good one!"},
 {"id":"r-reformat","cat":"trivial","gold":"local_small","ok":{"local_small"},
  "request":"Turn this into a bulleted list: eggs, milk, bread, coffee."},
 {"id":"r-arith","cat":"trivial","gold":"local_small","ok":{"local_small"},
  "request":"What's 18% of 250?"},
 # --- simple factual / extraction / classify -> local_small ---
 {"id":"r-fact","cat":"simple","gold":"local_small","ok":{"local_small"},
  "request":"What's the capital of Australia?"},
 {"id":"r-extract","cat":"simple","gold":"local_small","ok":{"local_small","local_large"},
  "request":"Pull the name and email out of this line: 'Contact: Dana Lee <dana.lee@corp.io>'."},
 {"id":"r-classify","cat":"simple","gold":"local_small","ok":{"local_small","local_large"},
  "request":"Is this review positive or negative? 'Honestly the best purchase I made this year.'"},
 {"id":"r-shortsum","cat":"simple","gold":"local_small","ok":{"local_small","local_large"},
  "request":"Give me a one-sentence summary of this paragraph: 'The meeting covered Q3 hiring, the "
            "office move, and the new expense policy. No decisions were finalized.'"},
 # --- medium coding / debug -> local_large ---
 {"id":"r-code-fn","cat":"code","gold":"local_large","ok":{"local_large"},
  "request":"Write a Go function that returns the median of a []float64, handling the even-length case."},
 {"id":"r-debug","cat":"code","gold":"local_large","ok":{"local_large"},
  "request":"This Python raises 'dict changed size during iteration' when I delete keys in the loop. Fix it."},
 {"id":"r-regex","cat":"code","gold":"local_large","ok":{"local_large","local_small"},
  "request":"Write a regex that matches an ISO-8601 date like 2026-08-17 and explain each part briefly."},
 {"id":"r-refactor","cat":"code","gold":"local_large","ok":{"local_large","cloud_ollama"},
  "request":"Refactor this 60-line Python module into three functions with type hints and docstrings."},
 # --- multi-step reasoning (medium-hard) -> local_large, cloud_ollama a fair tie ---
 {"id":"r-reason","cat":"reason","gold":"local_large","ok":{"local_large","cloud_ollama"},
  "request":"Given a 3-tier pricing table, compute the cheapest plan for a customer using 1.2M "
            "requests/mo with 40GB egress, and show the arithmetic."},
 {"id":"r-plan","cat":"reason","gold":"local_large","ok":{"local_large","cloud_ollama"},
  "request":"Draft a step-by-step migration plan to move a Postgres 14 DB to 18 with minimal downtime."},
 # --- hard / novel / high-stakes -> cloud_claude ---
 {"id":"r-hard-algo","cat":"hard","gold":"cloud_claude","ok":{"cloud_claude","cloud_ollama"},
  "request":"Design and prove the correctness of a lock-free MPMC ring buffer with wraparound, "
            "including the memory-ordering argument for each atomic."},
 {"id":"r-hard-novel","cat":"hard","gold":"cloud_claude","ok":{"cloud_claude"},
  "request":"We're being sued over an ambiguous SLA clause. Analyze the legal exposure, argue both "
            "sides, and recommend a settlement posture. This goes to our board."},
 {"id":"r-hard-arch","cat":"hard","gold":"cloud_claude","ok":{"cloud_claude","cloud_ollama"},
  "request":"Critique the trade-offs of event-sourcing vs CRUD for a fintech ledger that must pass a "
            "SOC 2 audit, and recommend one with justification."},
 # --- very-long-context synthesis -> cloud_claude (context, not just difficulty) ---
 {"id":"r-longdoc","cat":"longdoc","gold":"cloud_claude","ok":{"cloud_claude","cloud_ollama"},
  "request":"Here are five 40-page vendor contracts. Synthesize every conflicting indemnification "
            "clause across all of them into one comparison table."},
 {"id":"r-longcode","cat":"longdoc","gold":"cloud_claude","ok":{"cloud_claude","cloud_ollama"},
  "request":"Read this entire 12,000-line legacy codebase and produce an architecture overview plus "
            "the three riskiest coupling points."},
 # --- overflow / mid-cloud is the intended tier -> cloud_ollama ---
 {"id":"r-overflow","cat":"overflow","gold":"cloud_ollama","ok":{"cloud_ollama","local_large"},
  "request":"local_large is currently saturated with a long job. I need a medium-difficulty code "
            "review done now on a 200-line PR. Where should this go?"},
 # --- privacy: must stay local regardless of difficulty ---
 {"id":"r-priv-secret","cat":"privacy","gold":"local_large","ok":{"local_large","local_small"},"local_only":True,
  "request":"Here is our production database password and connection string. Write a script to rotate "
            "it. Do NOT send any of this off our network."},
 {"id":"r-priv-pii","cat":"privacy","gold":"local_small","ok":{"local_small","local_large"},"local_only":True,
  "request":"This spreadsheet has 500 customers' SSNs and home addresses. Just tell me how many rows "
            "have a missing ZIP code. Keep it on-prem — compliance rule."},
 {"id":"r-priv-hard","cat":"privacy","gold":"local_large","ok":{"local_large"},"local_only":True,
  "request":"Analyze this internal, unreleased financial model (highly confidential, must not leave "
            "the building) and find the three biggest risks in the assumptions."},
 # --- ambiguous / underspecified -> clarify ---
 {"id":"r-amb-vague","cat":"clarify","gold":"clarify","ok":{"clarify"},
  "request":"Can you help me with the thing from yesterday?"},
 {"id":"r-amb-it","cat":"clarify","gold":"clarify","ok":{"clarify"},
  "request":"Fix it."},
 {"id":"r-amb-empty","cat":"clarify","gold":"clarify","ok":{"clarify"},
  "request":"?"},
]

# ---------------------------------------------------------------- ORCH dataset
# Each compound request must decompose into steps tagged with a `kind` (from KINDS)
# and assigned a `worker`. Scoring is by KIND, not by exact wording:
#   coverage  = every required_kind appears in the plan
#   assign    = every produced step whose kind is known is routed to a capable worker
#   ordering  = for each (a,b) in deps, some kind-b step depends (directly/transitively)
#               on some kind-a step
KINDS = {
 "chat","extract","classify","summarize","math","code","debug","reason",
 "longdoc","draft_message","web_search",
}
# which workers are *capable* of each kind (cheapest-capable-first is the ideal, but
# any capable worker counts as a correct assignment; escalation cost is scored on ROUTE).
CAPABLE = {
 "chat":         {"local_small","local_large"},
 "extract":      {"local_small","local_large"},
 "classify":     {"local_small","local_large"},
 "summarize":    {"local_small","local_large","cloud_ollama","cloud_claude"},
 "math":         {"local_small","local_large"},
 "code":         {"local_large","cloud_ollama","cloud_claude"},
 "debug":        {"local_large","cloud_ollama","cloud_claude"},
 "reason":       {"local_large","cloud_ollama","cloud_claude"},
 "longdoc":      {"cloud_ollama","cloud_claude"},
 "draft_message":{"local_small","local_large"},
 "web_search":   {"local_small","local_large","cloud_ollama"},
}
ORCH = [
 {"id":"o-log-code-msg",
  "request":"Summarize what went wrong in this 2,000-line error log, then write a Python function to "
            "parse that log format, then draft a short Slack message telling the team the root cause.",
  "required_kinds":["summarize","code","draft_message"],
  "deps":[("summarize","draft_message")]},
 {"id":"o-extract-analyze-report",
  "request":"Extract the line items from this invoice, check the arithmetic, and draft an email to "
            "accounts payable flagging any discrepancy.",
  "required_kinds":["extract","math","draft_message"],
  "deps":[("extract","math"),("math","draft_message")]},
 {"id":"o-research-code",
  "request":"Look up the current recommended way to do structured logging in Go, then write a small "
            "logging wrapper using it, then write a unit test for the wrapper.",
  "required_kinds":["web_search","code"],
  "deps":[("web_search","code")]},
 {"id":"o-bigdoc-decide-draft",
  "request":"Read these three 50-page RFP responses, compare them on price and SLA, and draft a "
            "one-paragraph recommendation to leadership.",
  "required_kinds":["longdoc","reason","draft_message"],
  "deps":[("longdoc","reason"),("reason","draft_message")]},
 {"id":"o-classify-route",
  "request":"Classify these 20 support tickets by urgency, summarize the urgent ones, and write a "
            "message assigning them to on-call.",
  "required_kinds":["classify","summarize","draft_message"],
  "deps":[("classify","summarize"),("summarize","draft_message")]},
 {"id":"o-debug-explain",
  "request":"Figure out why this Go service deadlocks under load, fix it, and write a short postmortem "
            "message for the incident channel.",
  "required_kinds":["debug","draft_message"],
  "deps":[("debug","draft_message")]},
 {"id":"o-privacy-chain","local_only":True,
  "request":"Using this confidential internal salary dataset (must stay on-prem), compute the median "
            "pay gap by department and draft an internal memo summarizing it.",
  "required_kinds":["math","draft_message"],
  "deps":[("math","draft_message")]},
 {"id":"o-simple-two",
  "request":"Extract the phone numbers from this contact list and format them as an E.164 list.",
  "required_kinds":["extract"],
  "deps":[]},
]

# ---------------------------------------------------------------- local helpers
def _first_json(text, want="object"):
    """Extract the first top-level JSON object ({...}) or array ([...]) from text,
    tolerating <think>...</think> preambles and markdown fences."""
    text = re.sub(r"<think>.*?</think>", "", text, flags=re.DOTALL)
    opener, closer = ("{", "}") if want == "object" else ("[", "]")
    start = text.find(opener)
    if start < 0:
        return None
    depth, instr, esc = 0, False, False
    for i in range(start, len(text)):
        c = text[i]
        if instr:
            if esc: esc = False
            elif c == "\\": esc = True
            elif c == '"': instr = False
            continue
        if c == '"': instr = True
        elif c == opener: depth += 1
        elif c == closer:
            depth -= 1
            if depth == 0:
                try:
                    return json.loads(text[start:i+1])
                except Exception:
                    return None
    return None

# ---------------------------------------------------------------- ROUTE phase
MENU_TEXT = "\n".join("  - %s (cost %d, %s): %s" %
                      (w, m["cost"], "CLOUD/off-box" if m["cloud"] else "on-LAN", m["desc"])
                      for w, m in WORKERS.items())
ROUTE_SYSTEM = (
    "You are a dispatch router. Every incoming request must be handled by exactly one worker. "
    "Pick the CHEAPEST worker that can do the job well — keep easy work local, and only escalate "
    "to a bigger or cloud worker when the task genuinely needs it (hard novel reasoning, very long "
    "context, or high-stakes correctness). Hard rule: if the request contains sensitive/confidential "
    "data or says to keep data on-prem/on-LAN, you MUST choose an on-LAN worker (never a CLOUD one), "
    "even if a cloud worker would be more capable. If the request is too vague to route, choose "
    "'clarify'.\n\nWorkers:\n" + MENU_TEXT +
    "\n\nRespond with ONLY a single JSON object and nothing else:\n"
    '{"target": "<one worker name>", "reason": "<short>", "confidence": <0.0-1.0>}')

def score_route(item, resp):
    c = client.content(resp)
    obj = _first_json(c, "object")
    rec = {"id": item["id"], "cat": item["cat"], "gold": item["gold"]}
    if not obj or "target" not in obj:
        rec.update({"format_ok": False, "target": None, "exact": False,
                    "acceptable": False, "detail": ("no JSON target: %r" % c[:140])})
        return rec
    target = str(obj.get("target", "")).strip()
    valid = target in WORKERS
    ok_set = item["ok"]
    exact = target == item["gold"]
    acceptable = target in ok_set
    local_only = item.get("local_only", False)
    priv_violation = bool(local_only and target in CLOUD)
    esc = cost(target) - cost(item["gold"]) if valid and target != "clarify" and item["gold"] != "clarify" else 0
    rec.update({
        "format_ok": valid, "target": target, "confidence": obj.get("confidence"),
        "exact": exact, "acceptable": acceptable and not priv_violation,
        "priv_violation": priv_violation, "esc_err": esc,
        "detail": "" if (acceptable and not priv_violation) else
                  ("PRIVACY: sent local-only to cloud" if priv_violation else
                   "routed %s, ok=%s" % (target, sorted(ok_set))),
    })
    return rec

# ---------------------------------------------------------------- ORCH phase
ORCH_SYSTEM = (
    "You are an orchestrator. Break the user's compound request into an ordered plan of atomic steps. "
    "For EACH step choose a task `kind` and assign the cheapest capable `worker`, and list which "
    "earlier steps it depends on. Same routing rules as dispatch: keep cheap work local, escalate only "
    "when needed, and NEVER assign a CLOUD worker to a step that handles on-prem/confidential data.\n\n"
    "Valid kinds: " + ", ".join(sorted(KINDS)) + "\n"
    "Valid workers: " + ", ".join(w for w in WORKERS if w != "clarify") + "\n\n"
    "Respond with ONLY a JSON array (no prose), each element:\n"
    '{"step": <int, 1-based>, "task": "<what to do>", "kind": "<one kind>", '
    '"worker": "<one worker>", "depends_on": [<earlier step ints>]}')

def score_orch(item, resp):
    c = client.content(resp)
    plan = _first_json(c, "array")
    rec = {"id": item["id"], "required_kinds": item["required_kinds"]}
    if not isinstance(plan, list) or not plan:
        rec.update({"format_ok": False, "coverage": False, "assign_ok": 0.0,
                    "ordering_ok": False, "passed": False,
                    "detail": "no JSON array plan: %r" % c[:140]})
        return rec
    # normalize steps
    steps = []
    for s in plan:
        if not isinstance(s, dict):
            continue
        steps.append({
            "step": s.get("step"),
            "kind": str(s.get("kind", "")).strip(),
            "worker": str(s.get("worker", "")).strip(),
            "depends_on": s.get("depends_on") or [],
        })
    kinds_present = {s["kind"] for s in steps}
    coverage = set(item["required_kinds"]) <= kinds_present
    # assignment: fraction of known-kind steps routed to a capable worker (+privacy)
    local_only = item.get("local_only", False)
    known = [s for s in steps if s["kind"] in KINDS]
    def assign_good(s):
        if local_only and s["worker"] in CLOUD:
            return False
        return s["worker"] in CAPABLE.get(s["kind"], set())
    assign_ok = (sum(1 for s in known if assign_good(s)) / len(known)) if known else 0.0
    priv_violation = bool(local_only and any(s["worker"] in CLOUD for s in steps))
    # ordering: build step->kind and a reachability check over depends_on
    by_step = {s["step"]: s for s in steps if isinstance(s["step"], int)}
    def reaches_kind(start_step, target_kind, seen=None):
        seen = seen or set()
        for dep in (by_step.get(start_step, {}).get("depends_on") or []):
            if not isinstance(dep, int) or dep in seen:
                continue
            seen.add(dep)
            d = by_step.get(dep)
            if not d:
                continue
            if d["kind"] == target_kind or reaches_kind(dep, target_kind, seen):
                return True
        return False
    ordering_ok = True
    for a_kind, b_kind in item["deps"]:
        b_steps = [s["step"] for s in steps if s["kind"] == b_kind and isinstance(s["step"], int)]
        if not b_steps or not any(reaches_kind(bs, a_kind) for bs in b_steps):
            ordering_ok = False
            break
    passed = bool(coverage and assign_ok == 1.0 and ordering_ok and not priv_violation)
    rec.update({
        "format_ok": True, "n_steps": len(steps), "kinds": sorted(kinds_present),
        "coverage": coverage, "assign_ok": round(assign_ok, 2),
        "ordering_ok": ordering_ok, "priv_violation": priv_violation, "passed": passed,
        "detail": "" if passed else "cov=%s assign=%.2f order=%s priv=%s" %
                  (coverage, assign_ok, ordering_ok, priv_violation),
    })
    return rec

# ---------------------------------------------------------------- phases
def validate():
    """Prove the dataset is well-formed before trusting any model score."""
    ok = True
    for it in ROUTE:
        if it["gold"] not in it["ok"]:
            print("BAD route %s: gold %s not in ok %s" % (it["id"], it["gold"], it["ok"])); ok = False
        if it["gold"] not in WORKERS:
            print("BAD route %s: gold %s not a worker" % (it["id"], it["gold"])); ok = False
        if it.get("local_only") and (it["ok"] & CLOUD):
            print("BAD route %s: local_only but ok set includes cloud" % it["id"]); ok = False
    for it in ORCH:
        for k in it["required_kinds"]:
            if k not in KINDS:
                print("BAD orch %s: kind %s unknown" % (it["id"], k)); ok = False
        for a, b in it["deps"]:
            if a not in it["required_kinds"] or b not in it["required_kinds"]:
                print("BAD orch %s: dep (%s,%s) not both required" % (it["id"], a, b)); ok = False
    print("ROUTE items: %d   ORCH items: %d" % (len(ROUTE), len(ORCH)))
    print("VALIDATION %s" % ("PASSED" if ok else "FAILED"))
    return ok

def run():
    path = jsonl_path(client.MODEL, prefix="router")
    open(path, "w").close()
    recs = []
    print("=== ROUTER BENCH vs %s (%d route + %d orch) ===\n" %
          (client.MODEL, len(ROUTE), len(ORCH)), flush=True)
    print("-- ROUTE --", flush=True)
    for it in ROUTE:
        try:
            resp, dt = client.call_model([{"role": "system", "content": ROUTE_SYSTEM},
                                          {"role": "user", "content": it["request"]}])
            rec = score_route(it, resp)
            rec["latency_s"] = round(dt, 1)
            rec["phase"] = "route"
        except Exception as e:
            rec = {"id": it["id"], "cat": it["cat"], "phase": "route", "format_ok": False,
                   "exact": False, "acceptable": False, "detail": "ERROR %r" % e}
        recs.append(rec)
        append_jsonl(path, rec)
        print("  %-14s -> %-12s %s%s" % (it["id"], rec.get("target"),
              "OK" if rec.get("acceptable") else "MISS",
              "" if rec.get("acceptable") else "  (" + (rec.get("detail","") or "")[:60] + ")"), flush=True)
    print("\n-- ORCH --", flush=True)
    for it in ORCH:
        try:
            resp, dt = client.call_model([{"role": "system", "content": ORCH_SYSTEM},
                                          {"role": "user", "content": it["request"]}])
            rec = score_orch(it, resp)
            rec["latency_s"] = round(dt, 1)
            rec["phase"] = "orch"
        except Exception as e:
            rec = {"id": it["id"], "phase": "orch", "format_ok": False, "passed": False,
                   "detail": "ERROR %r" % e}
        recs.append(rec)
        append_jsonl(path, rec)
        print("  %-20s %s  %s" % (it["id"], "PASS" if rec.get("passed") else "miss",
              "" if rec.get("passed") else (rec.get("detail","") or "")[:70]), flush=True)
    report_from(recs, client.MODEL)
    write_json(recs, client.MODEL, prefix="router")

def report():
    path = jsonl_path(client.MODEL, prefix="router")
    recs = [json.loads(l) for l in open(path)]
    report_from(recs, client.MODEL)

def report_from(recs, model):
    route = [r for r in recs if r.get("phase") == "route"]
    orch  = [r for r in recs if r.get("phase") == "orch"]
    print("\n===================== ROUTER SUMMARY: %s =====================" % model)
    if route:
        n = len(route)
        fmt = sum(1 for r in route if r.get("format_ok"))
        exact = sum(1 for r in route if r.get("exact"))
        acc = sum(1 for r in route if r.get("acceptable"))
        priv = sum(1 for r in route if r.get("priv_violation"))
        over = sum(1 for r in route if (r.get("esc_err") or 0) > 0)
        under = sum(1 for r in route if (r.get("esc_err") or 0) < 0)
        mae = round(sum(abs(r.get("esc_err") or 0) for r in route) / n, 2)
        lat = round(sum(r.get("latency_s") or 0 for r in route) / n, 1)
        print("ROUTE  acceptable %d/%d (%.0f%%)  exact %d/%d  format %d/%d" %
              (acc, n, 100*acc/n, exact, n, fmt, n))
        print("       over-escalate %d  under-escalate %d  |esc| mean %.2f  privacy-violations %d  avg lat %.1fs" %
              (over, under, mae, priv, lat))
        # per-category acceptable
        cats = {}
        for r in route: cats.setdefault(r["cat"], []).append(r)
        print("       by-cat: " + "  ".join("%s %d/%d" %
              (c, sum(1 for r in rs if r.get("acceptable")), len(rs)) for c, rs in sorted(cats.items())))
    if orch:
        n = len(orch)
        p = sum(1 for r in orch if r.get("passed"))
        cov = sum(1 for r in orch if r.get("coverage"))
        aok = round(sum(r.get("assign_ok") or 0 for r in orch) / n, 2)
        ord_ok = sum(1 for r in orch if r.get("ordering_ok"))
        priv = sum(1 for r in orch if r.get("priv_violation"))
        lat = round(sum(r.get("latency_s") or 0 for r in orch) / n, 1)
        print("ORCH   pass %d/%d (%.0f%%)  coverage %d/%d  mean-assign %.2f  ordering %d/%d  privacy-viol %d  avg lat %.1fs" %
              (p, n, 100*p/n, cov, n, aok, ord_ok, n, priv, lat))
    print("=" * 70)



def main():
    mode = sys.argv[1] if len(sys.argv) > 1 else "validate"
    if mode == "validate":
        sys.exit(0 if validate() else 1)
    elif mode == "run":
        run()
    elif mode == "report":
        report()
    else:
        print(f"usage: python bench.py router [validate|run|report]")
        sys.exit(1)


if __name__ == "__main__":
    main()