"""Track A orchestrator. Runs the two real jobs over each candidate model on OpenRouter and writes raw per-call records to bench/results/raw_*.json. Budget: ~10 live calls per model by default (modest spend): - directive emission: 6 calls (3 fixtures x 2 reps) [Job 1] - narration: 2 calls (2 fixtures) [Job 2a] - command: 2 calls (2 fixtures) [Job 2b] Usage: uv run python bench/run_bench.py # all models, full uv run python bench/run_bench.py --models qwen/qwen3-8b google/gemma-3-4b-it uv run python bench/run_bench.py --quick # 1 rep, fewer calls uv run python bench/run_bench.py --anomaly # only the qwen3.5-9b probe The API key is loaded from .env.local via OpenRouterSettings; it is never printed. """ from __future__ import annotations import argparse import json import os import sys import time from typing import Dict, List _HERE = os.path.dirname(os.path.abspath(__file__)) if _HERE not in sys.path: sys.path.insert(0, _HERE) import bench_directive as jd # noqa: E402 import bench_translate as jt # noqa: E402 from bench_common import load_fixtures, load_models, settings # noqa: E402 RESULTS_DIR = os.path.join(_HERE, "results") def _save(name: str, obj: object) -> str: os.makedirs(RESULTS_DIR, exist_ok=True) path = os.path.join(RESULTS_DIR, name) with open(path, "w") as fh: json.dump(obj, fh, indent=2) return path def run_model(model_spec: Dict[str, object], fixtures: List[Dict], reps: int, reasoning_off: bool) -> Dict[str, object]: slug = str(model_spec["slug"]) print(f"\n=== {slug} ({model_spec.get('tier')}, {model_spec.get('params_b')}B) ===", flush=True) rec: Dict[str, object] = {"model": model_spec, "directive": [], "narration": [], "command": []} # Job 1: directive emission (strict structured output, like production) for fx in fixtures: for rep in range(reps): r = jd.run_one(slug, fx["cc"], strict=True, reasoning_off=reasoning_off) r["fixture"] = fx["label"] r["rep"] = rep rec["directive"].append(r) _log("directive", fx["label"], r) # Job 2a: narration (2 fixtures) for fx in fixtures[:2]: r = jt.run_narration(slug, fx["cc"], reasoning_off=reasoning_off) r["fixture"] = fx["label"] rec["narration"].append(r) _log("narrate", fx["label"], r) # Job 2b: command translation (2 fixtures with hand-authored intents) for fx in fixtures: intent = jt.INTENTS.get(fx["label"]) if not intent: continue r = jt.run_command(slug, fx["cc"], intent, strict=True, reasoning_off=reasoning_off) r["fixture"] = fx["label"] rec["command"].append(r) _log("command", fx["label"], r) if len(rec["command"]) >= 2: break return rec def _log(job: str, fixture: str, r: Dict[str, object]) -> None: bits = [f" [{job:9s}] {fixture:16s}"] if r.get("error"): bits.append(f"ERR {str(r['error'])[:80]}") else: ttft = r.get("ttft_s") tps = r.get("tokens_per_sec") bits.append(f"ttft={ttft:.2f}s" if ttft else "ttft=-") bits.append(f"tot={r.get('total_s', 0):.2f}s") bits.append(f"tps={tps:.0f}" if tps else "tps=-") if "schema_valid" in r: bits.append("valid" if r["schema_valid"] else "INVALID") if "command_correct" in r: bits.append("cmd_ok" if r["command_correct"] else "cmd_bad") if "grounded" in r: bits.append("grounded" if r["grounded"] else "ungrounded") if r.get("reasoning_chars"): bits.append(f"rc={r['reasoning_chars']}") print(" ".join(bits), flush=True) def run_anomaly(fixtures: List[Dict]) -> Dict[str, object]: """Diagnose the 76-110s structured-output anomaly. The original was qwen3.5-9b on deepinfra/together with strict json_schema. We sweep the controllable knobs: strict vs non-strict json_schema, and reasoning on vs off. Provider is logged per call (OpenRouter picks it); cold-start shows as an inflated first call. """ model = "qwen/qwen3.5-9b" fx = fixtures[0] print(f"\n### ANOMALY PROBE: {model} ###", flush=True) out: List[Dict[str, object]] = [] matrix = [ {"strict": True, "reasoning_off": True, "tag": "strict+noreason"}, {"strict": False, "reasoning_off": True, "tag": "nonstrict+noreason"}, {"strict": True, "reasoning_off": False, "tag": "strict+reason"}, {"strict": False, "reasoning_off": False, "tag": "nonstrict+reason"}, ] for cell in matrix: for rep in range(2): # rep0 may be cold, rep1 warm r = jd.run_one(model, fx["cc"], strict=cell["strict"], reasoning_off=cell["reasoning_off"]) r["cell"] = cell["tag"] r["rep"] = rep out.append(r) _log(cell["tag"], f"rep{rep}", r) return {"model": model, "matrix": out} def main() -> None: ap = argparse.ArgumentParser() ap.add_argument("--models", nargs="*", help="explicit slugs; default = all in models.json") ap.add_argument("--quick", action="store_true", help="1 rep per directive fixture") ap.add_argument("--anomaly", action="store_true", help="run only the qwen3.5-9b anomaly probe") ap.add_argument("--reasoning-on", action="store_true", help="do NOT request reasoning off") args = ap.parse_args() st = settings() if not st.api_key: print("ERROR: no OPENROUTER_API_KEY in .env.local", file=sys.stderr) sys.exit(1) print(f"key loaded (len {len(st.api_key)}), base={st.base_url}", flush=True) fixtures = load_fixtures() reps = 1 if args.quick else 2 reasoning_off = not args.reasoning_on stamp = time.strftime("%Y%m%d_%H%M%S") if args.anomaly: out = run_anomaly(fixtures) path = _save(f"anomaly_{stamp}.json", out) print(f"\nsaved {path}") return all_models = load_models() if args.models: wanted = set(args.models) all_models = [m for m in all_models if m["slug"] in wanted] if not all_models: print(f"no matching models for {args.models}", file=sys.stderr) sys.exit(1) results = [] for m in all_models: try: results.append(run_model(m, fixtures, reps, reasoning_off)) except Exception as exc: # noqa: BLE001 print(f" model {m['slug']} crashed: {type(exc).__name__}: {exc}", flush=True) results.append({"model": m, "crash": f"{type(exc).__name__}: {exc}"}) _save(f"raw_{stamp}.json", results) # incremental save path = _save(f"raw_{stamp}.json", results) print(f"\nsaved {path}") if __name__ == "__main__": main()