openaibot's picture
Submission deploy: app + terrarium + tracked artifacts + requirements
4755af4 verified
Raw History Blame Contribute Delete
6.82 kB
"""Track A orchestrator. Runs the two real jobs over each candidate model on
OpenRouter and writes raw per-call records to bench/results/raw_*.json.
Budget: ~10 live calls per model by default (modest spend):
- directive emission: 6 calls (3 fixtures x 2 reps) [Job 1]
- narration: 2 calls (2 fixtures) [Job 2a]
- command: 2 calls (2 fixtures) [Job 2b]
Usage:
uv run python bench/run_bench.py # all models, full
uv run python bench/run_bench.py --models qwen/qwen3-8b google/gemma-3-4b-it
uv run python bench/run_bench.py --quick # 1 rep, fewer calls
uv run python bench/run_bench.py --anomaly # only the qwen3.5-9b probe
The API key is loaded from .env.local via OpenRouterSettings; it is never printed.
"""
from __future__ import annotations
import argparse
import json
import os
import sys
import time
from typing import Dict, List
_HERE = os.path.dirname(os.path.abspath(__file__))
if _HERE not in sys.path:
sys.path.insert(0, _HERE)
import bench_directive as jd # noqa: E402
import bench_translate as jt # noqa: E402
from bench_common import load_fixtures, load_models, settings # noqa: E402
RESULTS_DIR = os.path.join(_HERE, "results")
def _save(name: str, obj: object) -> str:
os.makedirs(RESULTS_DIR, exist_ok=True)
path = os.path.join(RESULTS_DIR, name)
with open(path, "w") as fh:
json.dump(obj, fh, indent=2)
return path
def run_model(model_spec: Dict[str, object], fixtures: List[Dict], reps: int, reasoning_off: bool) -> Dict[str, object]:
slug = str(model_spec["slug"])
print(f"\n=== {slug} ({model_spec.get('tier')}, {model_spec.get('params_b')}B) ===", flush=True)
rec: Dict[str, object] = {"model": model_spec, "directive": [], "narration": [], "command": []}
# Job 1: directive emission (strict structured output, like production)
for fx in fixtures:
for rep in range(reps):
r = jd.run_one(slug, fx["cc"], strict=True, reasoning_off=reasoning_off)
r["fixture"] = fx["label"]
r["rep"] = rep
rec["directive"].append(r)
_log("directive", fx["label"], r)
# Job 2a: narration (2 fixtures)
for fx in fixtures[:2]:
r = jt.run_narration(slug, fx["cc"], reasoning_off=reasoning_off)
r["fixture"] = fx["label"]
rec["narration"].append(r)
_log("narrate", fx["label"], r)
# Job 2b: command translation (2 fixtures with hand-authored intents)
for fx in fixtures:
intent = jt.INTENTS.get(fx["label"])
if not intent:
continue
r = jt.run_command(slug, fx["cc"], intent, strict=True, reasoning_off=reasoning_off)
r["fixture"] = fx["label"]
rec["command"].append(r)
_log("command", fx["label"], r)
if len(rec["command"]) >= 2:
break
return rec
def _log(job: str, fixture: str, r: Dict[str, object]) -> None:
bits = [f" [{job:9s}] {fixture:16s}"]
if r.get("error"):
bits.append(f"ERR {str(r['error'])[:80]}")
else:
ttft = r.get("ttft_s")
tps = r.get("tokens_per_sec")
bits.append(f"ttft={ttft:.2f}s" if ttft else "ttft=-")
bits.append(f"tot={r.get('total_s', 0):.2f}s")
bits.append(f"tps={tps:.0f}" if tps else "tps=-")
if "schema_valid" in r:
bits.append("valid" if r["schema_valid"] else "INVALID")
if "command_correct" in r:
bits.append("cmd_ok" if r["command_correct"] else "cmd_bad")
if "grounded" in r:
bits.append("grounded" if r["grounded"] else "ungrounded")
if r.get("reasoning_chars"):
bits.append(f"rc={r['reasoning_chars']}")
print(" ".join(bits), flush=True)
def run_anomaly(fixtures: List[Dict]) -> Dict[str, object]:
"""Diagnose the 76-110s structured-output anomaly. The original was qwen3.5-9b
on deepinfra/together with strict json_schema. We sweep the controllable knobs:
strict vs non-strict json_schema, and reasoning on vs off. Provider is logged
per call (OpenRouter picks it); cold-start shows as an inflated first call.
"""
model = "qwen/qwen3.5-9b"
fx = fixtures[0]
print(f"\n### ANOMALY PROBE: {model} ###", flush=True)
out: List[Dict[str, object]] = []
matrix = [
{"strict": True, "reasoning_off": True, "tag": "strict+noreason"},
{"strict": False, "reasoning_off": True, "tag": "nonstrict+noreason"},
{"strict": True, "reasoning_off": False, "tag": "strict+reason"},
{"strict": False, "reasoning_off": False, "tag": "nonstrict+reason"},
]
for cell in matrix:
for rep in range(2): # rep0 may be cold, rep1 warm
r = jd.run_one(model, fx["cc"], strict=cell["strict"], reasoning_off=cell["reasoning_off"])
r["cell"] = cell["tag"]
r["rep"] = rep
out.append(r)
_log(cell["tag"], f"rep{rep}", r)
return {"model": model, "matrix": out}
def main() -> None:
ap = argparse.ArgumentParser()
ap.add_argument("--models", nargs="*", help="explicit slugs; default = all in models.json")
ap.add_argument("--quick", action="store_true", help="1 rep per directive fixture")
ap.add_argument("--anomaly", action="store_true", help="run only the qwen3.5-9b anomaly probe")
ap.add_argument("--reasoning-on", action="store_true", help="do NOT request reasoning off")
args = ap.parse_args()
st = settings()
if not st.api_key:
print("ERROR: no OPENROUTER_API_KEY in .env.local", file=sys.stderr)
sys.exit(1)
print(f"key loaded (len {len(st.api_key)}), base={st.base_url}", flush=True)
fixtures = load_fixtures()
reps = 1 if args.quick else 2
reasoning_off = not args.reasoning_on
stamp = time.strftime("%Y%m%d_%H%M%S")
if args.anomaly:
out = run_anomaly(fixtures)
path = _save(f"anomaly_{stamp}.json", out)
print(f"\nsaved {path}")
return
all_models = load_models()
if args.models:
wanted = set(args.models)
all_models = [m for m in all_models if m["slug"] in wanted]
if not all_models:
print(f"no matching models for {args.models}", file=sys.stderr)
sys.exit(1)
results = []
for m in all_models:
try:
results.append(run_model(m, fixtures, reps, reasoning_off))
except Exception as exc: # noqa: BLE001
print(f" model {m['slug']} crashed: {type(exc).__name__}: {exc}", flush=True)
results.append({"model": m, "crash": f"{type(exc).__name__}: {exc}"})
_save(f"raw_{stamp}.json", results) # incremental save
path = _save(f"raw_{stamp}.json", results)
print(f"\nsaved {path}")
if __name__ == "__main__":
main()