Download bench/run_bench.py from build-small-hackathon/thousand-token-terrarium: direct link, hf CLI and curl.
- Browser
- Download file 6.82 kB
-
https://huggingface.co/spaces/build-small-hackathon/thousand-token-terrarium/resolve/main/bench/run_bench.py
- Command line
-
hf download hf://spaces/build-small-hackathon/thousand-token-terrarium/bench/run_bench.py
-
curl -L -o run_bench.py https://huggingface.co/spaces/build-small-hackathon/thousand-token-terrarium/resolve/main/bench/run_bench.py
6.82 kB
| """Track A orchestrator. Runs the two real jobs over each candidate model on | |
| OpenRouter and writes raw per-call records to bench/results/raw_*.json. | |
| Budget: ~10 live calls per model by default (modest spend): | |
| - directive emission: 6 calls (3 fixtures x 2 reps) [Job 1] | |
| - narration: 2 calls (2 fixtures) [Job 2a] | |
| - command: 2 calls (2 fixtures) [Job 2b] | |
| Usage: | |
| uv run python bench/run_bench.py # all models, full | |
| uv run python bench/run_bench.py --models qwen/qwen3-8b google/gemma-3-4b-it | |
| uv run python bench/run_bench.py --quick # 1 rep, fewer calls | |
| uv run python bench/run_bench.py --anomaly # only the qwen3.5-9b probe | |
| The API key is loaded from .env.local via OpenRouterSettings; it is never printed. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| import os | |
| import sys | |
| import time | |
| from typing import Dict, List | |
| _HERE = os.path.dirname(os.path.abspath(__file__)) | |
| if _HERE not in sys.path: | |
| sys.path.insert(0, _HERE) | |
| import bench_directive as jd # noqa: E402 | |
| import bench_translate as jt # noqa: E402 | |
| from bench_common import load_fixtures, load_models, settings # noqa: E402 | |
| RESULTS_DIR = os.path.join(_HERE, "results") | |
| def _save(name: str, obj: object) -> str: | |
| os.makedirs(RESULTS_DIR, exist_ok=True) | |
| path = os.path.join(RESULTS_DIR, name) | |
| with open(path, "w") as fh: | |
| json.dump(obj, fh, indent=2) | |
| return path | |
| def run_model(model_spec: Dict[str, object], fixtures: List[Dict], reps: int, reasoning_off: bool) -> Dict[str, object]: | |
| slug = str(model_spec["slug"]) | |
| print(f"\n=== {slug} ({model_spec.get('tier')}, {model_spec.get('params_b')}B) ===", flush=True) | |
| rec: Dict[str, object] = {"model": model_spec, "directive": [], "narration": [], "command": []} | |
| # Job 1: directive emission (strict structured output, like production) | |
| for fx in fixtures: | |
| for rep in range(reps): | |
| r = jd.run_one(slug, fx["cc"], strict=True, reasoning_off=reasoning_off) | |
| r["fixture"] = fx["label"] | |
| r["rep"] = rep | |
| rec["directive"].append(r) | |
| _log("directive", fx["label"], r) | |
| # Job 2a: narration (2 fixtures) | |
| for fx in fixtures[:2]: | |
| r = jt.run_narration(slug, fx["cc"], reasoning_off=reasoning_off) | |
| r["fixture"] = fx["label"] | |
| rec["narration"].append(r) | |
| _log("narrate", fx["label"], r) | |
| # Job 2b: command translation (2 fixtures with hand-authored intents) | |
| for fx in fixtures: | |
| intent = jt.INTENTS.get(fx["label"]) | |
| if not intent: | |
| continue | |
| r = jt.run_command(slug, fx["cc"], intent, strict=True, reasoning_off=reasoning_off) | |
| r["fixture"] = fx["label"] | |
| rec["command"].append(r) | |
| _log("command", fx["label"], r) | |
| if len(rec["command"]) >= 2: | |
| break | |
| return rec | |
| def _log(job: str, fixture: str, r: Dict[str, object]) -> None: | |
| bits = [f" [{job:9s}] {fixture:16s}"] | |
| if r.get("error"): | |
| bits.append(f"ERR {str(r['error'])[:80]}") | |
| else: | |
| ttft = r.get("ttft_s") | |
| tps = r.get("tokens_per_sec") | |
| bits.append(f"ttft={ttft:.2f}s" if ttft else "ttft=-") | |
| bits.append(f"tot={r.get('total_s', 0):.2f}s") | |
| bits.append(f"tps={tps:.0f}" if tps else "tps=-") | |
| if "schema_valid" in r: | |
| bits.append("valid" if r["schema_valid"] else "INVALID") | |
| if "command_correct" in r: | |
| bits.append("cmd_ok" if r["command_correct"] else "cmd_bad") | |
| if "grounded" in r: | |
| bits.append("grounded" if r["grounded"] else "ungrounded") | |
| if r.get("reasoning_chars"): | |
| bits.append(f"rc={r['reasoning_chars']}") | |
| print(" ".join(bits), flush=True) | |
| def run_anomaly(fixtures: List[Dict]) -> Dict[str, object]: | |
| """Diagnose the 76-110s structured-output anomaly. The original was qwen3.5-9b | |
| on deepinfra/together with strict json_schema. We sweep the controllable knobs: | |
| strict vs non-strict json_schema, and reasoning on vs off. Provider is logged | |
| per call (OpenRouter picks it); cold-start shows as an inflated first call. | |
| """ | |
| model = "qwen/qwen3.5-9b" | |
| fx = fixtures[0] | |
| print(f"\n### ANOMALY PROBE: {model} ###", flush=True) | |
| out: List[Dict[str, object]] = [] | |
| matrix = [ | |
| {"strict": True, "reasoning_off": True, "tag": "strict+noreason"}, | |
| {"strict": False, "reasoning_off": True, "tag": "nonstrict+noreason"}, | |
| {"strict": True, "reasoning_off": False, "tag": "strict+reason"}, | |
| {"strict": False, "reasoning_off": False, "tag": "nonstrict+reason"}, | |
| ] | |
| for cell in matrix: | |
| for rep in range(2): # rep0 may be cold, rep1 warm | |
| r = jd.run_one(model, fx["cc"], strict=cell["strict"], reasoning_off=cell["reasoning_off"]) | |
| r["cell"] = cell["tag"] | |
| r["rep"] = rep | |
| out.append(r) | |
| _log(cell["tag"], f"rep{rep}", r) | |
| return {"model": model, "matrix": out} | |
| def main() -> None: | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--models", nargs="*", help="explicit slugs; default = all in models.json") | |
| ap.add_argument("--quick", action="store_true", help="1 rep per directive fixture") | |
| ap.add_argument("--anomaly", action="store_true", help="run only the qwen3.5-9b anomaly probe") | |
| ap.add_argument("--reasoning-on", action="store_true", help="do NOT request reasoning off") | |
| args = ap.parse_args() | |
| st = settings() | |
| if not st.api_key: | |
| print("ERROR: no OPENROUTER_API_KEY in .env.local", file=sys.stderr) | |
| sys.exit(1) | |
| print(f"key loaded (len {len(st.api_key)}), base={st.base_url}", flush=True) | |
| fixtures = load_fixtures() | |
| reps = 1 if args.quick else 2 | |
| reasoning_off = not args.reasoning_on | |
| stamp = time.strftime("%Y%m%d_%H%M%S") | |
| if args.anomaly: | |
| out = run_anomaly(fixtures) | |
| path = _save(f"anomaly_{stamp}.json", out) | |
| print(f"\nsaved {path}") | |
| return | |
| all_models = load_models() | |
| if args.models: | |
| wanted = set(args.models) | |
| all_models = [m for m in all_models if m["slug"] in wanted] | |
| if not all_models: | |
| print(f"no matching models for {args.models}", file=sys.stderr) | |
| sys.exit(1) | |
| results = [] | |
| for m in all_models: | |
| try: | |
| results.append(run_model(m, fixtures, reps, reasoning_off)) | |
| except Exception as exc: # noqa: BLE001 | |
| print(f" model {m['slug']} crashed: {type(exc).__name__}: {exc}", flush=True) | |
| results.append({"model": m, "crash": f"{type(exc).__name__}: {exc}"}) | |
| _save(f"raw_{stamp}.json", results) # incremental save | |
| path = _save(f"raw_{stamp}.json", results) | |
| print(f"\nsaved {path}") | |
| if __name__ == "__main__": | |
| main() | |