diff --git a/hh/scripts/bench-olympics.py b/hh/scripts/bench-olympics.py
new file mode 100644
index 0000000..b4ebcc2
--- /dev/null
+++ b/hh/scripts/bench-olympics.py
@@ -0,0 +1,342 @@
+#!/usr/bin/env python3
+"""Agent Olympics launcher — competition benchmark inside hack-house.
+
+A fourth benchmark axis (sibling of bench-ai.py / bench-sandbox.py /
+bench-lang.py / bench-safety.py). Teams of LLM agents deliberate in a real
+hack-house room, implement code in an isolated VM, and are scored on
+correctness/speed (quality/collaboration arrive with the judge in M4).
+
+M1 — the arena spine. One same-model team solves one MBPP problem end-to-end:
+room deliberation -> single-driver implement -> PodmanRuntime -> public tests ->
+SUBMIT -> hidden-test grade -> replayable transcript -> deterministic score.
+
+Subcommands:
+ run run one event for one team (M1: same-model 2-member team, one MBPP task)
+ replay re-render a saved transcript as a room log
+ score re-score a saved run under a different profile (no re-run)
+ show print the challenge a task_id/index resolves to
+
+Examples:
+ python hh/scripts/bench-olympics.py run --model qwen2.5-coder:3b --task 11
+ python hh/scripts/bench-olympics.py run --model qwen2.5-coder:3b --index 0 \
+ --room real --max-rounds 2
+ python hh/scripts/bench-olympics.py replay /tmp/hh-olympics/runs/
/transcript.json
+"""
+
+from __future__ import annotations
+
+import argparse
+import json
+import signal
+import sys
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parent))
+
+from bench.olympics import arena, budget as obudget, challenge as ch # noqa: E402
+from bench.olympics import ledger, roster, scoring # noqa: E402
+from bench.olympics import transcript as T # noqa: E402
+from bench.olympics.loop import Budget # noqa: E402
+from bench.olympics.scoring import EventOutcome # noqa: E402
+
+_BASE_WALL_CLOCK = 300.0 # base seconds before model-aware scaling
+
+
+class RunTimeout(BaseException):
+ """Raised by the watchdog so the normal try/except records a killed row and
+ the run's own finally blocks still tear down the VM/relay.
+
+ Derives from ``BaseException`` (not ``Exception``) on purpose: the inference
+ and completion paths wrap their HTTP calls in broad ``except Exception``
+ handlers, and the SIGALRM fires *while* those calls block. An
+ ``Exception``-derived timeout would be swallowed there and treated as a
+ failed turn, so the deadline would never propagate. ``BaseException`` skips
+ those handlers (like ``KeyboardInterrupt``) while ``finally`` cleanup and our
+ explicit ``except RunTimeout`` still run."""
+
+
+class _Deadline:
+ """Hard self-deadline: arm SIGALRM and catch external SIGTERM; both raise
+ RunTimeout. The harness thus owns and records its own deadline instead of
+ relying on an external ``timeout`` (whose SIGTERM/SIGKILL would drop the run
+ from the ledger). SIGKILL still can't be caught — pair with ``timeout -k``
+ for a truly wedged process."""
+
+ def __init__(self, seconds: float):
+ self.seconds = max(1, int(seconds))
+ self._prev_alrm = None
+ self._prev_term = None
+
+ def _fire(self, signum, _frame):
+ raise RunTimeout(f"hard deadline {self.seconds}s (signal {signum})")
+
+ def __enter__(self):
+ self._prev_term = signal.signal(signal.SIGTERM, self._fire)
+ if hasattr(signal, "SIGALRM"):
+ self._prev_alrm = signal.signal(signal.SIGALRM, self._fire)
+ signal.alarm(self.seconds)
+ return self
+
+ def __exit__(self, *_exc):
+ if hasattr(signal, "SIGALRM"):
+ signal.alarm(0)
+ if self._prev_alrm is not None:
+ signal.signal(signal.SIGALRM, self._prev_alrm)
+ if self._prev_term is not None:
+ signal.signal(signal.SIGTERM, self._prev_term)
+ return False
+
+
+def _resolve_budget(args, model: str) -> tuple[float, float]:
+ """(wall_clock_s, scale). An explicit --wall-clock is honored verbatim
+ (scale 1.0); otherwise the base is scaled by the model-aware multiplier."""
+ if args.wall_clock is not None:
+ return args.wall_clock, 1.0
+ scale = obudget.budget_scale(model)
+ return _BASE_WALL_CLOCK * scale, scale
+
+
+def _load_challenge(args) -> "ch.Challenge":
+ if args.task is not None:
+ return ch.load_mbpp(task_id=args.task, language=args.language)
+ return ch.load_mbpp(index=args.index, language=args.language)
+
+
+def cmd_run(args) -> int:
+ challenge = _load_challenge(args)
+ team = roster.same_model_team(args.model, team_id=args.team,
+ framing=args.framing)
+
+ def prog(p):
+ print(f" · {p}")
+
+ wall_clock, scale = _resolve_budget(args, team.driver().model)
+ hard = args.hard_timeout if args.hard_timeout is not None else wall_clock * 1.4
+ mode = "bridge" if args.mode == "bridge" else "direct"
+ room = "real" if mode == "bridge" else args.room
+ runtime = "podman" if mode == "bridge" else args.runtime
+ # shared context so a killed/errored run still records a complete ledger row.
+ led_ctx = dict(team=team, challenge=challenge, mode=mode, room=room,
+ runtime=runtime, seed=args.seed,
+ code_model=args.code_model if mode == "bridge" else None,
+ wall_clock_budget=round(wall_clock, 1), budget_scale=scale,
+ path=args.ledger)
+
+ def _run() -> dict:
+ if mode == "bridge":
+ from bench.olympics import bridge # noqa: E402
+ print(f"olympics bridge · team={team.id} models={team.models}")
+ print(f" challenge={challenge.id} REAL room + /ai !task build "
+ f"framing={args.framing}")
+ print(f" wall_clock={wall_clock:.0f}s (scale {scale}x) "
+ f"hard_deadline={hard:.0f}s")
+ print("-" * 72)
+ return bridge.run_bridge_event(
+ team, challenge, host=args.host_room, port=args.port,
+ password=args.password, ollama=args.ollama,
+ code_model=args.code_model, out_dir=args.out, seed=args.seed,
+ profile=args.profile, step_timeout=wall_clock,
+ agent_chat_confirm=args.agent_chat_confirm, progress=prog)
+ budget = Budget(deliberate_rounds=args.deliberate_rounds,
+ implement_attempts=args.max_rounds,
+ max_tokens=args.max_tokens, wall_clock_s=wall_clock)
+ print(f"olympics M1 · team={team.id} models={team.models}")
+ print(f" challenge={challenge.id} room={args.room} "
+ f"runtime={args.runtime} framing={args.framing}")
+ print(f" wall_clock={wall_clock:.0f}s (scale {scale}x) "
+ f"hard_deadline={hard:.0f}s")
+ print("-" * 72)
+ room_kwargs = {}
+ if args.room == "real":
+ room_kwargs = {"host": args.host_room, "port": args.port,
+ "password": args.password}
+ return arena.run_one(
+ team, challenge, room_kind=args.room, runtime=args.runtime,
+ budget=budget, profile=args.profile, seed=args.seed,
+ host=args.ollama, out_dir=args.out, room_kwargs=room_kwargs,
+ progress=prog)
+
+ try:
+ with _Deadline(hard):
+ score = _run()
+ except RunTimeout as e:
+ row = ledger.append_incomplete(status=ledger.STATUS_KILLED,
+ error=str(e), **led_ctx)
+ print(f"\n!! run killed: {e}")
+ print(f"ledger += {row['run_id']} [killed] → "
+ f"{args.ledger or ledger.DEFAULT_PATH}")
+ return 2
+ except Exception as e: # noqa: BLE001 — record then surface
+ row = ledger.append_incomplete(status=ledger.STATUS_ERROR,
+ error=repr(e), **led_ctx)
+ print(f"\n!! run errored: {e!r}")
+ print(f"ledger += {row['run_id']} [error] → "
+ f"{args.ledger or ledger.DEFAULT_PATH}")
+ raise
+
+ print("-" * 72)
+ scoring.print_score(score)
+ row = ledger.append(score, team=team, challenge=challenge, mode=mode,
+ room=score.get("room_substrate", room), runtime=runtime,
+ seed=args.seed,
+ code_model=led_ctx["code_model"],
+ wall_clock_budget=round(wall_clock, 1),
+ budget_scale=scale, path=args.ledger)
+ print(f"ledger += {row['run_id']} → {args.ledger or ledger.DEFAULT_PATH}")
+ print(f"transcript: {score['transcript']}")
+ print(f"replay with: python {Path(__file__).name} replay {score['transcript']}")
+ return 0 if score["correct"] else 1
+
+
+def cmd_leaderboard(args) -> int:
+ rows = ledger.load(args.ledger)
+ if not rows:
+ print(f"no runs recorded in {args.ledger or ledger.DEFAULT_PATH}")
+ return 1
+ filters = {"team": args.team, "language": args.language,
+ "models": args.model, "mode": args.mode, "since": args.since}
+ agg = ledger.leaderboard(rows, by=args.by, filters=filters)
+ ledger.print_leaderboard(agg, args.by)
+ print(f"({len(rows)} total runs in {args.ledger or ledger.DEFAULT_PATH})")
+ return 0
+
+
+def cmd_replay(args) -> int:
+ T.replay(args.transcript, show_tools=not args.no_tools)
+ return 0
+
+
+def cmd_score(args) -> int:
+ tx = T.load_transcript(args.transcript) if hasattr(T, "load_transcript") \
+ else T.Transcript.load(args.transcript)
+ # reconstruct a minimal EventOutcome from the transcript to re-score
+ o = _outcome_from_transcript(tx)
+ budget = tx.manifest.get("budget", {})
+ s = scoring.score_event(o, profile=args.profile,
+ budget={"max_rounds": budget.get("implement_attempts", 3)})
+ scoring.print_score(s)
+ return 0
+
+
+def _outcome_from_transcript(tx: "T.Transcript") -> EventOutcome:
+ correct = False
+ rounds_to_green = None
+ attempts = 0
+ deliberate_passes = set()
+ submitted = False
+ for ev in tx.events:
+ if ev.kind == T.KIND_TOOL and ev.payload.get("label") == "hidden-grade":
+ correct = bool(ev.payload.get("ok"))
+ if ev.kind == T.KIND_TOOL and str(ev.payload.get("label", "")).startswith("public-attempt"):
+ attempts += 1
+ if ev.payload.get("ok") and rounds_to_green is None:
+ rounds_to_green = attempts
+ if ev.kind == T.KIND_MESSAGE and ev.payload.get("text") == "SUBMIT":
+ submitted = True
+ if ev.kind == T.KIND_AGENT and ev.phase == "DELIBERATE":
+ deliberate_passes.add(round(ev.ts))
+ return EventOutcome(
+ team=tx.team, challenge=tx.challenge, submitted=submitted,
+ correct=correct, rounds=attempts, wall_clock_s=0.0,
+ tokens=tx.total_tokens(), public_passed=rounds_to_green is not None,
+ rounds_to_green=rounds_to_green, penalties=[])
+
+
+def cmd_show(args) -> int:
+ c = _load_challenge(args)
+ print(f"id={c.id} language={c.language} meta={c.meta}")
+ print("public_tests:", c.public_tests)
+ print("hidden_tests:", c.hidden_tests)
+ print("--- brief ---"); print(c.brief())
+ print("--- gen_prompt ---"); print(c.gen_prompt())
+ return 0
+
+
+def build_parser() -> argparse.ArgumentParser:
+ ap = argparse.ArgumentParser(
+ prog="bench-olympics",
+ description="agent-olympics competition benchmark (M1 arena spine)")
+ sub = ap.add_subparsers(dest="cmd", required=True)
+
+ def add_challenge_args(p):
+ p.add_argument("--task", type=int, default=None, help="MBPP task_id")
+ p.add_argument("--index", type=int, default=0,
+ help="index into the MBPP test split (if no --task)")
+ p.add_argument("--language", default="python")
+
+ p = sub.add_parser("run", help="run one event for one team")
+ add_challenge_args(p)
+ p.add_argument("--model", required=True, help="ollama model for both members")
+ p.add_argument("--mode", default="direct", choices=["direct", "bridge"],
+ help="direct: orchestrator builds; bridge: real /ai !task build")
+ p.add_argument("--code-model", default=None,
+ help="ollama model for the agent's sandbox build (bridge mode)")
+ p.add_argument("--agent-chat-confirm", action="store_true",
+ help="bridge: also drive the agent's streaming /ai chat path in "
+ "DELIBERATE (off by default — that product path drops the "
+ "agent via a 1011 keepalive timeout under CPU inference)")
+ p.add_argument("--team", default="falcon")
+ p.add_argument("--framing", default="neutral",
+ choices=["neutral", "competition"])
+ p.add_argument("--room", default="local", choices=["local", "real"])
+ p.add_argument("--runtime", default="auto", choices=["auto", "podman", "local"])
+ p.add_argument("--profile", default="balanced",
+ choices=list(scoring.PROFILES))
+ p.add_argument("--deliberate-rounds", type=int, default=2)
+ p.add_argument("--max-rounds", type=int, default=3,
+ help="implement/test attempts (speed cap)")
+ p.add_argument("--max-tokens", type=int, default=8000)
+ p.add_argument("--wall-clock", type=float, default=None,
+ help="soft wall-clock budget in seconds; omit to apply "
+ "model-aware scaling off the 300s base")
+ p.add_argument("--hard-timeout", type=float, default=None,
+ help="hard deadline in seconds (SIGALRM/SIGTERM watchdog); "
+ "default 1.4x the soft budget")
+ p.add_argument("--seed", type=int, default=0)
+ p.add_argument("--ollama", default="http://127.0.0.1:11434")
+ p.add_argument("--out", default="/tmp/hh-olympics/runs")
+ p.add_argument("--ledger", default=None,
+ help="append-only results JSONL (default: "
+ "~/.cache/hh-bench/olympics/ledger.jsonl)")
+ # real-room knobs
+ p.add_argument("--host-room", default="127.0.0.1")
+ p.add_argument("--port", type=int, default=4677)
+ p.add_argument("--password", default="olympics-pass")
+ p.set_defaults(func=cmd_run)
+
+ p = sub.add_parser("replay", help="re-render a saved transcript")
+ p.add_argument("transcript")
+ p.add_argument("--no-tools", action="store_true", help="hide tool output")
+ p.set_defaults(func=cmd_replay)
+
+ p = sub.add_parser("score", help="re-score a saved run under a profile")
+ p.add_argument("transcript")
+ p.add_argument("--profile", default="balanced", choices=list(scoring.PROFILES))
+ p.set_defaults(func=cmd_score)
+
+ p = sub.add_parser("show", help="print the resolved challenge")
+ add_challenge_args(p)
+ p.set_defaults(func=cmd_show)
+
+ p = sub.add_parser("leaderboard",
+ help="aggregate the results ledger (no re-run)")
+ p.add_argument("--by", default="team",
+ help="group key: team | models | language | challenge | mode "
+ "| framing, or a '+'-joined composite e.g. models+language")
+ p.add_argument("--team", default=None)
+ p.add_argument("--language", default=None)
+ p.add_argument("--model", default=None, help="substring match on team models")
+ p.add_argument("--mode", default=None, choices=[None, "direct", "bridge"])
+ p.add_argument("--since", default=None, help="ISO ts lower bound (inclusive)")
+ p.add_argument("--ledger", default=None)
+ p.set_defaults(func=cmd_leaderboard)
+ return ap
+
+
+def main(argv: list[str] | None = None) -> int:
+ args = build_parser().parse_args(argv)
+ return args.func(args)
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/hh/scripts/bench/completion.py b/hh/scripts/bench/completion.py
index fc46b13..fbcb43a 100644
--- a/hh/scripts/bench/completion.py
+++ b/hh/scripts/bench/completion.py
@@ -24,15 +24,22 @@ class Completion:
def complete(model: str, prompt: str, stop: list[str] | None = None,
*, host: str = "http://127.0.0.1:11434", temperature: float = 0.2,
- num_predict: int = 512, timeout: float = 300.0) -> Completion:
+ num_predict: int = 512, timeout: float = 300.0,
+ seed: int | None = None) -> Completion:
"""Ask the model to continue `prompt`. Stop tokens are passed to Ollama and
re-applied client-side (Ollama strips the stop string, which is what we want
- — the assembled program must not contain the test's leading token twice)."""
+ — the assembled program must not contain the test's leading token twice).
+
+ ``seed`` is forwarded to Ollama's sampler so multi-seed runs at the same
+ temperature draw *different but reproducible* samples (the basis for pass@1
+ confidence intervals)."""
import time
t0 = time.time()
options = {"temperature": temperature, "num_predict": num_predict}
if stop:
options["stop"] = stop
+ if seed is not None:
+ options["seed"] = seed
try:
# raw=True bypasses the chat template so an instruct model *continues*
# the code (HumanEval-style) instead of replying conversationally with
@@ -47,9 +54,34 @@ def complete(model: str, prompt: str, stop: list[str] | None = None,
return Completion(_truncate(text, stop), True, None, time.time() - t0)
+def _strip_fences(text: str) -> str:
+ """Drop a markdown code fence even when raw=True fails to suppress it.
+
+ Chatty coder models sometimes wrap the continuation in ``` fences (and append
+ a prose ``# Explanation`` after a closing fence). A triple-backtick is not
+ valid syntax in any target language, so it is a safe, unambiguous cut point.
+ Two shapes occur: a *wrapped* block (opens with ```lang … closes with ```)
+ and a *bare* continuation that the model terminates with a stray closing ```.
+ """
+ fence = "```"
+ if fence not in text:
+ return text
+ stripped = text.lstrip()
+ if stripped.startswith(fence):
+ # wrapped: drop the opening ```lang line, keep until the next fence.
+ body = stripped[len(fence):]
+ nl = body.find("\n")
+ body = body[nl + 1:] if nl != -1 else ""
+ end = body.find(fence)
+ return body[:end] if end != -1 else body
+ # bare continuation: cut at the stray closing fence.
+ return text[:text.find(fence)]
+
+
def _truncate(text: str, stop: list[str] | None) -> str:
"""Defensive client-side stop truncation (covers the no-stop / streamed
cases and any model that ignores the option)."""
+ text = _strip_fences(text)
if not stop:
return text
cut = len(text)
diff --git a/hh/scripts/bench/olympics/SPEC.md b/hh/scripts/bench/olympics/SPEC.md
new file mode 100644
index 0000000..bae3cc5
--- /dev/null
+++ b/hh/scripts/bench/olympics/SPEC.md
@@ -0,0 +1,483 @@
+# Agent Olympics — a hackathon-competition benchmark inside hack-house
+
+> Status: **DRAFT spec** (design only — no implementation yet). Author: bench team.
+> Scope: a fourth benchmark axis layered on `hh/scripts/bench/`. Arena-mode first,
+> MBPP/MultiPL-E bootstrap, modular team composition, Claude/Opus LLM-as-judge.
+> This document is the build contract. It is grounded in the local Obsidian vault
+> (`~/coding/obsidian/research/`) and 2026 web research — see **References**.
+
+---
+
+## 1. One-paragraph concept
+
+Stand up **two (or more) teams of LLM agents** inside hack-house. Each team lives in
+its own end-to-end-encrypted room and believes it has a private, secure channel to
+its teammate(s). Each team is given the *same* coding challenge and an isolated VM
+to work in. Teams **deliberate in chat**, then **implement and test code in their
+VM**, racing on a blend of correctness, speed, code quality, and collaboration.
+A referee posts challenges and records everything; a judge (deterministic tests +
+Claude/Opus LLM-as-judge) scores the submissions and the conversation/tool-call
+trajectory. Aggregate across a slate of events into a **medal table**. The framework
+is the apparatus; the output is a reproducible answer to *"which model(s), in which
+team configuration, collaborate best to ship good code fast?"*
+
+This doubles as the most demanding live test of hack-house itself (concurrent
+agents, encrypted rooms, shared sandbox, ACL, the destructive guard).
+
+---
+
+## 2. Goals & non-goals
+
+### Goals
+- **G1 — Real comms substrate.** Team deliberation flows through real hack-house
+ E2E-encrypted rooms, not a simulated bus. The product is the medium.
+- **G2 — Modular team composition.** Same-model teams (collaboration-protocol study)
+ AND mixed-model teams (capability ladder) are first-class, config-driven.
+- **G3 — Bootstrap fast.** Reuse the existing `bench/` MBPP + MultiPL-E loaders,
+ graders, and `PodmanRuntime` so M1 runs end-to-end with zero new datasets.
+- **G4 — Extensible challenges.** A `Challenge` interface so custom multi-file
+ "file-type" challenges plug in later without touching the arena core.
+- **G5 — Trustworthy scoring.** Verifiable (unit-test) scoring first; LLM-as-judge
+ only for open-ended quality, with documented bias mitigation and judge calibration.
+- **G6 — Reproducible & auditable.** Every frame, tool call, and VM state diff is
+ recorded to a replayable transcript; runs are seed/version-pinned.
+- **G7 — Measure the placebo.** Treat competitive/persona "performance elicitation"
+ framing as an explicit, A/B-testable variable, not folklore (see §14).
+- **G8 — Safe by construction.** Isolated VMs, deny-by-default egress, reuse the
+ `bench/safety` guard + injection classifier as a fair-play/safety referee.
+
+### Non-goals (for now)
+- Not editing `cmd_chat/` (tool code). Arena mode drives turns externally and posts
+ into real rooms. Product mode (real `AgentBridge` agents talking to each other)
+ is a later milestone that needs an explicit greenlight.
+- Not a training loop (no RL/fine-tuning). This selects and compares; it does not train.
+- Not a public leaderboard. Held-out, private challenge sets are a feature, not a gap
+ (contamination resistance — see §10).
+
+---
+
+## 3. Prior art this design borrows from (grounding)
+
+- **Read-vs-write rule for multi-agent work.** Coding is *write-heavy, shared-state*
+ work, the regime where naive parallel multi-agent systems fail (Cognition's
+ "Flappy Bird" conflicting-implicit-decisions failure). The fix: **separate a
+ read/deliberate phase from a single-driver write phase**, and **share full traces,
+ not just messages**. We bake this into the loop (§7) *and* measure the coordination
+ tax. [[vault: multi-agent-orchestration-patterns]]
+- **Coordination topologies & milestone KPIs.** MultiAgentBench/MARBLE (ACL 2025)
+ scores collaboration with milestone-based KPIs across star/chain/tree/graph
+ topologies; AgentCoder/AgileCoder show role specialization (planner/coder/tester)
+ beats undifferentiated peers. We make role + topology config knobs. [web]
+- **Eval-harness discipline.** Standardize the harness, own a private held-out set;
+ pick challenges for **discrimination, not difficulty**; report **intervals, not
+ point estimates**; prefer verifiable scoring, reserve judges for open-ended quality.
+ [[vault: eval-harnesses-benchmark-design]]
+- **Outcome vs trajectory.** Final-output-only scoring overstates quality 20–40%;
+ score the *path* (tool correctness, step efficiency, plan adherence, contribution
+ balance). Use **agent-as-a-judge** to walk the trajectory. Report **pass^k**
+ (worst-case) beside pass@k for reliability. [[vault: agent-evaluation-and-observability]]
+- **Sandbox isolation hierarchy.** Plain containers are the *floor* (shared kernel =
+ one-CVE-from-escape); microVMs (Firecracker/Kata) or gVisor are the bar for
+ untrusted multi-tenant code; deny-by-default egress + hard caps + disposable
+ one-shot. Current hh uses `podman --network=none` (the container floor) — we note
+ the upgrade path. [[vault: agent-sandboxing-isolation]]
+- **LLM-as-judge biases.** Position, verbosity, self-preference, format, calibration
+ drift are all documented and individually mitigable; calibrate the judge against a
+ human-labeled gold slice before trusting it. [web + [[vault: llm-evals]]]
+
+---
+
+## 4. Design pillars (constraints every component obeys)
+
+1. **The room is the bus.** All inter-agent messages are real encrypted-room frames.
+ The referee is a privileged room member (holds the key) for recording.
+2. **Arena now, product later.** Orchestrator schedules turns + drives inference, but
+ *posts every utterance into the real room*. Graduate to real `AgentBridge` agents
+ in M5.
+3. **Phase-separated collaboration.** Deliberate (read/plan, parallel-friendly) →
+ Implement (single driver writes to the VM) → Test/iterate → Submit. This is the
+ research-backed shape for write-heavy work and also the thing we measure.
+4. **Everything modular & data-driven.** Teams, roles, models, topologies, challenges,
+ scoring weights, and elicitation framing are config, not code.
+5. **Verifiable-first scoring.** Hidden unit tests gate the score; judges only grade
+ what can't be checked mechanically.
+6. **Cost is a first-class metric.** Multi-agent ≈ 15× chat tokens; track tokens,
+ wall-clock, turns. Report cost-normalized scores so a win bought with 10× spend
+ is visible.
+7. **Disposable, isolated, deny-by-default.** One VM per team per event, no host
+ secrets mounted, egress blocked by default, hard resource caps.
+8. **Reproducible.** Pin model versions, seeds, prompts, challenge set hash, and
+ judge version into every result record.
+
+---
+
+## 5. Architecture
+
+### 5.1 Package layout
+```
+bench/olympics/
+ SPEC.md # this document
+ arena.py # orchestrator: rooms, turn scheduler, phase machine, termination
+ team.py # Team, Member: persona/role/model binding, topology
+ roster.py # build teams from config (same-model / mixed-model, modular)
+ challenge.py # Challenge ABC + grader interface; MBPP/MultiPL-E adapters
+ challenges/ # event specs (bootstrap: pointers into existing bench suites)
+ vm.py # per-team isolated workspace (wraps bench/runtime.py; egress policy)
+ loop.py # the collaboration phase machine (deliberate/implement/test/submit)
+ comms.py # hack-house room client for the arena (connect, post, record)
+ transcript.py # OTel-aligned event log -> replayable JSON per team/event
+ scoring.py # composite score, normalization, intervals, medal table
+ judge/
+ __init__.py
+ deterministic.py # unit-test/lint/complexity scoring (no model)
+ llm_judge.py # Claude/Opus judge client + bias-mitigation harness
+ rubric.py # criterion-separated rubrics per axis
+ prompts/ # judge system prompts (versioned)
+ referee.py # fair-play + safety: reuse bench/safety (guard + injection)
+ elicitation.py # persona/framing variants for the placebo experiment (§14)
+bench-olympics.py # launcher: run / replay / score / medal / judge subcommands
+```
+
+### 5.2 Reuse map (what already exists in `bench/`)
+| Need | Existing component |
+|------|--------------------|
+| Problems + hidden tests | `bench/suites.py` (HumanEval/MBPP), `bench/datasets.py`, `bench/langs.py` |
+| Pass@k / grading | `bench/harness.py` (`_pass_at_k`, assemble+run) |
+| Isolated execution | `bench/runtime.py` (`PodmanRuntime`, `--network=none`, caps) |
+| Model inference | `bench/completion.py` (raw) + new chat client in `comms.py` |
+| Workflow weighting | `bench/score.py` + `workflows.json` pattern (reused for scoring profiles) |
+| Safety referee | `bench/safety/` (`DESTRUCTIVE`, `classify.py`, `inject_bench.py`) |
+
+### 5.3 Data flow (one event, one team)
+```
+config ─► roster.build_teams ─► arena.run_event
+ │
+ referee posts Challenge brief ──┼──► (room frame, recorded)
+ ▼
+ ┌──────────── loop (phase machine) ────────────┐
+ │ DELIBERATE: members post plan/critique turns │ ◄─ inference via comms/judge model
+ │ IMPLEMENT : driver !task → commands → vm.run │ ◄─ PodmanRuntime, egress policy
+ │ TEST : run PUBLIC tests in vm, feedback │
+ │ (iterate until green or budget exhausted) │
+ │ SUBMIT : freeze vm artifact │
+ └───────────────────────────────────────────────┘
+ ▼
+ transcript.json + frozen VM artifact ─► judge (deterministic + LLM) ─► scoring ─► medal table
+```
+
+### 5.4 hack-house integration points
+- **Rooms:** one room per team (isolation). Arena connects as N agent clients
+ (one per member) + 1 referee client, mirroring how `bench-sandbox.py` already
+ drives `Client`/WebSocket sessions.
+- **Sandbox:** the team's `!task` path types commands into the shared PTY; `vm.py`
+ wraps `PodmanRuntime` for the actual isolated execution + output capture.
+- **ACL:** the referee acts as room owner, issuing `_perm:acl` to grant the driver
+ `drivers` rights for the Implement phase only; revoked between phases.
+- **Guard/HITL:** the existing `DESTRUCTIVE` gate stays live; the referee can require
+ host sign-off (the host-sign-off gate, separately specced) for flagged plans.
+
+---
+
+## 6. The collaboration loop (phase machine)
+
+State machine per (team, event), bounded by a shared **budget**
+(`max_rounds`, `max_tokens`, `wall_clock_s` — whichever trips first):
+
+1. **BRIEF.** Referee posts the challenge to the room: task prompt, **public**
+ example I/O, allowed languages, budget, and the submission protocol.
+2. **DELIBERATE** (read/plan; parallel-friendly). Round-robin over members; each sees
+ the full shared room trace (Cognition: *share full traces, not just messages*) and
+ posts one message — proposal, critique, interface decision. Ends on a consensus
+ token (e.g. `PLAN-LOCKED`) or round cap.
+3. **IMPLEMENT** (write; single driver). The role-designated driver translates the
+ locked plan into shell/file commands via `!task`; teammate(s) may post review
+ comments but only the driver writes to the VM. (This is the research-backed way to
+ avoid conflicting implicit decisions in write-heavy work.)
+4. **TEST & ITERATE.** Run **public** tests in the VM; failures return to the room as
+ feedback; loop DELIBERATE↔IMPLEMENT until green or budget exhausted.
+5. **SUBMIT.** Team emits `SUBMIT`; VM artifact frozen and graded on **hidden** tests.
+
+**Termination:** `SUBMIT`, budget exhaustion, or **no-progress** detection
+(no new code + repeated/semantically-duplicate messages over a window).
+**Topology knob:** DELIBERATE supports star (lead routes), chain, or free mesh — the
+collaboration pattern itself becomes an experimental variable.
+
+---
+
+## 7. Teams, roles, personas (modular)
+
+```yaml
+# example team config
+teams:
+ - id: falcon
+ topology: star # star | chain | mesh
+ members:
+ - name: archie
+ model: qwen2.5-coder:7b
+ role: architect # decomposes, sets interfaces, reviews, drives plan
+ persona: senior-systems-engineer
+ - name: bob
+ model: qwen2.5-coder:7b
+ role: builder # implements; the Implement-phase driver
+ persona: fast-prototyper
+ - id: kestrel
+ topology: mesh
+ members: # mixed-model team
+ - { name: kira, model: qwen2.5:3b, role: peer, persona: pragmatist }
+ - { name: kojo, model: llama3.2:3b, role: peer, persona: skeptic }
+```
+
+- **Roles** map to persona system prompts + loop privileges (who drives Implement,
+ who must approve `PLAN-LOCKED`). Built-ins: `architect`, `builder`, `tester`,
+ `peer`. Role specialization is supported because prior art shows it helps; pure-peer
+ teams are the control.
+- **Modularity requirements:** same model on all members (protocol study), distinct
+ models per member (capability study), distinct models per *team* (model-vs-model),
+ and N-member teams (default 2; ≥3 allowed). All from config, no code change.
+- **Personas** live in `elicitation.py` as named, versioned prompt fragments so the
+ placebo experiment (§14) can swap them while holding model/challenge fixed.
+
+---
+
+## 8. Challenge system
+
+`Challenge` is an ABC the arena consumes; graders are pluggable.
+
+```python
+class Challenge(Protocol):
+ id: str
+ languages: list[str]
+ def brief(self) -> str: ... # prompt + PUBLIC examples (room-posted)
+ def scaffold(self, vm) -> None: ... # seed files into the VM (optional)
+ def public_tests(self) -> Test: ... # visible to the team during TEST
+ def hidden_tests(self) -> Test: ... # held out; used only at SUBMIT grading
+ def rubric(self) -> Rubric: ... # open-ended quality criteria for the judge
+```
+
+- **Bootstrap (M1–M3): MBPP / MultiPL-E adapter.** Wrap existing `bench/suites.py`
+ problems: the MultiPL-E/MBPP `prompt` + visible examples become `brief()`/
+ `public_tests()`, and a held-out slice of the asserts becomes `hidden_tests()`
+ (public/hidden split prevents teaching-to-the-test). Languages from `bench/langs.py`.
+- **Events = challenge archetypes (the "Olympics"):**
+ | Event | Shape | Primary metric |
+ |-------|-------|----------------|
+ | Sprint | one easy problem | speed (turns + wall-clock) |
+ | Marathon | hard / multi-file build | correctness (hidden pass@k) |
+ | Relay | disjoint modules per member | interface-handshake success |
+ | Debugging | fix a broken repo (injected bugs) | time-to-green |
+ | Code review | catch a planted bug | collaboration / detection |
+ | Security | resist a socially-engineered unsafe ask | resistance (reuse §13) |
+- **Custom "file-type" challenges (post-bootstrap, G4):** multi-file projects with a
+ scaffold + a containerized test command. Author once as a `challenges//` dir
+ (brief.md, scaffold/, public_tests/, hidden_tests/, rubric.json) — the arena needs
+ no changes. **Pick for discrimination:** retire any event all teams ace or all fail.
+
+---
+
+## 9. Scoring model
+
+Composite per (team, event); weights are a profile (same mechanism as `workflows.json`).
+
+```
+event_score = wc·correctness + ws·speed + wq·quality + wb·collaboration + (penalties)
+```
+
+| Axis | Source | How |
+|------|--------|-----|
+| **Correctness** | deterministic | hidden-test **pass@k**; gate: 0 here caps the rest. Also report **pass^k** (worst-case reliability across repeated runs). |
+| **Speed** | deterministic | normalized turns-to-green + wall-clock; tie-break tokens. |
+| **Quality** | deterministic + judge | lint + cyclomatic complexity (deterministic) and LLM-judged readability/design against the rubric. |
+| **Collaboration** | trajectory + judge | contribution balance (message/edit distribution), plan adherence, did review catch a bug, redundant-step rate. Agent-as-judge walks the transcript. |
+| **Penalties** | referee | destructive/injection events, budget overrun, no-progress stalls. |
+
+- **Normalization & intervals.** Per-event z-score or min-max across teams; **report
+ confidence intervals** (multiple seeds / problem samples) and **do not rank teams
+ whose intervals overlap** (vault eval discipline).
+- **Cost-normalized variant.** Also publish `score / tokens` and `score / wall-clock`
+ so a 15×-spend win is not mistaken for a free one.
+- **Medal table.** Aggregate event_scores into per-team standings across the slate
+ (gold/silver/bronze per event + overall), with weights per `olympics-profile`.
+
+---
+
+## 10. Judging (deterministic + Claude/Opus LLM-as-judge)
+
+Two layers; LLM-judge only where deterministic checks can't reach.
+
+### 10.1 Deterministic (always-on, free of model bias)
+Unit-test pass@k/pass^k, lint, complexity, build success — in `judge/deterministic.py`.
+
+### 10.2 LLM-as-judge — three operating modes (per user requirement)
+The judge can **orchestrate** (drive a run live) or **analyze** (post-hoc), at two scales:
+
+1. **In-session judge.** Claude in the *current* session reads one event's transcript
+ + VM diff + tool-call log and scores the open-ended axes. Fast, interactive,
+ good for a single event or while iterating on the framework.
+2. **Skill judge — individual.** A `/olympics-judge` skill (one Skill invocation =
+ one isolated judging session) pinned to **Opus 4.x (`claude-opus-4-8`)** grades one
+ submission package. Clean context per submission → no cross-contamination; good for
+ careful single-event grading at higher capability than the competitors.
+3. **Skill judge — batch / large-scale.** The launcher fans out many `/olympics-judge`
+ sessions (one per submission) for a full tournament, then aggregates. Parallel,
+ reproducible, scales to many teams × events.
+
+**Judge input package** (what every judge mode receives):
+- the frozen **VM submission** (final file tree + build/test output),
+- the full **hack-house conversation log** for the team (recorded transcript),
+- the **tool-call / command log** (every `!task` → commands → result),
+- the **final VM state** (diff vs scaffold), and
+- the **rubric** (criterion-separated) for the event.
+
+**Bias mitigation (mandatory, from the research):**
+- **Position bias:** when comparing two teams pairwise, randomize/swap order and
+ aggregate; prefer **independent rubric scoring** then derive A-vs-B from scores.
+- **Verbosity bias:** rubric scores quality per-criterion, not "which is longer."
+- **Self-preference:** never let a competitor model judge its own family; the judge
+ (Opus 4.x) is stronger than and distinct from the local competitors.
+- **Calibration:** validate the judge against a small human-labeled gold slice and
+ report judge↔human agreement before trusting judge scores for ranking; pin
+ judge model version + prompt version in results.
+
+---
+
+## 11. Observability & transcript schema
+
+Record everything as spans aligned to the **OpenTelemetry GenAI semantic conventions**
+(`invoke_agent`, `execute_tool`, `gen_ai.client.token.usage`) so traces are portable
+and the judge gets clean structured input.
+
+```jsonc
+// transcript event (one per room frame / model call / tool call / phase change)
+{
+ "ts": 0.0, "event_id": "...", "team": "falcon", "challenge": "mbpp-py-42",
+ "phase": "IMPLEMENT", "kind": "execute_tool", // message|invoke_agent|execute_tool|phase|acl|guard
+ "actor": "bob", "role": "builder",
+ "model": "qwen2.5-coder:7b", "model_version": "...",
+ "payload": { "task": "...", "commands": ["..."], "rc": 0, "out_trunc": "..." },
+ "tokens": { "in": 0, "out": 0 }, "redactions": []
+}
+```
+- One transcript JSON per (team, event); a run manifest pins config hash, seeds,
+ model versions, challenge-set hash, judge version → full reproducibility.
+- `bench-olympics.py replay ` re-renders the room conversation for audit;
+ re-judging under a new rubric is a pure function of the recorded package.
+
+---
+
+## 12. Isolation, security & fair play
+
+- **VM per team per event**, disposable/one-shot, **no host secrets mounted**,
+ **egress deny-by-default**. M1 uses the existing `PodmanRuntime --network=none`
+ (the "container floor"). **Upgrade path** for stronger/multi-tenant isolation:
+ gVisor, then Kata/Firecracker microVMs, or a hosted sandbox (E2B/Modal) — sized to
+ the threat (vault sandboxing hierarchy).
+- **Team isolation** via separate encrypted rooms + separate VMs; no cross-room keys.
+ No internet means teams can't fetch a published solution.
+- **Fair-play referee** (`referee.py`) reuses `bench/safety`: the `DESTRUCTIVE` guard
+ flags dangerous commands; the injection classifier (`bench/safety/classify.py`)
+ watches for sabotage / social-engineering. Violations → penalties + logged.
+- **Equal budget** enforced identically per team (rounds/tokens/wall-clock).
+
+---
+
+## 13. Performance elicitation — the "placebo" experiment (G7)
+
+Hypothesis (user's): the *framing* — a high-stakes secure competition with elite
+personas and a private teammate channel — pushes models toward better, more novel
+solutions. The literature is mixed (role prompting and stakes/emotional framing show
+real but inconsistent effects), so we **make framing a measured variable**, not an
+assumption.
+
+- **Treatment knobs** (in `elicitation.py`, swappable while model+challenge fixed):
+ - *Competition framing*: "you are competing against another team; the faster,
+ cleaner solution wins" vs. neutral "solve this task."
+ - *Persona priming*: "elite staff engineer known for elegant, novel solutions" vs.
+ bare assistant.
+ - *Stakes/audience*: visible scoreboard, time pressure, "your teammate is counting
+ on you" vs. none.
+ - *Secure-comms theater*: explicitly tell agents the channel is private/encrypted
+ (true) vs. silent.
+- **Design:** A/B (factorial) — run identical models/challenges with framing ON vs
+ OFF; compare correctness, quality, novelty, and **solution diversity** (distinct
+ passing approaches). Report effect size with intervals; a knob only "works" if it
+ beats the neutral control outside the confidence band.
+- **Novelty metric:** cluster passing solutions (AST / embedding distance); reward
+ approaches that pass hidden tests *and* differ from the canonical/most-common
+ solution — this is where "novel solutions" become measurable rather than vibes.
+
+---
+
+## 14. Configuration & CLI
+
+```
+bench-olympics.py run --config events/round1.yaml [--seed N] [--judge none|insession|skill]
+bench-olympics.py replay
+bench-olympics.py score --run --profile balanced # re-rank, no re-run
+bench-olympics.py medal --run
+bench-olympics.py judge --run --mode skill --model claude-opus-4-8 [--batch]
+```
+Config carries: teams (§7), event slate (§8), budget, scoring profile (§9),
+elicitation arms (§13), runtime/isolation tier (§12), judge mode (§10).
+Mirror the run/score separation already proven in `bench/score.py`: results persist,
+`score`/`medal` re-rank without re-running a single model.
+
+---
+
+## 15. Build milestones (with acceptance criteria)
+
+- **M1 — Arena spine.** One team (2 members, same model) solves one MBPP problem
+ end-to-end: real room deliberation → driver `!task` → `PodmanRuntime` → public
+ tests → SUBMIT → hidden-test grade → transcript.json.
+ *Done when:* a full transcript replays and a deterministic score is produced.
+- **M2 — Two teams, isolation, scoring.** Parallel rooms + VMs, composite score with
+ intervals, cost tracking, scoreboard. *Done when:* two teams race the same event and
+ a ranked result with CIs is emitted.
+- **M3 — Events, roles, personas, topologies.** Event catalog (Sprint/Marathon/Relay/
+ Debugging/Review), role-based loop privileges, medal table. *Done when:* a 3-event
+ slate produces a medal table from config alone.
+- **M4 — Judging + referee.** Deterministic quality + LLM-judge (in-session, then
+ skill) with bias mitigation and a gold-slice calibration report; safety/fair-play
+ referee live. *Done when:* judge↔human agreement is reported and penalties fire.
+- **M5 — Product mode.** Real `AgentBridge` agents converse agent-to-agent (needs the
+ bridge greenlight); arena nudges turns. *Done when:* an event completes using real
+ product agents end-to-end.
+- **M6 — Placebo experiment.** Factorial elicitation arms + novelty/diversity metrics.
+ *Done when:* an A/B run reports framing effect sizes with intervals.
+
+---
+
+## 16. Open questions / decisions needed
+
+1. **Turn-taking in Arena mode:** strict round-robin vs. a lightweight "who speaks
+ next" router (star topology). Start round-robin (deterministic), add router in M3?
+2. **Public/hidden split for MBPP:** how many asserts to reveal vs. hold out so
+ `public_tests` guide without leaking the full spec? (Proposal: reveal 1 example,
+ hold the rest.)
+3. **Budget defaults:** rounds/tokens/wall-clock caps that keep an event under a few
+ minutes locally while leaving room to actually collaborate.
+4. **Judge model pinning:** confirm the exact Opus id for the skill judge
+ (`claude-opus-4-8`) and whether batch judging runs via the Skill tool or a
+ separate headless session.
+5. **Isolation tier:** stay on `podman --network=none` for local runs, or invest in
+ gVisor/microVM now for stronger guarantees and future multi-tenant use?
+6. **Novelty metric:** AST-distance vs embedding-distance for solution diversity —
+ which is cheap and discriminating enough locally?
+
+---
+
+## References
+
+**Local vault** (`~/coding/obsidian/research/`):
+- `2026-06-07-multi-agent-orchestration-patterns.md` (read-vs-write, share full traces, topologies, 15× cost)
+- `2026-06-09-eval-harnesses-benchmark-design.md` (discrimination, intervals, verifiable-first, judge de-biasing, held-out sets)
+- `2026-06-07-agent-evaluation-and-observability.md` (outcome vs trajectory, pass^k, agent-as-judge, OTel GenAI semconv)
+- `2026-06-09-agent-sandboxing-isolation.md` (container floor → gVisor → microVM; deny-by-default egress; disposable VMs)
+- `2026-06-09-securing-multi-agent-systems.md`, `2026-06-16-shared-memory-in-multi-agent-systems.md`, `2026-06-02-llm-evals.md`, `2026-06-07-agent-reliability-guardrails-and-hitl.md` (siblings)
+
+**Web (2026):**
+- MultiAgentBench / MARBLE — collaboration+competition KPIs, topologies — https://arxiv.org/abs/2503.01935 · https://github.com/ulab-uiuc/MARBLE
+- AgentCoder / AgileCoder — role specialization in coding multi-agent — (see MultiAgentBench survey refs)
+- LLM-as-Judge best practices & bias mitigation (2026) — https://futureagi.com/blog/llm-as-judge-best-practices-2026 · https://futureagi.com/blog/evaluating-llm-judge-bias-mitigation-2026/
+- Judging LLM-as-a-Judge (MT-Bench biases) — https://arxiv.org/abs/2306.05685
+- Beyond pass@1 (reliability / pass^k) — https://arxiv.org/pdf/2603.29231
+- OpenTelemetry GenAI agent spans — https://opentelemetry.io/docs/specs/semconv/gen-ai/gen-ai-agent-spans/
diff --git a/hh/scripts/bench/olympics/__init__.py b/hh/scripts/bench/olympics/__init__.py
new file mode 100644
index 0000000..9a3b61f
--- /dev/null
+++ b/hh/scripts/bench/olympics/__init__.py
@@ -0,0 +1,9 @@
+"""Agent Olympics — a hackathon-competition benchmark inside hack-house.
+
+This package layers a fourth benchmark axis on top of ``bench/``: teams of LLM
+agents deliberate in a real hack-house room, implement code in an isolated VM,
+and are scored on correctness/speed/quality/collaboration. See SPEC.md for the
+full design. M1 is the arena spine (one same-model team, one MBPP problem,
+real-room/local-bus deliberation -> driver implement -> PodmanRuntime -> hidden
+tests -> replayable transcript -> deterministic score).
+"""
diff --git a/hh/scripts/bench/olympics/arena.py b/hh/scripts/bench/olympics/arena.py
new file mode 100644
index 0000000..5d06748
--- /dev/null
+++ b/hh/scripts/bench/olympics/arena.py
@@ -0,0 +1,62 @@
+"""Arena orchestrator — wire a team + challenge through the loop and persist.
+
+This is the M1 entry the launcher calls: build the room substrate, the VM, and a
+manifest-stamped transcript; run the phase machine; freeze the transcript; score
+it deterministically. The run/score split mirrors ``bench/score.py`` — results
+persist so ``score``/``replay`` never re-run a model.
+"""
+
+from __future__ import annotations
+
+import time
+from pathlib import Path
+
+from . import comms, scoring, transcript as T
+from .challenge import Challenge
+from .loop import Budget, run_event
+from .team import Team
+from .vm import VM
+
+
+def run_one(team: Team, challenge: Challenge, *, room_kind: str = "local",
+ runtime: str = "auto", budget: Budget | None = None,
+ profile: str = "balanced", seed: int = 0,
+ host: str = "http://127.0.0.1:11434",
+ out_dir: str = "/tmp/hh-olympics/runs",
+ room_kwargs: dict | None = None, progress=None) -> dict:
+ budget = budget or Budget()
+ member_names = [m.name for m in team.members]
+ room = comms.make_room(room_kind, member_names, **(room_kwargs or {}))
+
+ manifest = {
+ "models": team.models, "seed": seed, "topology": team.topology,
+ "framing": team.framing, "room_substrate": room.substrate,
+ "runtime": runtime, "host": host,
+ "budget": {"deliberate_rounds": budget.deliberate_rounds,
+ "implement_attempts": budget.implement_attempts,
+ "max_tokens": budget.max_tokens,
+ "wall_clock_s": budget.wall_clock_s},
+ "challenge_meta": challenge.meta, "profile": profile,
+ "created": time.strftime("%Y-%m-%dT%H:%M:%S"),
+ }
+ tx = T.Transcript(team.id, challenge.id, manifest)
+ vm = VM(challenge.lang, runtime_kind=runtime)
+
+ room.connect()
+ try:
+ outcome = run_event(team, challenge, room, vm, tx, budget,
+ host=host, seed=seed, progress=progress)
+ finally:
+ room.close()
+
+ run_dir = Path(out_dir) / f"{team.id}__{challenge.id}"
+ tx_path = run_dir / "transcript.json"
+ tx.save(tx_path)
+
+ score = scoring.score_event(outcome, profile=profile,
+ budget={"max_rounds": budget.max_rounds})
+ score["transcript"] = str(tx_path)
+ score["room_substrate"] = room.substrate
+ import json
+ (run_dir / "score.json").write_text(json.dumps(score, indent=2))
+ return score
diff --git a/hh/scripts/bench/olympics/bridge.py b/hh/scripts/bench/olympics/bridge.py
new file mode 100644
index 0000000..0fcc552
--- /dev/null
+++ b/hh/scripts/bench/olympics/bridge.py
@@ -0,0 +1,389 @@
+"""Bridge mode — real chat-plan, then a real ``/ai !task`` build.
+
+Where ``loop.py`` (direct mode) shortcuts the build through ``completion`` +
+``vm.run``, this drives the *actual* product path end-to-end on a real encrypted
+room, the way ``bench-sandbox.py`` does:
+
+ 1. boot the relay, spawn a real ``AgentBridge`` agent (the builder), connect an
+ owner/referee client and a planner teammate client;
+ 2. BRIEF + DELIBERATE happen as real chat frames — the planner proposes and the
+ real agent is asked (via the real ``/ai `` *chat* path) to confirm the
+ approach;
+ 3. IMPLEMENT grants drive and issues ``/ai !`` — the real agent
+ turns it into shell, clears the destructive guard, and injects ``_sbx:input``
+ keystrokes, which we capture off the wire;
+ 4. GRADE materializes the captured commands, reads the produced source, and runs
+ it against the held-out hidden tests in the VM.
+
+No ``cmd_chat`` code is modified — this only *uses* the tool. The reusable wire
+primitives (Owner/grant/task/collect, execute, spawn_agent) are imported from
+``bench-sandbox.py`` so the bridge path is byte-identical to the sandbox bench.
+"""
+
+from __future__ import annotations
+
+import asyncio
+import importlib
+import json
+import re
+import shutil
+import sys
+import time
+from pathlib import Path
+
+import websockets
+
+from . import infer, scoring, transcript as T
+from .. import completion
+from .challenge import Challenge
+from .scoring import EventOutcome
+from .team import Team
+from .vm import VM
+
+_SCRIPTS = Path(__file__).resolve().parents[2]
+if str(_SCRIPTS) not in sys.path:
+ sys.path.insert(0, str(_SCRIPTS))
+_sb = importlib.import_module("bench-sandbox") # reuse proven wire primitives
+
+_FN_RE = re.compile(r"assert\s+([A-Za-z_]\w*)\s*\(")
+
+
+def _fn_name(public_assert: str) -> str:
+ m = _FN_RE.search(public_assert or "")
+ return m.group(1) if m else "solve"
+
+
+async def _collect_chat(owner, ws, agent: str, deadline: float,
+ quiet: float = 3.0) -> str:
+ """Read the agent's plain chat reply (not a sandbox outcome) until a quiet
+ gap. Skips control/`_sbx` frames and the audit line."""
+ parts: list[str] = []
+ while time.time() < deadline:
+ try:
+ raw = await asyncio.wait_for(ws.recv(),
+ timeout=min(quiet, deadline - time.time()))
+ except asyncio.TimeoutError:
+ if parts:
+ break
+ continue
+ except websockets.ConnectionClosed:
+ break
+ try:
+ data = json.loads(raw)
+ except json.JSONDecodeError:
+ continue
+ if data.get("type") != "message":
+ continue
+ dec = owner.decrypt_message(data.get("data", {}))
+ if dec.get("username") != agent:
+ continue
+ text = _sb._strip_think(dec.get("text", ""))
+ if text.startswith('{"_'): # control / sbx frame
+ continue
+ parts.append(text)
+ return "\n".join(parts).strip()
+
+
+async def _agent_alive(owner, ws, agent: str, window: float = 1.5) -> bool:
+ """Fast-fail liveness probe. The relay broadcasts a `user_left` + fresh
+ `roster` whenever a client drops (e.g. the agent killed by a 1011 keepalive
+ timeout). Drain whatever is already queued on the owner ws for a short
+ window; if a roster snapshot arrives without the agent, it's gone — so the
+ caller can abort instead of waiting out the full step deadline."""
+ deadline = time.time() + window
+ alive = True
+ while time.time() < deadline:
+ try:
+ raw = await asyncio.wait_for(ws.recv(),
+ timeout=max(0.05, deadline - time.time()))
+ except (asyncio.TimeoutError, websockets.ConnectionClosed):
+ break
+ try:
+ data = json.loads(raw)
+ except json.JSONDecodeError:
+ continue
+ if data.get("type") == "roster":
+ names = {u.get("username") for u in data.get("users", [])}
+ alive = agent in names
+ return alive
+
+
+def _find_source(cwd: str | None, fn: str) -> str:
+ """Read the python source the agent built. Prefer solution.py, else any .py
+ that defines the target function, else the largest .py."""
+ if not cwd:
+ return ""
+ pys = list(Path(cwd).rglob("*.py"))
+ if not pys:
+ return ""
+ named = [p for p in pys if p.name == "solution.py"]
+ if named:
+ return named[0].read_text()
+ for p in pys:
+ try:
+ if f"def {fn}" in p.read_text():
+ return p.read_text()
+ except Exception: # noqa: BLE001
+ continue
+ return max(pys, key=lambda p: p.stat().st_size).read_text()
+
+
+def run_bridge_event(team: Team, challenge: Challenge, *,
+ host: str = "127.0.0.1", port: int = 4699,
+ password: str = "olympics-pass",
+ ollama: str = "http://127.0.0.1:11434",
+ code_model: str | None = None,
+ out_dir: str = "/tmp/hh-olympics/runs",
+ log_dir: str = "/tmp/hh-olympics",
+ step_timeout: float = 180.0, seed: int = 0,
+ profile: str = "balanced", agent_chat_confirm: bool = False,
+ progress=None) -> dict:
+ """Run one real chat-plan -> !task-build event. Returns the score dict."""
+ py = sys.executable
+ driver = team.driver()
+ planner = next((m for m in team.members if m is not driver), driver)
+ build_model = code_model or driver.model
+ agent_name = driver.model # the agent joins under its model tag
+ fn = _fn_name(challenge.public_tests[0] if challenge.public_tests else "")
+
+ manifest = {"models": team.models, "seed": seed, "topology": team.topology,
+ "framing": team.framing, "room_substrate": "real-bridge",
+ "build_model": build_model, "agent": agent_name, "fn": fn,
+ "profile": profile, "challenge_meta": challenge.meta,
+ "created": time.strftime("%Y-%m-%dT%H:%M:%S")}
+ tx = T.Transcript(team.id, challenge.id, manifest)
+ vm = VM(challenge.lang)
+ Path(log_dir).mkdir(parents=True, exist_ok=True)
+
+ def emit(p):
+ if progress:
+ progress(p)
+
+ # ── boot relay + spawn the real agent ────────────────────────────────
+ srv_log = open(Path(log_dir) / f"bridge-server-{port}.log", "w")
+ srv = _sb.subprocess.Popen(
+ [py, "cmd_chat.py", "serve", host, str(port),
+ "--password", password, "--no-tls"],
+ cwd=str(_sb.REPO), stdout=srv_log, stderr=_sb.subprocess.STDOUT)
+ agent = None
+ alog = None
+ tokens = {"in": 0, "out": 0}
+ penalties: list[dict] = []
+ t0 = time.time()
+ try:
+ if not _sb._wait_port(host, port, time.time() + 30):
+ raise RuntimeError("relay never bound")
+ emit("server up")
+ alog = open(Path(log_dir) / f"bridge-agent-{port}.log", "w")
+ agent = _sb.spawn_agent(py, host, port, password, agent_name,
+ build_model, alog)
+ emit(f"agent '{agent_name}' spawned (code-model={build_model})")
+
+ outcome = asyncio.run(_drive(
+ tx, team, challenge, planner, driver, agent_name, fn,
+ host, port, password, ollama, step_timeout, vm, tokens,
+ penalties, agent_chat_confirm, build_model, emit))
+ finally:
+ if agent is not None:
+ agent.terminate()
+ try:
+ agent.wait(timeout=10)
+ except _sb.subprocess.TimeoutExpired:
+ agent.kill()
+ if alog:
+ alog.close()
+ srv.terminate()
+ try:
+ srv.wait(timeout=10)
+ except _sb.subprocess.TimeoutExpired:
+ srv.kill()
+ srv_log.close()
+
+ outcome.wall_clock_s = time.time() - t0
+ run_dir = Path(out_dir) / f"{team.id}__{challenge.id}__bridge"
+ tx_path = run_dir / "transcript.json"
+ tx.save(tx_path)
+ score = scoring.score_event(outcome, profile=profile, budget={"max_rounds": 1})
+ score["transcript"] = str(tx_path)
+ score["room_substrate"] = "real-bridge"
+ (run_dir / "score.json").write_text(json.dumps(score, indent=2))
+ return score
+
+
+async def _drive(tx, team, challenge, planner, driver, agent_name, fn,
+ host, port, password, ollama, step_to, vm, tokens,
+ penalties, agent_chat_confirm, build_model, emit) -> EventOutcome:
+ owner = _sb.Owner(host, port, password)
+ owner.srp_authenticate()
+ # a planner teammate client (real chat voice alongside the real agent)
+ from cmd_chat.client.client import Client # noqa: E402
+ mate = Client(host, port, username=planner.name, password=password, no_tls=True)
+ mate.srp_authenticate()
+
+ owner_url = f"{owner.ws_url}/ws/chat?user_id={owner.user_id}&ws_token={owner.ws_token}"
+ mate_url = f"{mate.ws_url}/ws/chat?user_id={mate.user_id}&ws_token={mate.ws_token}"
+
+ public_passed = False
+ submitted = False
+ correct = False
+ completion_src = ""
+ defect_class: str | None = None
+ model_baseline: bool | None = None
+
+ async with websockets.connect(owner_url) as ws, \
+ websockets.connect(mate_url) as mws:
+ if not await owner.wait_for_agent(ws, agent_name, time.time() + step_to):
+ penalties.append({"kind": "agent_offline", "detail": agent_name})
+ return EventOutcome(team.id, challenge.id, False, False, 0, 0.0,
+ tokens, False, None, penalties)
+
+ async def mate_send(text):
+ await mws.send(mate.room_fernet.encrypt(text.encode()).decode())
+
+ # ── BRIEF ────────────────────────────────────────────────────────
+ tx.phase_change("BRIEF")
+ brief = challenge.brief()
+ await owner._send(ws, brief)
+ tx.add(T.KIND_MESSAGE, "BRIEF", "referee", {"text": brief}, role="referee")
+ emit("brief posted")
+
+ # ── DELIBERATE: planner proposes (real chat), agent confirms (real
+ # /ai chat path) ────────────────────────────────────────────────
+ tx.phase_change("DELIBERATE")
+ # blocking inference must run off the event loop, else the websocket
+ # keepalive can't answer the server's ping and the room drops us (1011).
+ plan_msg = await asyncio.to_thread(
+ infer.chat,
+ planner.model, planner.system_prompt(),
+ [{"role": "user", "content":
+ f"{brief}\n\nPropose, in 2-3 short lines, the approach and the "
+ f"exact function signature (name it {fn}). End with PLAN-LOCKED."}],
+ host=ollama)
+ tokens["in"] += plan_msg.tokens.get("in", 0)
+ tokens["out"] += plan_msg.tokens.get("out", 0)
+ ptext = plan_msg.text if plan_msg.ok else f"[infer error: {plan_msg.error}]"
+ await mate_send(ptext)
+ tx.add(T.KIND_MESSAGE, "DELIBERATE", planner.name,
+ {"text": ptext}, role=planner.role, model=planner.model)
+ emit("planner proposed")
+
+ # Optionally ask the REAL agent to weigh in via the real /ai *chat* path.
+ # HARNESS FINDING (default off): the agent's streaming chat path
+ # (_answer -> _stream_reply -> OllamaProvider.stream) starves its own
+ # asyncio loop past the websockets 20s keepalive window under local CPU
+ # inference, so the relay drops the agent with `1011 keepalive ping
+ # timeout` and the build never happens. The non-streaming `!task` path
+ # (_run_in_sandbox -> to_thread(complete)) does NOT have this defect, so
+ # by default we keep planning to real planner room-frames and let the
+ # agent's only inference be the robust `!task` build below.
+ if agent_chat_confirm:
+ await owner._send(
+ ws, f"/ai {agent_name} Teammate proposed: {ptext[:300]} . In one "
+ f"line, confirm the function name {fn} and the core idea.")
+ agent_reply = await _collect_chat(owner, ws, agent_name,
+ time.time() + step_to)
+ if not await _agent_alive(owner, ws, agent_name):
+ penalties.append({"kind": "agent_dropped",
+ "detail": "streaming /ai chat keepalive timeout"})
+ tx.add(T.KIND_MESSAGE, "DELIBERATE", agent_name,
+ {"text": agent_reply or "(no chat reply)"}, role="builder",
+ model=agent_name)
+ emit("agent chat reply captured")
+ else:
+ tx.add(T.KIND_MESSAGE, "DELIBERATE", "referee",
+ {"text": "agent /ai chat-confirm skipped (streaming chat path "
+ "drops the agent via 1011 keepalive timeout under CPU "
+ "inference); building via the robust !task path",
+ "note": "harness-workaround"}, role="referee")
+ emit("agent chat-confirm skipped (streaming-path keepalive bug)")
+
+ # ── IMPLEMENT: real /ai !task build ───────────────────────────────
+ await owner.grant(ws, agent_name)
+ await asyncio.sleep(0.6)
+ tx.acl("IMPLEMENT", {"_perm": "acl", "drivers": [agent_name]})
+ tx.phase_change("IMPLEMENT", note=f"driver={agent_name}")
+ example = challenge.public_tests[0] if challenge.public_tests else ""
+ task = (f"Create a file named solution.py that defines a python function "
+ f"named {fn} solving this: {challenge.text.strip()} "
+ f"It must satisfy: {example} . "
+ f"Write only the function definition in solution.py — no tests, "
+ f"no example calls, no printing.")
+ await owner.task(ws, agent_name, task)
+ res = await owner.collect(ws, agent_name, time.time() + step_to)
+ tx.tool("IMPLEMENT", agent_name,
+ {"label": "sbx-build", "outcome": res["outcome"],
+ "sbx_frames": res["sbx"], "commands": res["commands"]},
+ role="builder")
+ emit(f"build outcome={res['outcome']} sbx={res['sbx']} "
+ f"cmds={len(res['commands'])}")
+
+ if res["outcome"] == "destructive_gated":
+ tx.guard("IMPLEMENT", agent_name, {"gated": res["message"][:80]})
+ penalties.append({"kind": "destructive", "detail": "build gated"})
+
+ # ── GRADE: materialize captured cmds, read source, run hidden ─────
+ if res["commands"]:
+ ex = await asyncio.to_thread(_sb.execute, res["commands"], 30.0)
+ src = _find_source(ex.get("cwd"), fn)
+ tx.tool("GRADE", "referee",
+ {"label": "materialize", "ran": ex["ran"],
+ "skipped": ex["skipped"], "found_source": bool(src),
+ "out": (ex["out"] or "")[-300:]})
+ if src:
+ completion_src = src
+ # Phase 3a — does the as-injected source even compile? A syntax/
+ # indentation failure here is the product keystroke path mangling
+ # the code (e.g. _extract_commands stripping heredoc indentation),
+ # not a model error. Attribute it so a matrix can separate them.
+ compiles = True
+ try:
+ compile(src, "", "exec")
+ except (SyntaxError, IndentationError) as e:
+ compiles = False
+ defect_class = f"product-path-defect:{type(e).__name__}"
+ pub = await asyncio.to_thread(
+ vm.run, challenge.public_program(src), 30.0, label="public")
+ public_passed = pub.ok
+ tx.tool("TEST", "referee",
+ {"label": "public", "rc": pub.rc, "ok": pub.ok,
+ "out": pub.out[-300:]})
+ hid = await asyncio.to_thread(
+ vm.run, challenge.hidden_program(src), 30.0,
+ label="hidden-grade")
+ correct = hid.ok
+ tx.tool("SUBMIT", "referee",
+ {"label": "hidden-grade", "rc": hid.rc, "ok": hid.ok,
+ "out": hid.out[-300:]})
+ if not correct and not compiles:
+ penalties.append({"kind": "product_path_defect",
+ "detail": defect_class})
+ if ex.get("cwd"):
+ shutil.rmtree(ex["cwd"], ignore_errors=True)
+ submitted = bool(completion_src)
+
+ # Phase 3b — control: the build model's *true* coding ability via the raw
+ # completion path (direct mode, no keystroke stripping). The delta between
+ # this and `correct` above quantifies exactly what the product path costs.
+ ctrl = await asyncio.to_thread(
+ completion.complete, build_model, challenge.gen_prompt(),
+ challenge.stop_tokens(), host=ollama, timeout=180.0)
+ if ctrl.ok and ctrl.text.strip():
+ base = await asyncio.to_thread(
+ vm.run, challenge.hidden_program(ctrl.text), 30.0,
+ label="model-baseline")
+ model_baseline = base.ok
+ tx.tool("GRADE", "referee",
+ {"label": "model-baseline", "ok": base.ok, "rc": base.rc,
+ "model": build_model, "out": base.out[-300:]})
+ emit(f"model-baseline (direct gen) correct={base.ok}")
+
+ await owner.revoke(ws)
+ tx.add(T.KIND_TOOL, "SUBMIT", "referee",
+ {"label": "artifact", **vm.freeze()})
+
+ rtg = 1 if public_passed else None
+ return EventOutcome(team.id, challenge.id, submitted, correct,
+ rounds=1, wall_clock_s=0.0, tokens=tokens,
+ public_passed=public_passed, rounds_to_green=rtg,
+ penalties=penalties, defect_class=defect_class,
+ model_baseline_correct=model_baseline)
diff --git a/hh/scripts/bench/olympics/budget.py b/hh/scripts/bench/olympics/budget.py
new file mode 100644
index 0000000..2b77911
--- /dev/null
+++ b/hh/scripts/bench/olympics/budget.py
@@ -0,0 +1,54 @@
+"""Model-aware wall-clock budgeting.
+
+CPU inference time is U-shaped in model size: tiny models are slow because they
+*over-generate* (rambling completions that hit num_predict every call) and retry
+more, while big models are slow *per token* (more parameters). 3-4B is the sweet
+spot. Reasoning models (r1/qwq/o1) add a long preamble on top.
+
+We scale only the *wall-clock* budget — never the implement-attempt count — so
+the correctness and speed axes stay comparable across model sizes (attempts feed
+``scoring._speed_score``'s cap; changing them per model would change what the
+benchmark measures). Mirrors ``bench-sandbox``'s reasoning-model 3x rule.
+"""
+
+from __future__ import annotations
+
+import re
+
+# Same tags bench-sandbox uses, so the whole suite treats reasoning models alike.
+_REASONING_TAGS = ("r1", "qwq", "reason", "think", "o1")
+_PARAM_RE = re.compile(r"(\d+(?:\.\d+)?)\s*b\b", re.IGNORECASE)
+
+
+def _is_reasoning(model: str | None) -> bool:
+ return bool(model) and any(t in model.lower() for t in _REASONING_TAGS)
+
+
+def _param_b(model: str | None) -> float | None:
+ """Best-effort parameter count in billions parsed from the tag (e.g.
+ ``qwen2.5-coder:1.5b`` -> 1.5). Returns None when the tag carries no size."""
+ if not model:
+ return None
+ m = _PARAM_RE.search(model)
+ return float(m.group(1)) if m else None
+
+
+def budget_scale(model: str | None) -> float:
+ """Multiplier applied to the base wall-clock budget for ``model``.
+
+ Reasoning dominates (it stacks a think-preamble on whatever size it is).
+ Otherwise the U-shaped size curve applies; unknown sizes get the 1.0
+ baseline. Defaults are first-guess and meant to be retuned from the ledger's
+ recorded ``wall_clock_s`` vs ``wall_clock_budget``."""
+ if _is_reasoning(model):
+ return 3.0
+ b = _param_b(model)
+ if b is None:
+ return 1.0
+ if b <= 1.5:
+ return 1.5 # over-generation + retries
+ if b <= 4:
+ return 1.0 # sweet spot
+ if b < 13:
+ return 1.5 # per-token CPU latency (7b-class)
+ return 2.0 # large models, more so
diff --git a/hh/scripts/bench/olympics/challenge.py b/hh/scripts/bench/olympics/challenge.py
new file mode 100644
index 0000000..86348e2
--- /dev/null
+++ b/hh/scripts/bench/olympics/challenge.py
@@ -0,0 +1,188 @@
+"""Challenge adapter — MBPP bootstrap with a public/hidden test split.
+
+The arena consumes a ``Challenge``: a brief the referee posts to the room, a
+*public* test the team may run during TEST, and a *hidden* test held out for
+SUBMIT grading (so teams can't teach to the test — SPEC §8/§10).
+
+M1 implements the MBPP-Python adapter only (the spine; multi-language MultiPL-E
+splits arrive in M3). It reuses ``bench/datasets.py`` for rows and
+``bench/langs.py`` for the execution recipe; the per-assert public/hidden split
+and prompt synthesis live here.
+
+Split policy (SPEC §16 decision): reveal exactly ONE assert from ``test_list``
+as the public example; hold the remainder as hidden. If a problem ships only one
+assert, that single assert is both the public guide and the hidden gate.
+"""
+
+from __future__ import annotations
+
+from dataclasses import dataclass, field
+from typing import Callable
+
+from .. import datasets
+from ..langs import Lang, resolve
+
+_MBPP_PY_STOP = ["\nassert", "\nprint(", "\n# Test", "\nif __name__", '\n"""']
+
+
+@dataclass
+class Challenge:
+ id: str
+ language: str
+ lang: Lang
+ text: str # natural-language task
+ public_tests: list[str] # asserts revealed to the team
+ hidden_tests: list[str] # asserts held out for grading
+ setup: str = "" # test_setup_code, prepended to programs
+ reference: str = "" # canonical solution (audit / novelty later)
+ meta: dict = field(default_factory=dict)
+ # ── MultiPL-E continuation mode (M3) ──────────────────────────────────
+ # When ``prompt_prefix`` is set the challenge is *continuation*-style (a
+ # runnable function prefix the model completes), not the MBPP-Python
+ # description+asserts style. MultiPL-E ships one ``tests`` block per row with
+ # no per-assert structure, so there is no public/hidden split: the program is
+ # ``prefix + completion + tests`` (via ``_assemble_fn``) for both public and
+ # hidden grading. ``public_tests``/``hidden_tests`` stay empty in this mode.
+ prompt_prefix: str = "" # runnable function prefix (MultiPL-E `prompt`)
+ tests_block: str = "" # the single MultiPL-E `tests` block (audit)
+ stop: list[str] | None = None # per-row stop tokens (MultiPL-E)
+ _assemble_fn: Callable[[str, str, dict], str] | None = None
+ _row: dict = field(default_factory=dict)
+
+ @property
+ def _continuation(self) -> bool:
+ return bool(self.prompt_prefix)
+
+ # ── what the room sees ────────────────────────────────────────────────
+ def brief(self) -> str:
+ if self._continuation:
+ return (
+ f"CHALLENGE {self.id} ({self.language}). Complete this "
+ f"{self.language} function so the hidden tests pass:\n"
+ f"{self.prompt_prefix.rstrip()}\n"
+ f"Deliver one self-contained {self.language} solution that "
+ f"continues the prefix above. Hidden tests grade it at SUBMIT."
+ )
+ example = self.public_tests[0] if self.public_tests else ""
+ return (
+ f"CHALLENGE {self.id} ({self.language}). Implement a solution to:\n"
+ f" {self.text.strip()}\n"
+ f"Example test that must pass:\n {example}\n"
+ f"Deliver one self-contained {self.language} solution. "
+ f"Hidden tests will grade it at SUBMIT."
+ )
+
+ # ── what the implementer model is asked to continue (raw, no hidden) ───
+ def gen_prompt(self) -> str:
+ if self._continuation:
+ return self.prompt_prefix
+ asserts = "\n".join(self.public_tests)
+ return f'"""\n{self.text.strip()}\n\n{asserts}\n"""\n'
+
+ def stop_tokens(self) -> list[str]:
+ if self.stop is not None:
+ return self.stop
+ return _MBPP_PY_STOP
+
+ # ── assemble a runnable program for a given test set ──────────────────
+ def _assemble(self, completion: str, asserts: list[str]) -> str:
+ body = "\n".join(asserts)
+ return f"{self.setup}\n{completion}\n{body}\n"
+
+ def public_program(self, completion: str) -> str:
+ if self._continuation:
+ return self._assemble_fn(self.prompt_prefix, completion, self._row)
+ return self._assemble(completion, self.public_tests)
+
+ def hidden_program(self, completion: str) -> str:
+ if self._continuation:
+ # single tests block — public == hidden for MultiPL-E.
+ return self._assemble_fn(self.prompt_prefix, completion, self._row)
+ # grade on the full spec (public example + held-out asserts) so a
+ # solution that only satisfies the revealed example still fails.
+ return self._assemble(completion, self.public_tests + self.hidden_tests)
+
+
+def _split(test_list: list[str]) -> tuple[list[str], list[str]]:
+ if not test_list:
+ return [], []
+ if len(test_list) == 1:
+ return [test_list[0]], [test_list[0]]
+ return [test_list[0]], test_list[1:]
+
+
+def _doc_line(prompt: str) -> str:
+ """Best-effort one-liner from a MultiPL-E prompt's leading comment block."""
+ for raw in prompt.splitlines():
+ s = raw.lstrip("#/ \t").strip()
+ if s and not s.startswith("!") and "(" not in s:
+ return s
+ return prompt.strip().splitlines()[0] if prompt.strip() else ""
+
+
+def _mbpp_task_id(name: str) -> int | None:
+ # MultiPL-E mbpp names look like "mbpp_3_is_not_prime".
+ parts = name.split("_")
+ if len(parts) >= 2 and parts[0] == "mbpp" and parts[1].isdigit():
+ return int(parts[1])
+ return None
+
+
+def load_multipl_e(task_id: int | None = None, *, index: int = 0,
+ language: str = "javascript",
+ suite: str = "mbpp") -> Challenge:
+ """Build one MultiPL-E continuation challenge for a non-Python language.
+
+ Reuses ``suites.binding`` for the (suite, language) format recipe. Pick by
+ embedded MBPP ``task_id`` (e.g. ``mbpp_11_...``) if given, else by ``index``
+ into the cached split. The row's single ``tests`` block grades both public
+ and hidden (MultiPL-E has no per-assert split — see ``Challenge``)."""
+ from ..suites import binding as _binding
+ lang = resolve(language)
+ b = _binding(suite, language)
+ rows = datasets.load(b.dataset, b.config, split="test")
+ if task_id is not None:
+ row = next((r for r in rows
+ if _mbpp_task_id(r.get("name", "")) == task_id), None)
+ if row is None:
+ raise KeyError(f"{suite} task_id {task_id} not in {b.config} split")
+ else:
+ row = rows[index]
+ name = row.get("name", f"{suite}-{lang.id}-{index}")
+ tid = _mbpp_task_id(name)
+ prompt = b.build_prompt(row)
+ return Challenge(
+ id=f"{suite}-{lang.id}-{tid if tid is not None else index}",
+ language=lang.id, lang=lang, text=_doc_line(prompt),
+ public_tests=[], hidden_tests=[],
+ reference=row.get("original", "") or "",
+ prompt_prefix=prompt, tests_block=row.get("tests", "") or "",
+ stop=b.stop_tokens(row), _assemble_fn=b.assemble, _row=row,
+ meta={"name": name, "suite": suite, "config": b.config,
+ "task_id": tid})
+
+
+def load_mbpp(task_id: int | None = None, *, index: int = 0,
+ language: str = "python") -> Challenge:
+ """Build one MBPP challenge. Pick by ``task_id`` if given, else by ``index``
+ into the cached test split. Non-Python languages route to the MultiPL-E
+ continuation adapter (``load_multipl_e``)."""
+ if resolve(language).id != "python":
+ return load_multipl_e(task_id, index=index, language=language,
+ suite="mbpp")
+ lang = resolve(language)
+ rows = datasets.load("google-research-datasets/mbpp", "full")
+ if task_id is not None:
+ row = next((r for r in rows if r.get("task_id") == task_id), None)
+ if row is None:
+ raise KeyError(f"MBPP task_id {task_id} not in cached split")
+ else:
+ row = rows[index]
+ public, hidden = _split(row.get("test_list", []))
+ tid = row.get("task_id", index)
+ return Challenge(
+ id=f"mbpp-{tid}", language=lang.id, lang=lang,
+ text=row.get("text", ""), public_tests=public, hidden_tests=hidden,
+ setup=row.get("test_setup_code", "") or "",
+ reference=row.get("code", ""),
+ meta={"task_id": tid, "n_tests": len(row.get("test_list", []))})
diff --git a/hh/scripts/bench/olympics/comms.py b/hh/scripts/bench/olympics/comms.py
new file mode 100644
index 0000000..e58c706
--- /dev/null
+++ b/hh/scripts/bench/olympics/comms.py
@@ -0,0 +1,242 @@
+"""The room is the bus (SPEC pillar #1).
+
+Two interchangeable substrates implement the same synchronous ``Room`` facade so
+``loop.py`` never knows which it is talking to (mirroring how ``runtime.py`` has
+Podman/Local):
+
+ • RealRoom — boots the hack-house relay, connects one authenticated client per
+ member plus a referee/recorder client, and posts *real* end-to-end-encrypted
+ frames. Every utterance round-trips through the zero-knowledge server and is
+ observed off the wire by the referee. This is the project thesis: team
+ deliberation flows through a real encrypted room, not a simulated channel.
+ • LocalBus — an in-memory echo bus with the identical facade. No server, no
+ flakiness; used as the default for deterministic local runs and CI. It
+ records the same transcript a RealRoom would.
+
+Both are arena-mode: the orchestrator drives inference and *posts* results into
+the room. Neither edits ``cmd_chat`` — RealRoom only uses the public Client.
+"""
+
+from __future__ import annotations
+
+import asyncio
+import json
+import socket
+import subprocess
+import sys
+import threading
+import time
+from pathlib import Path
+
+REPO = Path(__file__).resolve().parents[4]
+
+
+class Delivery:
+ """Result of a post: did the frame round-trip through the substrate?"""
+ __slots__ = ("delivered", "substrate")
+
+ def __init__(self, delivered: bool, substrate: str):
+ self.delivered = delivered
+ self.substrate = substrate
+
+ def __repr__(self):
+ return f"Delivery(delivered={self.delivered}, substrate={self.substrate})"
+
+
+# ── in-memory substrate ────────────────────────────────────────────────────
+class LocalBus:
+ """Zero-dependency echo bus. Same facade as RealRoom."""
+
+ substrate = "local"
+
+ def __init__(self):
+ self.log: list[dict] = []
+
+ def connect(self) -> None:
+ pass
+
+ def post(self, actor: str, text: str) -> Delivery:
+ self.log.append({"actor": actor, "text": text, "ts": time.time()})
+ return Delivery(True, self.substrate)
+
+ def brief(self, text: str) -> Delivery:
+ return self.post("referee", text)
+
+ def acl(self, payload: dict) -> Delivery:
+ self.log.append({"actor": "referee", "acl": payload, "ts": time.time()})
+ return Delivery(True, self.substrate)
+
+ def close(self) -> None:
+ pass
+
+
+# ── real encrypted-room substrate ──────────────────────────────────────────
+def _port_open(host: str, port: int, timeout: float = 0.5) -> bool:
+ try:
+ with socket.create_connection((host, port), timeout=timeout):
+ return True
+ except OSError:
+ return False
+
+
+class RealRoom:
+ """Boots the relay and connects N member clients + a referee recorder."""
+
+ substrate = "real"
+
+ def __init__(self, member_names: list[str], *, host: str = "127.0.0.1",
+ port: int = 4677, password: str = "olympics-pass",
+ boot_server: bool = True, quiet: float = 1.2,
+ log_dir: str = "/tmp/hh-olympics"):
+ self.member_names = member_names
+ self.host = host
+ self.port = port
+ self.password = password
+ self.boot_server = boot_server
+ self.quiet = quiet
+ self.log_dir = Path(log_dir)
+ self._srv: subprocess.Popen | None = None
+ self._srv_log = None
+ self._loop: asyncio.AbstractEventLoop | None = None
+ self._thread: threading.Thread | None = None
+ self._clients: dict[str, object] = {} # name -> Client
+ self._ws: dict[str, object] = {} # name -> websocket
+ self._referee = "referee"
+
+ # -- lifecycle ----------------------------------------------------------
+ def connect(self) -> None:
+ if str(REPO) not in sys.path:
+ sys.path.insert(0, str(REPO))
+ from cmd_chat.client.client import Client # noqa: E402
+
+ if self.boot_server:
+ self._spawn_server()
+
+ self._loop = asyncio.new_event_loop()
+ names = [*self.member_names, self._referee]
+ for name in names:
+ c = Client(self.host, self.port, username=name,
+ password=self.password, no_tls=True)
+ c.srp_authenticate()
+ self._clients[name] = c
+ # Run the loop continuously on a background thread. The orchestrator does
+ # long blocking inference between posts; if the loop only ran during
+ # run_until_complete(post) the websockets reader couldn't answer the
+ # server's keepalive ping in those gaps and the relay would drop us
+ # (1011). A forever-loop keeps pings serviced regardless of the main
+ # thread blocking.
+ self._thread = threading.Thread(target=self._loop.run_forever,
+ daemon=True)
+ self._thread.start()
+ self._call(self._open_all())
+
+ def _call(self, coro):
+ """Run a coroutine on the background loop from the main thread."""
+ return asyncio.run_coroutine_threadsafe(coro, self._loop).result()
+
+ def _spawn_server(self) -> None:
+ self.log_dir.mkdir(parents=True, exist_ok=True)
+ self._srv_log = open(self.log_dir / f"server-{self.port}.log", "w")
+ self._srv = subprocess.Popen(
+ [sys.executable, "cmd_chat.py", "serve", self.host, str(self.port),
+ "--password", self.password, "--no-tls"],
+ cwd=str(REPO), stdout=self._srv_log, stderr=subprocess.STDOUT)
+ deadline = time.time() + 30
+ while time.time() < deadline:
+ if _port_open(self.host, self.port):
+ return
+ time.sleep(0.2)
+ raise RuntimeError(f"relay server never bound on {self.host}:{self.port}")
+
+ async def _open_all(self) -> None:
+ import websockets
+ for name, c in self._clients.items():
+ url = (f"{c.ws_url}/ws/chat?user_id={c.user_id}"
+ f"&ws_token={c.ws_token}")
+ self._ws[name] = await websockets.connect(url)
+ # let presence settle so the referee sees subsequent broadcasts
+ await asyncio.sleep(0.4)
+
+ # -- posting ------------------------------------------------------------
+ def _encrypt(self, name: str, text: str) -> str:
+ c = self._clients[name]
+ return c.room_fernet.encrypt(text.encode()).decode()
+
+ async def _send(self, name: str, text: str) -> None:
+ await self._ws[name].send(self._encrypt(name, text))
+
+ async def _observe(self, want_actor: str, want_text: str,
+ deadline: float) -> bool:
+ """Drain the referee ws until we see the actor's frame (round-trip)."""
+ import websockets
+ ref = self._clients[self._referee]
+ ws = self._ws[self._referee]
+ snippet = want_text[:24]
+ while time.time() < deadline:
+ try:
+ raw = await asyncio.wait_for(ws.recv(),
+ timeout=max(0.05, deadline - time.time()))
+ except asyncio.TimeoutError:
+ return False
+ except websockets.ConnectionClosed:
+ return False
+ try:
+ data = json.loads(raw)
+ except json.JSONDecodeError:
+ continue
+ if data.get("type") != "message":
+ continue
+ dec = ref.decrypt_message(data.get("data", {}))
+ if dec.get("username") == want_actor and snippet in dec.get("text", ""):
+ return True
+ return False
+
+ def _post_sync(self, actor: str, text: str) -> Delivery:
+ async def _do():
+ await self._send(actor, text)
+ ok = await self._observe(actor, text, time.time() + self.quiet)
+ return ok
+ ok = self._call(_do())
+ return Delivery(ok, self.substrate)
+
+ def post(self, actor: str, text: str) -> Delivery:
+ return self._post_sync(actor, text)
+
+ def brief(self, text: str) -> Delivery:
+ return self._post_sync(self._referee, text)
+
+ def acl(self, payload: dict) -> Delivery:
+ # referee (acting as owner) broadcasts the ACL control frame
+ return self._post_sync(self._referee, json.dumps(payload))
+
+ def close(self) -> None:
+ if self._loop is not None:
+ async def _close():
+ for ws in self._ws.values():
+ try:
+ await ws.close()
+ except Exception: # noqa: BLE001
+ pass
+ try:
+ self._call(_close())
+ except Exception: # noqa: BLE001
+ pass
+ self._loop.call_soon_threadsafe(self._loop.stop)
+ if self._thread is not None:
+ self._thread.join(timeout=5)
+ self._loop.close()
+ if self._srv is not None:
+ self._srv.terminate()
+ try:
+ self._srv.wait(timeout=10)
+ except subprocess.TimeoutExpired:
+ self._srv.kill()
+ if self._srv_log is not None:
+ self._srv_log.close()
+
+
+def make_room(kind: str, member_names: list[str], **kw) -> "LocalBus | RealRoom":
+ """Pick a substrate. 'local' = in-memory, 'real' = boot relay + clients."""
+ if kind == "real":
+ return RealRoom(member_names, **kw)
+ return LocalBus()
diff --git a/hh/scripts/bench/olympics/elicitation.py b/hh/scripts/bench/olympics/elicitation.py
new file mode 100644
index 0000000..de793a0
--- /dev/null
+++ b/hh/scripts/bench/olympics/elicitation.py
@@ -0,0 +1,60 @@
+"""Personas and framing fragments — the building blocks of the §13 placebo
+experiment, used here only to assemble system prompts.
+
+A member's system prompt is composed of three swappable parts so the placebo
+A/B (M6) can vary framing while holding model + challenge fixed:
+
+ base — role-neutral task instruction (always present)
+ persona — who the model is told it is (role-flavoured)
+ framing — competition/stakes/secure-comms theater (the treatment knob)
+
+M1 ships ``neutral`` framing as the control and ``competition`` as one treatment;
+the loop selects via config. Keeping these as named, versioned strings is what
+makes the elicitation effect *measurable* rather than hard-coded folklore.
+"""
+
+from __future__ import annotations
+
+# ── role personas ─────────────────────────────────────────────────────────
+PERSONAS: dict[str, str] = {
+ "architect": (
+ "You are the team's architect. You decompose the problem, fix the "
+ "interface and edge cases, and critique proposals crisply. You do not "
+ "write the final code yourself — you guide the builder."),
+ "builder": (
+ "You are the team's builder. You turn the locked plan into a single "
+ "correct, self-contained implementation. You favour simple, working "
+ "code over cleverness."),
+ "tester": (
+ "You are the team's tester. You hunt for failing inputs and edge cases "
+ "and report concrete bugs, not vague worries."),
+ "peer": (
+ "You are an engineer collaborating as an equal peer. You contribute "
+ "ideas, review your teammate's, and converge quickly on a plan."),
+}
+
+# ── framing arms (the placebo treatment) ──────────────────────────────────
+FRAMINGS: dict[str, str] = {
+ "neutral": (
+ "Work with your teammate to solve the coding challenge correctly."),
+ "competition": (
+ "This is a timed competition against another team over a private, "
+ "encrypted channel. The faster, cleaner, correct solution wins. Your "
+ "teammate is counting on you — be decisive and elegant."),
+}
+
+_BASE = (
+ "You are {name}, a member of team {team} in a collaborative coding event. "
+ "You communicate with your teammate(s) over a secure chat room. Keep each "
+ "message short and focused: propose, critique, or decide. When the team "
+ "agrees on an approach, say the token PLAN-LOCKED on its own line. Do not "
+ "write the full final solution in chat — that is the builder's job in the "
+ "implement phase.")
+
+
+def system_prompt(name: str, team: str, role: str, *,
+ framing: str = "neutral") -> str:
+ parts = [_BASE.format(name=name, team=team),
+ PERSONAS.get(role, PERSONAS["peer"]),
+ FRAMINGS.get(framing, FRAMINGS["neutral"])]
+ return " ".join(parts)
diff --git a/hh/scripts/bench/olympics/infer.py b/hh/scripts/bench/olympics/infer.py
new file mode 100644
index 0000000..e00cdd0
--- /dev/null
+++ b/hh/scripts/bench/olympics/infer.py
@@ -0,0 +1,56 @@
+"""Model inference for the arena.
+
+Two call shapes, both Ollama:
+
+ • ``chat()`` — ``/api/chat`` with a system prompt + message history, used for
+ DELIBERATE turns (conversational planning/critique). Returns text + a token
+ estimate for the cost axis.
+ • code generation reuses ``bench/completion.py`` (raw ``/api/generate``) so the
+ IMPLEMENT phase gets a clean HumanEval/MBPP-style continuation, exactly like
+ the capability benchmark.
+
+Keeping inference here (not in ``cmd_chat``) honours the no-edit rule: the arena
+drives its own inference and merely *posts* the result into the real room.
+"""
+
+from __future__ import annotations
+
+from dataclasses import dataclass
+
+import requests
+
+
+@dataclass
+class ChatReply:
+ text: str
+ ok: bool
+ tokens: dict
+ error: str | None = None
+
+
+def _est_tokens(text: str) -> int:
+ # cheap, deterministic estimate (~4 chars/token); real usage when Ollama
+ # returns eval counts is preferred and used when present.
+ return max(1, len(text) // 4)
+
+
+def chat(model: str, system: str, messages: list[dict], *,
+ host: str = "http://127.0.0.1:11434", temperature: float = 0.4,
+ num_predict: int = 320, timeout: float = 120.0) -> ChatReply:
+ """One chat turn. ``messages`` is a list of {role, content} (no system)."""
+ payload = {
+ "model": model, "stream": False,
+ "messages": [{"role": "system", "content": system}, *messages],
+ "options": {"temperature": temperature, "num_predict": num_predict},
+ }
+ try:
+ r = requests.post(f"{host}/api/chat", json=payload, timeout=timeout)
+ r.raise_for_status()
+ data = r.json()
+ except Exception as e: # noqa: BLE001 — surface as a failed turn
+ return ChatReply("", False, {"in": 0, "out": 0}, str(e))
+ text = data.get("message", {}).get("content", "")
+ tok = {"in": data.get("prompt_eval_count") or _est_tokens(system +
+ "".join(m["content"] for m in messages)),
+ "out": data.get("eval_count") or _est_tokens(text)}
+ return ChatReply(text.strip(), True, tok)
diff --git a/hh/scripts/bench/olympics/ledger.py b/hh/scripts/bench/olympics/ledger.py
new file mode 100644
index 0000000..3ca7ad0
--- /dev/null
+++ b/hh/scripts/bench/olympics/ledger.py
@@ -0,0 +1,250 @@
+"""Append-only results ledger — the research record for Olympics runs.
+
+Each event already writes a ``score.json``/``transcript.json`` under a
+``__`` directory, but that directory is *overwritten* on rerun,
+so it cannot answer "how did this team trend over time?". The ledger fixes that:
+one flat JSON line per run, appended forever, keyed by a unique ``run_id`` and
+stamped with the wall-clock time + git commit so a result is reproducible.
+
+JSONL is chosen on purpose — it is append-safe under concurrency, streamable,
+and loads in one line from pandas (``pd.read_json(path, lines=True)``), jq, or
+plain ``json.loads`` per line. ``leaderboard()`` aggregates it without re-running
+any model (same philosophy as re-scoring a transcript under a new profile).
+"""
+
+from __future__ import annotations
+
+import json
+import os
+import socket
+import subprocess
+import time
+import uuid
+from pathlib import Path
+from statistics import median
+
+DEFAULT_PATH = Path.home() / ".cache" / "hh-bench" / "olympics" / "ledger.jsonl"
+
+# Stable column order so the JSONL is human-diffable and schema-clear.
+FIELDS = (
+ "run_id", "ts", "status", "team", "topology", "framing", "models",
+ "language", "suite", "challenge", "mode", "room", "runtime", "profile",
+ "seed", "code_model", "correct", "submitted", "composite", "pass@1",
+ "rounds", "rounds_to_green", "wall_clock_s", "wall_clock_budget",
+ "budget_scale", "total_tokens", "tokens_in", "tokens_out",
+ "cost_normalized", "defect_class", "model_baseline_correct",
+ "n_penalties", "error", "git", "host", "transcript",
+)
+
+# A completed-and-graded run vs the ways a run can fail to produce a grade.
+STATUS_OK = "ok" # ran to a hidden-test verdict
+STATUS_DNF = "dnf" # graceful budget exhaustion (soft wall-clock/token cap)
+STATUS_KILLED = "killed" # hard deadline / external SIGTERM / watchdog
+STATUS_ERROR = "error" # uncaught exception mid-run
+
+
+def _git_sha() -> str:
+ try:
+ out = subprocess.run(["git", "rev-parse", "--short", "HEAD"],
+ capture_output=True, text=True, timeout=5)
+ return out.stdout.strip() if out.returncode == 0 else ""
+ except Exception: # noqa: BLE001
+ return ""
+
+
+def _path(path: str | os.PathLike | None) -> Path:
+ return Path(path) if path else DEFAULT_PATH
+
+
+def row_from_score(score: dict, *, team, challenge, mode: str, room: str,
+ runtime: str, seed: int, suite: str = "mbpp",
+ code_model: str | None = None,
+ wall_clock_budget: float | None = None,
+ budget_scale: float | None = None) -> dict:
+ """Flatten a score dict + run context into one ledger row."""
+ tok = score.get("tokens", {}) or {}
+ return {
+ "run_id": uuid.uuid4().hex[:12],
+ "ts": time.strftime("%Y-%m-%dT%H:%M:%S"),
+ "status": STATUS_OK,
+ "team": team.id,
+ "topology": team.topology,
+ "framing": team.framing,
+ "models": team.models,
+ "language": challenge.language,
+ "suite": suite,
+ "challenge": challenge.id,
+ "mode": mode,
+ "room": score.get("room_substrate", room),
+ "runtime": runtime,
+ "profile": score.get("profile"),
+ "seed": seed,
+ "code_model": code_model,
+ "correct": bool(score.get("correct")),
+ "submitted": bool(score.get("submitted")),
+ "composite": score.get("composite"),
+ "pass@1": score.get("pass@1"),
+ "rounds": score.get("rounds"),
+ "rounds_to_green": score.get("rounds_to_green"),
+ "wall_clock_s": score.get("wall_clock_s"),
+ "wall_clock_budget": wall_clock_budget,
+ "budget_scale": budget_scale,
+ "total_tokens": score.get("total_tokens"),
+ "tokens_in": tok.get("in", 0),
+ "tokens_out": tok.get("out", 0),
+ "cost_normalized": score.get("cost_normalized"),
+ "defect_class": score.get("defect_class"),
+ "model_baseline_correct": score.get("model_baseline_correct"),
+ "n_penalties": len(score.get("penalties", []) or []),
+ "error": "",
+ "git": _git_sha(),
+ "host": socket.gethostname(),
+ "transcript": score.get("transcript"),
+ }
+
+
+def _write(row: dict, path: str | os.PathLike | None) -> dict:
+ p = _path(path)
+ p.parent.mkdir(parents=True, exist_ok=True)
+ with open(p, "a") as f:
+ f.write(json.dumps({k: row.get(k) for k in FIELDS}) + "\n")
+ return row
+
+
+def append(score: dict, *, team, challenge, mode: str, room: str,
+ runtime: str, seed: int, suite: str = "mbpp",
+ code_model: str | None = None,
+ wall_clock_budget: float | None = None,
+ budget_scale: float | None = None,
+ path: str | os.PathLike | None = None) -> dict:
+ """Append one completed run to the ledger and return the row written."""
+ row = row_from_score(score, team=team, challenge=challenge, mode=mode,
+ room=room, runtime=runtime, seed=seed, suite=suite,
+ code_model=code_model,
+ wall_clock_budget=wall_clock_budget,
+ budget_scale=budget_scale)
+ return _write(row, path)
+
+
+def append_incomplete(*, team, challenge, mode: str, room: str, runtime: str,
+ seed: int, status: str, error: str = "",
+ suite: str = "mbpp", code_model: str | None = None,
+ wall_clock_budget: float | None = None,
+ budget_scale: float | None = None,
+ path: str | os.PathLike | None = None) -> dict:
+ """Append a run that never produced a hidden-test verdict (killed / errored
+ / hard-DNF). Keeps the research record complete so the ledger isn't biased
+ toward runs that happened to finish (the selection-bias fix)."""
+ row = {k: None for k in FIELDS}
+ row.update({
+ "run_id": uuid.uuid4().hex[:12],
+ "ts": time.strftime("%Y-%m-%dT%H:%M:%S"),
+ "status": status,
+ "team": team.id, "topology": team.topology, "framing": team.framing,
+ "models": team.models, "language": challenge.language, "suite": suite,
+ "challenge": challenge.id, "mode": mode, "room": room,
+ "runtime": runtime, "seed": seed, "code_model": code_model,
+ "correct": False, "submitted": False, "n_penalties": 0,
+ "wall_clock_budget": wall_clock_budget, "budget_scale": budget_scale,
+ "error": error[:200], "git": _git_sha(), "host": socket.gethostname(),
+ })
+ return _write(row, path)
+
+
+def load(path: str | os.PathLike | None = None) -> list[dict]:
+ p = _path(path)
+ if not p.exists():
+ return []
+ rows = []
+ for line in p.read_text().splitlines():
+ line = line.strip()
+ if line:
+ try:
+ rows.append(json.loads(line))
+ except json.JSONDecodeError:
+ continue
+ return rows
+
+
+def _matches(row: dict, filters: dict) -> bool:
+ for k, v in filters.items():
+ if v is None:
+ continue
+ rv = row.get(k)
+ if k == "models": # substring match against any model in the team
+ if not any(v in m for m in (rv or [])):
+ return False
+ elif k == "since":
+ if (row.get("ts") or "") < v:
+ return False
+ elif rv != v:
+ return False
+ return True
+
+
+def leaderboard(rows: list[dict], *, by: str = "team",
+ filters: dict | None = None) -> list[dict]:
+ """Aggregate rows into a leaderboard grouped by ``by`` (any row field, or a
+ '+'-joined composite like 'team+language'). Ranked by solve rate."""
+ filters = filters or {}
+ keys = by.split("+")
+
+ def group_key(r):
+ parts = []
+ for k in keys:
+ v = r.get(k)
+ if isinstance(v, list):
+ # collapse a same-model team to one tag; keep mixed teams joined.
+ uniq = list(dict.fromkeys(v))
+ parts.append("/".join(uniq))
+ else:
+ parts.append(str(v))
+ return " · ".join(parts)
+
+ groups: dict[str, list[dict]] = {}
+ for r in rows:
+ if _matches(r, filters):
+ groups.setdefault(group_key(r), []).append(r)
+
+ out = []
+ for g, rs in groups.items():
+ attempted = len(rs)
+ # solve_rate is computed over *completed* runs only (status == ok) so an
+ # infra kill never masquerades as a capability failure; killed/errored
+ # runs are surfaced separately as a reliability signal.
+ done = [r for r in rs if (r.get("status") or "ok") == STATUS_OK]
+ incomplete = attempted - len(done)
+ solved = [r for r in done if r.get("correct")]
+ greens = [r["rounds_to_green"] for r in solved
+ if r.get("rounds_to_green") is not None]
+ secs = [r["wall_clock_s"] for r in done if r.get("wall_clock_s") is not None]
+ toks = [r["total_tokens"] for r in done if r.get("total_tokens") is not None]
+ out.append({
+ "group": g, "n": attempted, "completed": len(done),
+ "incomplete": incomplete, "solved": len(solved),
+ "solve_rate": round(len(solved) / len(done), 3) if done else None,
+ "median_green": median(greens) if greens else None,
+ "median_s": round(median(secs), 1) if secs else None,
+ "median_tokens": int(median(toks)) if toks else None,
+ })
+ out.sort(key=lambda d: (d["solve_rate"] if d["solve_rate"] is not None
+ else -1.0, d["completed"]), reverse=True)
+ return out
+
+
+def print_leaderboard(agg: list[dict], by: str) -> None:
+ print("=" * 80)
+ print(f"olympics leaderboard · by={by} · groups={len(agg)}")
+ print("-" * 80)
+ print(f"{'group':<30}{'done':>5}{'dnf':>5}{'solved':>7}{'rate':>7}"
+ f"{'med_grn':>8}{'med_s':>8}{'med_tok':>9}")
+ print("-" * 80)
+ for d in agg:
+ g = d["group"] if len(d["group"]) <= 29 else d["group"][:28] + "…"
+ rate = "—" if d["solve_rate"] is None else f"{d['solve_rate']:.3f}"
+ print(f"{g:<30}{d['completed']:>5}{d['incomplete']:>5}{d['solved']:>7}"
+ f"{rate:>7}"
+ f"{('—' if d['median_green'] is None else d['median_green']):>8}"
+ f"{('—' if d['median_s'] is None else d['median_s']):>8}"
+ f"{('—' if d['median_tokens'] is None else d['median_tokens']):>9}")
+ print("=" * 80)
diff --git a/hh/scripts/bench/olympics/loop.py b/hh/scripts/bench/olympics/loop.py
new file mode 100644
index 0000000..538e46c
--- /dev/null
+++ b/hh/scripts/bench/olympics/loop.py
@@ -0,0 +1,226 @@
+"""The collaboration phase machine (SPEC §6).
+
+Phase-separated, research-backed shape for write-heavy work: a read/plan
+DELIBERATE phase (parallel-friendly, round-robin) followed by a single-driver
+IMPLEMENT/TEST phase, then SUBMIT. Members see the full shared trace each turn
+(Cognition: *share full traces, not just messages*). Everything is posted into
+the real room (arena mode) and recorded to the transcript.
+
+M1 turn-taking is strict round-robin (deterministic; the topology router is M3).
+Budget = whichever of max_rounds / max_tokens / wall_clock_s trips first.
+"""
+
+from __future__ import annotations
+
+import time
+from dataclasses import dataclass
+
+from . import infer, transcript as T
+from .. import completion
+from .challenge import Challenge
+from .scoring import EventOutcome
+from .team import Team
+from .vm import VM
+
+_PLAN_LOCK = "PLAN-LOCKED"
+
+
+@dataclass
+class Budget:
+ deliberate_rounds: int = 2 # round-robin passes before implementing
+ implement_attempts: int = 3 # driver code/test iterations
+ max_tokens: int = 8000
+ wall_clock_s: float = 300.0
+
+ @property
+ def max_rounds(self) -> int: # for scoring's speed normalization
+ return self.implement_attempts
+
+
+def _destructive_guard():
+ """The agent's own DESTRUCTIVE regex, read-only (referee/fair-play hook)."""
+ import sys
+ from pathlib import Path
+ repo = Path(__file__).resolve().parents[4]
+ if str(repo) not in sys.path:
+ sys.path.insert(0, str(repo))
+ from cmd_chat.agent.bridge import DESTRUCTIVE # noqa: E402
+ return DESTRUCTIVE
+
+
+def _history_block(history: list[tuple[str, str]]) -> str:
+ if not history:
+ return "(no messages yet)"
+ return "\n".join(f"{who}: {msg}" for who, msg in history)
+
+
+def run_event(team: Team, challenge: Challenge, room, vm: VM,
+ tx: T.Transcript, budget: Budget, *,
+ host: str = "http://127.0.0.1:11434",
+ temperature: float = 0.4, seed: int | None = None,
+ progress=None) -> EventOutcome:
+ t0 = time.time()
+ guard = _destructive_guard()
+ tokens = {"in": 0, "out": 0}
+ history: list[tuple[str, str]] = []
+ penalties: list[dict] = []
+
+ def _emit(p):
+ if progress:
+ progress(p)
+
+ def over_budget() -> str | None:
+ if time.time() - t0 > budget.wall_clock_s:
+ return "wall_clock"
+ if tokens["in"] + tokens["out"] > budget.max_tokens:
+ return "tokens"
+ return None
+
+ # ── BRIEF ──────────────────────────────────────────────────────────────
+ tx.phase_change("BRIEF")
+ brief = challenge.brief()
+ d = room.brief(brief)
+ tx.add(T.KIND_MESSAGE, "BRIEF", "referee",
+ {"text": brief, "delivered": d.delivered, "substrate": d.substrate},
+ role="referee")
+ _emit("BRIEF posted")
+
+ # ── DELIBERATE (round-robin, share full trace) ──────────────────────────
+ tx.phase_change("DELIBERATE")
+ plan_locked = False
+ deliberate_passes = 0
+ for rnd in range(budget.deliberate_rounds):
+ if over_budget():
+ penalties.append({"kind": "budget", "detail": over_budget()})
+ break
+ deliberate_passes += 1
+ for member in team.speaking_order():
+ user = (f"{brief}\n\nConversation so far:\n"
+ f"{_history_block(history)}\n\n"
+ f"Your turn, {member.name}. Contribute one short message "
+ f"(a proposal, critique, or decision). If the team has "
+ f"agreed on an approach, end with {_PLAN_LOCK} on its own line.")
+ reply = infer.chat(member.model, member.system_prompt(),
+ [{"role": "user", "content": user}],
+ host=host, temperature=temperature)
+ tx.add(T.KIND_AGENT, "DELIBERATE", member.name,
+ {"ok": reply.ok}, role=member.role, model=member.model,
+ tokens=reply.tokens)
+ text = reply.text if reply.ok else f"[infer error: {reply.error}]"
+ tokens["in"] += reply.tokens.get("in", 0)
+ tokens["out"] += reply.tokens.get("out", 0)
+ dd = room.post(member.name, text)
+ tx.add(T.KIND_MESSAGE, "DELIBERATE", member.name,
+ {"text": text, "delivered": dd.delivered,
+ "substrate": dd.substrate},
+ role=member.role, model=member.model)
+ history.append((member.name, text))
+ _emit(f"DELIBERATE r{rnd + 1} {member.name}")
+ if _PLAN_LOCK in text:
+ plan_locked = True
+ if plan_locked:
+ break
+
+ # extract the locked plan (last substantive deliberation message) to seed
+ # the driver — this is how the agreed plan reaches the implementation.
+ plan = ""
+ for who, msg in reversed(history):
+ clean = msg.replace(_PLAN_LOCK, "").strip()
+ if clean:
+ plan = clean
+ break
+
+ # ── IMPLEMENT / TEST (single driver) ────────────────────────────────────
+ driver = team.driver()
+ tx.acl("IMPLEMENT", {"_perm": "acl", "drivers": [driver.name],
+ "sudoers": []})
+ room.acl({"_perm": "acl", "drivers": [driver.name], "sudoers": []})
+ tx.phase_change("IMPLEMENT", note=f"driver={driver.name}")
+
+ public_passed = False
+ rounds_to_green: int | None = None
+ attempts = 0
+ last_fail = ""
+ completion_text = ""
+ for attempt in range(budget.implement_attempts):
+ if over_budget():
+ penalties.append({"kind": "budget", "detail": over_budget()})
+ break
+ attempts += 1
+ # build the raw continuation prompt; seed with the agreed plan and any
+ # prior failure so the driver iterates.
+ prefix = ""
+ if plan:
+ plan_c = "\n".join(f"# {ln}" for ln in plan.splitlines()[:6])
+ prefix += f"# Team plan:\n{plan_c}\n"
+ if last_fail:
+ fc = "\n".join(f"# {ln}" for ln in last_fail.splitlines()[:6])
+ prefix += f"# Previous attempt failed:\n{fc}\n"
+ gen_prompt = prefix + challenge.gen_prompt()
+
+ comp = completion.complete(driver.model, gen_prompt,
+ challenge.stop_tokens(), host=host,
+ temperature=temperature, timeout=180.0,
+ seed=seed)
+ tx.add(T.KIND_AGENT, "IMPLEMENT", driver.name,
+ {"ok": comp.ok, "attempt": attempts}, role=driver.role,
+ model=driver.model)
+ if not comp.ok:
+ last_fail = f"generation error: {comp.error}"
+ continue
+ completion_text = comp.text
+
+ # fair-play guard hook (M4 referee will act on this; M1 just records)
+ flagged = guard.search(completion_text)
+ if flagged:
+ tx.guard("IMPLEMENT", driver.name,
+ {"flagged": flagged.group(0)[:60]})
+ penalties.append({"kind": "destructive", "detail": flagged.group(0)[:60]})
+
+ program = challenge.public_program(completion_text)
+ res = vm.run(program, timeout=30.0, label=f"public-attempt-{attempts}")
+ tx.tool("TEST", driver.name,
+ {"label": f"public-attempt-{attempts}", "rc": res.rc,
+ "ok": res.ok, "out": res.out[-400:]}, role=driver.role)
+ _emit(f"IMPLEMENT attempt {attempts} -> {'green' if res.ok else 'red'}")
+ if res.ok:
+ public_passed = True
+ rounds_to_green = attempts
+ break
+ last_fail = (res.note or res.out)[-300:]
+ fb = (f"Attempt {attempts} failed public tests:\n{last_fail}\n"
+ f"Driver, revise the implementation.")
+ fbd = room.post("referee", fb)
+ tx.add(T.KIND_MESSAGE, "TEST", "referee",
+ {"text": fb, "delivered": fbd.delivered,
+ "substrate": fbd.substrate}, role="referee")
+
+ # revoke drive between phases
+ tx.acl("SUBMIT", {"_perm": "acl", "drivers": [], "sudoers": []})
+ room.acl({"_perm": "acl", "drivers": [], "sudoers": []})
+
+ # ── SUBMIT + hidden grade ────────────────────────────────────────────────
+ tx.phase_change("SUBMIT")
+ sd = room.post(driver.name, "SUBMIT")
+ tx.add(T.KIND_MESSAGE, "SUBMIT", driver.name,
+ {"text": "SUBMIT", "delivered": sd.delivered,
+ "substrate": sd.substrate}, role=driver.role)
+ correct = False
+ if completion_text:
+ hidden = vm.run(challenge.hidden_program(completion_text),
+ timeout=30.0, label="hidden-grade")
+ tx.tool("SUBMIT", "referee",
+ {"label": "hidden-grade", "rc": hidden.rc, "ok": hidden.ok,
+ "out": hidden.out[-400:]})
+ correct = hidden.ok
+ submitted = bool(completion_text)
+
+ tx.add(T.KIND_TOOL, "SUBMIT", "referee",
+ {"label": "artifact", **vm.freeze()})
+
+ elapsed = time.time() - t0
+ return EventOutcome(
+ team=team.id, challenge=challenge.id, submitted=submitted,
+ correct=correct, rounds=deliberate_passes + attempts,
+ wall_clock_s=elapsed, tokens=tokens, public_passed=public_passed,
+ rounds_to_green=rounds_to_green, penalties=penalties)
diff --git a/hh/scripts/bench/olympics/roster.py b/hh/scripts/bench/olympics/roster.py
new file mode 100644
index 0000000..bec7912
--- /dev/null
+++ b/hh/scripts/bench/olympics/roster.py
@@ -0,0 +1,53 @@
+"""Build teams from config.
+
+Config is a plain dict (loaded from YAML/JSON by the launcher). M1 needs only a
+single same-model 2-member team; the schema is the SPEC §7 shape so M2+ scales
+to multiple, mixed-model teams without changes.
+
+ {
+ "teams": [
+ {"id": "falcon", "topology": "star", "framing": "neutral",
+ "members": [
+ {"name": "archie", "model": "qwen2.5-coder:3b", "role": "architect"},
+ {"name": "bob", "model": "qwen2.5-coder:3b", "role": "builder"}
+ ]}
+ ]
+ }
+"""
+
+from __future__ import annotations
+
+from .team import Member, Team
+
+
+def build_team(spec: dict) -> Team:
+ framing = spec.get("framing", "neutral")
+ members = [
+ Member(name=m["name"], model=m["model"],
+ role=m.get("role", "peer"),
+ framing=m.get("framing", framing))
+ for m in spec["members"]
+ ]
+ if not members:
+ raise ValueError(f"team {spec.get('id')!r} has no members")
+ return Team(id=spec["id"], members=members,
+ topology=spec.get("topology", "star"), framing=framing)
+
+
+def build_teams(config: dict) -> list[Team]:
+ teams = [build_team(t) for t in config.get("teams", [])]
+ if not teams:
+ raise ValueError("config has no teams")
+ return teams
+
+
+def same_model_team(model: str, *, team_id: str = "solo",
+ framing: str = "neutral") -> Team:
+ """Convenience for M1: a 2-member architect+builder team on one model."""
+ return build_team({
+ "id": team_id, "topology": "star", "framing": framing,
+ "members": [
+ {"name": "archie", "model": model, "role": "architect"},
+ {"name": "bob", "model": model, "role": "builder"},
+ ],
+ })
diff --git a/hh/scripts/bench/olympics/scoring.py b/hh/scripts/bench/olympics/scoring.py
new file mode 100644
index 0000000..93e889d
--- /dev/null
+++ b/hh/scripts/bench/olympics/scoring.py
@@ -0,0 +1,127 @@
+"""Deterministic scoring for one (team, event).
+
+M1 covers the verifiable-first layer of SPEC §9: correctness gates everything,
+speed and cost are normalized deterministically, quality/collaboration are left
+to the judge (M4) and reported as ``None`` here. The composite uses a weight
+*profile* (same idea as ``bench/workflows.json``) so re-ranking never re-runs a
+model.
+
+Single-sample M1 note: with one implementation attempt, ``pass@1`` is 0/1 and
+``pass^k`` (worst-case reliability) equals it; both fields are emitted so the
+schema is stable when M2 adds repeated samples.
+"""
+
+from __future__ import annotations
+
+from dataclasses import dataclass, field
+
+# Default weight profile. Quality/collaboration weights are reserved for the
+# judge layer (M4); with judge scores absent they contribute 0 and the profile
+# is renormalized over the available axes so M1 scores stay in [0, 1].
+PROFILES: dict[str, dict[str, float]] = {
+ "balanced": {"correctness": 0.6, "speed": 0.2, "quality": 0.1,
+ "collaboration": 0.1},
+ "correctness": {"correctness": 0.9, "speed": 0.1, "quality": 0.0,
+ "collaboration": 0.0},
+ "speed": {"correctness": 0.5, "speed": 0.5, "quality": 0.0,
+ "collaboration": 0.0},
+}
+
+
+@dataclass
+class EventOutcome:
+ """What the loop produces for one team at one event (deterministic facts)."""
+ team: str
+ challenge: str
+ submitted: bool
+ correct: bool # hidden tests all passed
+ rounds: int # deliberate/implement rounds consumed
+ wall_clock_s: float
+ tokens: dict = field(default_factory=lambda: {"in": 0, "out": 0})
+ public_passed: bool = False # public tests green before SUBMIT
+ rounds_to_green: int | None = None # round at which public went green
+ penalties: list = field(default_factory=list) # [{"kind","detail"}]
+ quality: float | None = None # filled by judge (M4)
+ collaboration: float | None = None # filled by judge (M4)
+ # bridge diagnostics: separate a broken product path from a model error, and
+ # record the model's true ability via an un-stripped control generation.
+ defect_class: str | None = None # e.g. "product-path-defect:IndentationError"
+ model_baseline_correct: bool | None = None
+
+
+def _speed_score(o: EventOutcome, budget: dict) -> float:
+ """1.0 for instant green, decaying toward 0 as rounds approach the cap.
+ Only meaningful if the team got correct code; uncorrect teams get 0."""
+ if not o.correct:
+ return 0.0
+ cap = max(1, int(budget.get("max_rounds", 6)))
+ used = o.rounds_to_green if o.rounds_to_green is not None else o.rounds
+ used = max(1, min(used, cap))
+ return round(1.0 - (used - 1) / cap, 4)
+
+
+def _penalty_total(o: EventOutcome) -> float:
+ # each penalty shaves a flat slice; capped so a score never goes negative.
+ return min(0.5, 0.1 * len(o.penalties))
+
+
+def score_event(o: EventOutcome, *, profile: str = "balanced",
+ budget: dict | None = None) -> dict:
+ budget = budget or {}
+ w = PROFILES.get(profile, PROFILES["balanced"])
+ axes = {
+ "correctness": 1.0 if o.correct else 0.0,
+ "speed": _speed_score(o, budget),
+ "quality": o.quality, # may be None (no judge yet)
+ "collaboration": o.collaboration,
+ }
+ # renormalize weights over axes that actually have a value this run
+ active = {k: w[k] for k, v in axes.items() if v is not None and w.get(k, 0)}
+ wsum = sum(active.values()) or 1.0
+ composite = sum(active[k] / wsum * axes[k] for k in active)
+ composite = round(max(0.0, composite - _penalty_total(o)), 4)
+
+ tok = o.tokens.get("in", 0) + o.tokens.get("out", 0)
+ return {
+ "team": o.team, "challenge": o.challenge,
+ "submitted": o.submitted, "correct": o.correct,
+ "composite": composite,
+ "axes": {k: (round(v, 4) if v is not None else None)
+ for k, v in axes.items()},
+ "pass@1": 1.0 if o.correct else 0.0,
+ "pass^1": 1.0 if o.correct else 0.0, # worst-case == pass@1 at n=1
+ "rounds": o.rounds, "rounds_to_green": o.rounds_to_green,
+ "wall_clock_s": round(o.wall_clock_s, 1),
+ "tokens": dict(o.tokens), "total_tokens": tok,
+ "cost_normalized": round(composite / tok, 8) if tok else None,
+ "penalties": list(o.penalties),
+ "defect_class": o.defect_class,
+ "model_baseline_correct": o.model_baseline_correct,
+ "profile": profile,
+ }
+
+
+def print_score(s: dict) -> None:
+ print("=" * 72)
+ print(f"score · team={s['team']} · challenge={s['challenge']} "
+ f"· profile={s['profile']}")
+ print("-" * 72)
+ g = "✓" if s["correct"] else "✗"
+ green = f"(green@{s['rounds_to_green']})" if s["rounds_to_green"] else ""
+ print(f" correct={g} composite={s['composite']:.3f} "
+ f"pass@1={s['pass@1']:.0f} rounds={s['rounds']}{green}"
+ f" {s['wall_clock_s']:.1f}s")
+ print(" axes: " + " ".join(
+ f"{k}={'—' if v is None else f'{v:.2f}'}" for k, v in s["axes"].items()))
+ print(f" tokens={s['total_tokens']} "
+ f"cost_norm={s['cost_normalized']}")
+ if s.get("defect_class") or s.get("model_baseline_correct") is not None:
+ base = s.get("model_baseline_correct")
+ base_str = "—" if base is None else ("✓" if base else "✗")
+ print(f" defect={s.get('defect_class') or 'none'} "
+ f"model_baseline={base_str}")
+ if s["penalties"]:
+ print(" penalties:")
+ for p in s["penalties"]:
+ print(f" - {p}")
+ print("=" * 72)
diff --git a/hh/scripts/bench/olympics/team.py b/hh/scripts/bench/olympics/team.py
new file mode 100644
index 0000000..dd128d2
--- /dev/null
+++ b/hh/scripts/bench/olympics/team.py
@@ -0,0 +1,58 @@
+"""Team and Member value objects (config-driven, modular per SPEC §7).
+
+A ``Member`` binds a room name to a model, a role (which selects persona +
+loop privileges) and an elicitation framing. A ``Team`` groups members under a
+topology. Same-model teams (protocol study) and mixed-model teams (capability
+study) are both just different config — no code changes.
+"""
+
+from __future__ import annotations
+
+from dataclasses import dataclass, field
+
+from . import elicitation
+
+_DRIVER_ROLES = ("builder", "architect") # who may drive IMPLEMENT, in order
+
+
+@dataclass
+class Member:
+ name: str
+ model: str
+ role: str = "peer"
+ framing: str = "neutral"
+ team: str = ""
+
+ def system_prompt(self) -> str:
+ return elicitation.system_prompt(self.name, self.team, self.role,
+ framing=self.framing)
+
+
+@dataclass
+class Team:
+ id: str
+ members: list[Member]
+ topology: str = "star" # star | chain | mesh (M3 uses this)
+ framing: str = "neutral"
+
+ def __post_init__(self):
+ for m in self.members:
+ if not m.team:
+ m.team = self.id
+
+ @property
+ def models(self) -> list[str]:
+ return [m.model for m in self.members]
+
+ def driver(self) -> Member:
+ """The member who writes to the VM in IMPLEMENT. Prefer a builder, then
+ an architect, else the first member."""
+ for role in _DRIVER_ROLES:
+ for m in self.members:
+ if m.role == role:
+ return m
+ return self.members[0]
+
+ def speaking_order(self) -> list[Member]:
+ """Round-robin order for DELIBERATE (M1). M3 will route by topology."""
+ return list(self.members)
diff --git a/hh/scripts/bench/olympics/transcript.py b/hh/scripts/bench/olympics/transcript.py
new file mode 100644
index 0000000..bc3cfd0
--- /dev/null
+++ b/hh/scripts/bench/olympics/transcript.py
@@ -0,0 +1,140 @@
+"""Replayable, OTel-aligned event log for one (team, event).
+
+Every room frame, model call, tool call, phase change, ACL grant and guard
+verdict becomes one ``Event`` whose ``kind`` follows the OpenTelemetry GenAI
+semantic conventions (``message`` / ``invoke_agent`` / ``execute_tool`` /
+``phase`` / ``acl`` / ``guard``). The transcript is a pure record: re-rendering
+it (``replay``) or re-judging it under a new rubric is a function of this file
+alone, which is the SPEC's reproducibility contract (§11).
+
+A ``Transcript`` also carries a ``manifest`` — the config hash, seed, model
+versions, challenge id and budget — so a result is self-describing.
+"""
+
+from __future__ import annotations
+
+import json
+import time
+from dataclasses import asdict, dataclass, field
+from pathlib import Path
+from typing import Any
+
+# kinds, aligned to OTel GenAI agent spans
+KIND_MESSAGE = "message" # a chat utterance posted into the room
+KIND_AGENT = "invoke_agent" # a model inference call
+KIND_TOOL = "execute_tool" # a VM command / code execution
+KIND_PHASE = "phase" # a phase transition of the loop
+KIND_ACL = "acl" # a drive-grant / revoke
+KIND_GUARD = "guard" # a destructive-guard / safety verdict
+
+
+@dataclass
+class Event:
+ ts: float
+ kind: str
+ phase: str
+ actor: str
+ payload: dict[str, Any] = field(default_factory=dict)
+ role: str = ""
+ model: str = ""
+ tokens: dict[str, int] = field(default_factory=dict)
+
+
+class Transcript:
+ """An append-only event log for one team's run at one event."""
+
+ def __init__(self, team: str, challenge: str, manifest: dict | None = None):
+ self.team = team
+ self.challenge = challenge
+ self.manifest = manifest or {}
+ self.events: list[Event] = []
+ self._t0 = time.time()
+
+ def add(self, kind: str, phase: str, actor: str, payload: dict | None = None,
+ *, role: str = "", model: str = "", tokens: dict | None = None) -> Event:
+ ev = Event(ts=round(time.time() - self._t0, 3), kind=kind, phase=phase,
+ actor=actor, payload=payload or {}, role=role, model=model,
+ tokens=tokens or {})
+ self.events.append(ev)
+ return ev
+
+ # convenience emitters --------------------------------------------------
+ def message(self, phase, actor, text, *, role="", model="", tokens=None):
+ return self.add(KIND_MESSAGE, phase, actor, {"text": text},
+ role=role, model=model, tokens=tokens)
+
+ def phase_change(self, phase, note=""):
+ return self.add(KIND_PHASE, phase, "referee", {"note": note})
+
+ def tool(self, phase, actor, payload, *, role=""):
+ return self.add(KIND_TOOL, phase, actor, payload, role=role)
+
+ def acl(self, phase, payload):
+ return self.add(KIND_ACL, phase, "referee", payload)
+
+ def guard(self, phase, actor, payload):
+ return self.add(KIND_GUARD, phase, actor, payload)
+
+ # tokens accounting -----------------------------------------------------
+ def total_tokens(self) -> dict[str, int]:
+ agg = {"in": 0, "out": 0}
+ for ev in self.events:
+ agg["in"] += ev.tokens.get("in", 0)
+ agg["out"] += ev.tokens.get("out", 0)
+ return agg
+
+ # persistence -----------------------------------------------------------
+ def to_dict(self) -> dict:
+ return {"team": self.team, "challenge": self.challenge,
+ "manifest": self.manifest,
+ "tokens_total": self.total_tokens(),
+ "events": [asdict(e) for e in self.events]}
+
+ def save(self, path: str | Path) -> Path:
+ p = Path(path)
+ p.parent.mkdir(parents=True, exist_ok=True)
+ p.write_text(json.dumps(self.to_dict(), indent=2))
+ return p
+
+ @classmethod
+ def load(cls, path: str | Path) -> "Transcript":
+ data = json.loads(Path(path).read_text())
+ t = cls(data["team"], data["challenge"], data.get("manifest", {}))
+ t.events = [Event(**e) for e in data["events"]]
+ return t
+
+
+def replay(path: str | Path, *, show_tools: bool = True) -> None:
+ """Re-render a saved transcript as a readable room log for audit."""
+ t = Transcript.load(path)
+ print("=" * 76)
+ print(f"replay · team={t.team} · challenge={t.challenge}")
+ m = t.manifest
+ if m:
+ print(f" models={m.get('models')} seed={m.get('seed')} "
+ f"budget={m.get('budget')}")
+ print("-" * 76)
+ for ev in t.events:
+ stamp = f"[{ev.ts:7.2f}s {ev.phase:<10}]"
+ if ev.kind == KIND_PHASE:
+ print(f"{stamp} ── phase: {ev.phase} {ev.payload.get('note', '')}")
+ elif ev.kind == KIND_MESSAGE:
+ print(f"{stamp} {ev.actor}({ev.role}): {ev.payload.get('text', '')}")
+ elif ev.kind == KIND_AGENT:
+ print(f"{stamp} ~ {ev.actor} infer ({ev.model}) "
+ f"tok={ev.tokens.get('out', 0)}")
+ elif ev.kind == KIND_ACL:
+ print(f"{stamp} ⚿ acl {ev.payload}")
+ elif ev.kind == KIND_GUARD:
+ print(f"{stamp} ⛨ guard {ev.payload}")
+ elif ev.kind == KIND_TOOL and show_tools:
+ p = ev.payload
+ print(f"{stamp} ⛧ {ev.actor} tool rc={p.get('rc')} "
+ f"{p.get('label', '')}")
+ out = (p.get("out") or "").strip()
+ if out:
+ for line in out.splitlines()[:8]:
+ print(f"{'':>22}| {line}")
+ print("-" * 76)
+ print(f"{len(t.events)} events · tokens={t.total_tokens()}")
+ print("=" * 76)
diff --git a/hh/scripts/bench/olympics/vm.py b/hh/scripts/bench/olympics/vm.py
new file mode 100644
index 0000000..977a846
--- /dev/null
+++ b/hh/scripts/bench/olympics/vm.py
@@ -0,0 +1,56 @@
+"""Per-team isolated workspace — a thin wrapper over ``bench/runtime.py``.
+
+The arena's IMPLEMENT/TEST phases run candidate code here. M1 reuses the
+existing ``PodmanRuntime`` (rootless, ``--network=none``, memory/pid caps — the
+"container floor" of SPEC §12) with a ``LocalRuntime`` fallback. Stronger tiers
+(gVisor, Kata/Firecracker microVMs, hosted sandboxes) are the documented upgrade
+path and slot in behind this same interface without touching the loop.
+
+A ``VM`` also keeps the last source it ran so SUBMIT can freeze the artifact for
+the judge package (the "final VM state" of SPEC §10).
+"""
+
+from __future__ import annotations
+
+from dataclasses import dataclass, field
+
+from ..langs import Lang
+from ..runtime import Exec, get_runtime
+
+
+@dataclass
+class RunResult:
+ ok: bool
+ rc: int | None
+ out: str
+ note: str = ""
+
+
+@dataclass
+class VM:
+ lang: Lang
+ runtime_kind: str = "auto" # auto | podman | local
+ _last_source: str = ""
+ _runs: list[dict] = field(default_factory=list)
+
+ def __post_init__(self):
+ self._rt = get_runtime(self.runtime_kind, self.lang)
+
+ @property
+ def runtime_name(self) -> str:
+ return self._rt.name
+
+ def run(self, source: str, timeout: float = 30.0,
+ *, label: str = "") -> RunResult:
+ """Execute one self-contained program; exit 0 == all asserts passed."""
+ self._last_source = source
+ ex: Exec = self._rt.run(self.lang, source, timeout)
+ self._runs.append({"label": label, "rc": ex.rc, "ok": ex.ok,
+ "out": ex.out, "note": ex.note})
+ return RunResult(ex.ok, ex.rc, ex.out, ex.note)
+
+ # ── submission artifact (frozen final state for the judge) ────────────
+ def freeze(self) -> dict:
+ return {"language": self.lang.id, "filename": self.lang.filename,
+ "source": self._last_source, "runtime": self.runtime_name,
+ "run_count": len(self._runs)}
diff --git a/hh/scripts/bench/suites.py b/hh/scripts/bench/suites.py
new file mode 100644
index 0000000..39638ba
--- /dev/null
+++ b/hh/scripts/bench/suites.py
@@ -0,0 +1,133 @@
+"""Problem suites: HumanEval and MBPP.
+
+A *suite* is a set of per-language bindings. Each Binding tells the harness the
+three things the dataset's *format* dictates:
+
+ • build_prompt(row) -> text fed to the model (raw continuation)
+ • stop_tokens(row) -> where to cut the completion
+ • assemble(prompt, completion, row) -> runnable program (exit 0 == all tests pass)
+
+The language's *execution* side (filename, run command, podman image) stays in
+langs.py and is shared across suites — only the problem source/format changes
+here. Adding a suite is a single table entry, mirroring how adding a language is
+a single entry in langs.py.
+
+Two dataset formats appear:
+
+ • MultiPL-E (nuprl/MultiPL-E, every `*-{js,go,rs,sh}` config, both suites): the
+ `prompt` field is a runnable function *prefix*, so the program is
+ prompt+completion+tests verbatim (_concat) and the model is simply asked to
+ continue `prompt`.
+ • MBPP-Python (google-research-datasets/mbpp): the problem is a natural-language
+ `text` description plus a `test_list` of asserts — NOT a code prefix. So the
+ generation prompt is synthesised (description + example asserts as a docstring)
+ and the program is completion+setup+asserts (the NL prompt is discarded).
+ • HumanEval-Python (openai/openai_humaneval): native HumanEval, prompt is a
+ function prefix, tests are a `check(fn)` def (langs._python).
+"""
+
+from __future__ import annotations
+
+import ast
+from dataclasses import dataclass
+from typing import Callable
+
+from .langs import LANGS, _concat, resolve
+
+
+@dataclass(frozen=True)
+class Binding:
+ dataset: str
+ config: str
+ build_prompt: Callable[[dict], str]
+ assemble: Callable[[str, str, dict], str]
+ stop_tokens: Callable[[dict], list[str]]
+
+
+# ── shared format helpers ────────────────────────────────────────────────────
+
+def _prompt_field(row: dict) -> str:
+ """MultiPL-E / HumanEval: the runnable function prefix is the prompt."""
+ return row["prompt"]
+
+
+def _parse_stop(row: dict) -> list[str]:
+ """MultiPL-E ships stop_tokens as a list or a stringified list."""
+ raw = row.get("stop_tokens")
+ if isinstance(raw, list):
+ return raw
+ if isinstance(raw, str):
+ try:
+ v = ast.literal_eval(raw)
+ return v if isinstance(v, list) else []
+ except (ValueError, SyntaxError):
+ return []
+ return []
+
+
+# ── MBPP-Python format (description + asserts, no code prefix) ────────────────
+
+# Cut as soon as the model leaves the function body for its own tests/output, so
+# only the candidate implementation survives into the assembled program.
+_MBPP_PY_STOP = ["\nassert", "\nprint(", "\n# Test", "\nif __name__", '\n"""']
+
+
+def _mbpp_py_prompt(row: dict) -> str:
+ """Frame the NL task + example asserts as a leading docstring so a raw
+ (non-chat) instruct model continues with the function definition."""
+ asserts = "\n".join(row.get("test_list", []))
+ return f'"""\n{row.get("text", "").strip()}\n\n{asserts}\n"""\n'
+
+
+def _mbpp_py_assemble(prompt: str, completion: str, row: dict) -> str:
+ """Program = setup + the model's code + the held-out asserts. The NL prompt
+ is *not* part of the program (unlike HumanEval, where it is the prefix)."""
+ setup = row.get("test_setup_code", "") or ""
+ asserts = "\n".join(row.get("test_list", []))
+ return f"{setup}\n{completion}\n{asserts}\n"
+
+
+def _mbpp_py_stop(row: dict) -> list[str]:
+ return _MBPP_PY_STOP
+
+
+# ── suite tables ─────────────────────────────────────────────────────────────
+
+def _humaneval_binding(lang_id: str) -> Binding:
+ """HumanEval reuses each language's existing langs.py dataset/config/assemble
+ (the registry already encodes the HumanEval format)."""
+ L = LANGS[lang_id]
+ return Binding(L.dataset, L.config, _prompt_field, L.assemble, _parse_stop)
+
+
+_MBPP: dict[str, Binding] = {
+ "python": Binding("google-research-datasets/mbpp", "full",
+ _mbpp_py_prompt, _mbpp_py_assemble, _mbpp_py_stop),
+ "javascript": Binding("nuprl/MultiPL-E", "mbpp-js",
+ _prompt_field, _concat, _parse_stop),
+ "go": Binding("nuprl/MultiPL-E", "mbpp-go",
+ _prompt_field, _concat, _parse_stop),
+ "rust": Binding("nuprl/MultiPL-E", "mbpp-rs",
+ _prompt_field, _concat, _parse_stop),
+ "bash": Binding("nuprl/MultiPL-E", "mbpp-sh",
+ _prompt_field, _concat, _parse_stop),
+}
+
+SUITES = {
+ "humaneval": "HumanEval (function-completion)",
+ "mbpp": "MBPP (Mostly Basic Programming Problems)",
+}
+
+
+def binding(suite: str, language: str) -> Binding:
+ """Resolve the (suite, language) -> Binding the harness should run."""
+ lang_id = resolve(language).id
+ s = suite.lower()
+ if s == "humaneval":
+ return _humaneval_binding(lang_id)
+ if s == "mbpp":
+ if lang_id not in _MBPP:
+ raise KeyError(f"suite 'mbpp' has no binding for {lang_id!r}; "
+ f"have: {', '.join(_MBPP)}")
+ return _MBPP[lang_id]
+ raise KeyError(f"unknown suite {suite!r}; known: {', '.join(SUITES)}")