From 1dac661cd6dd11b38c98ce157b6d26bfd4700b50 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Thu, 1 Oct 2026 23:50:20 +0800 Subject: [PATCH 01/32] feat(bench): add harness-free direct adapter and the evaluate command --- benchmarks/agent/adapters/direct.py | 189 ++++++++++++++++++++++++++++ benchmarks/agent/agent_bench.py | 65 +++++++++- benchmarks/agent/test_adapters.py | 67 ++++++++++ docs/agent-benchmark.md | 23 ++++ 4 files changed, 341 insertions(+), 3 deletions(-) create mode 100644 benchmarks/agent/adapters/direct.py diff --git a/benchmarks/agent/adapters/direct.py b/benchmarks/agent/adapters/direct.py new file mode 100644 index 00000000..a0d1ddd0 --- /dev/null +++ b/benchmarks/agent/adapters/direct.py @@ -0,0 +1,189 @@ +#!/usr/bin/env python3 +"""Built-in agent loop: run one benchmark trial against a model API with no agent harness around it. + +Honors the BENCH_* contract (docs/agent-benchmark.md). BENCH_MODEL is `anthropic/` (ANTHROPIC_API_KEY, optional +ANTHROPIC_BASE_URL) or `openai/` (OPENAI_API_KEY, optional OPENAI_BASE_URL, so any OpenAI-compatible endpoint works). +The model gets one `bash` tool that runs in the workspace on the shimmed PATH, plus `fetch` only when knowledge is `web`. +The system prompt lists the installed skills by name and description, and the model reads their files itself. +The loop, its limits, and the tool set are fixed here so every model faces the same protocol; they are part of the result identity. +It does not sandbox the network: pair it with the harness --canary-cmd. Exit 75 marks a provider or infrastructure failure. +""" + +from __future__ import annotations + +import json +import os +import re +import shutil +import subprocess +import sys +import time +import urllib.error +import urllib.request +from pathlib import Path + +INFRA_EXIT = 75 +MAX_TURNS = 60 +COMMAND_SECONDS = 120 +OUTPUT_CHARS = 20_000 +MAX_OUTPUT_TOKENS = 16_000 +RETRIES = 3 +SYSTEM = "You are a coding agent working in the current directory. Use the bash tool to inspect and edit files. Finish with a short summary of what you did." +BASH = {"name": "bash", "description": "Run a shell command in the workspace and return its output.", "schema": {"type": "object", "properties": {"command": {"type": "string"}}, "required": ["command"]}} +FETCH = {"name": "fetch", "description": "HTTP GET a URL and return the start of the body.", "schema": {"type": "object", "properties": {"url": {"type": "string"}}, "required": ["url"]}} + + +class ProviderError(Exception): + def __init__(self, message: str, transient: bool): + super().__init__(message) + self.transient = transient + + +def post(url: str, headers: dict, body: dict) -> dict: + request = urllib.request.Request(url, json.dumps(body).encode(), {"content-type": "application/json", **headers}) + for attempt in range(RETRIES + 1): + try: + with urllib.request.urlopen(request, timeout=600) as response: + return json.load(response) + except urllib.error.HTTPError as error: + text = error.read().decode(errors="replace")[:500] + if error.code in (408, 429) or error.code >= 500: + if attempt < RETRIES: + time.sleep(2 ** attempt * 5) + continue + raise ProviderError(f"HTTP {error.code}: {text}", True) + raise ProviderError(f"HTTP {error.code}: {text}", False) + except (urllib.error.URLError, TimeoutError, ConnectionError) as error: + if attempt < RETRIES: + time.sleep(2 ** attempt * 5) + continue + raise ProviderError(str(error), True) + raise AssertionError + + +def usage_row(inp: int, out: int, cache_read: int, cache_write: int, reasoning: int | None) -> dict: + return {"t": time.time(), "input": inp, "output": out, "cache_read": cache_read, "cache_write": cache_write, "reasoning": reasoning, "context": inp + cache_read + cache_write, "context_limit": None} + + +class Anthropic: + def __init__(self, model: str, tools: list[dict], system: str): + self.model, self.system = model, system + self.url = os.environ.get("ANTHROPIC_BASE_URL", "https://api.anthropic.com").rstrip("/") + "/v1/messages" + self.headers = {"x-api-key": os.environ["ANTHROPIC_API_KEY"], "anthropic-version": "2023-06-01"} + self.tools = [{"name": t["name"], "description": t["description"], "input_schema": t["schema"]} for t in tools] + self.messages: list[dict] = [] + + def user(self, text: str) -> None: + self.messages.append({"role": "user", "content": text}) + + def results(self, results: list[tuple[str, str]]) -> None: + self.messages.append({"role": "user", "content": [{"type": "tool_result", "tool_use_id": i, "content": out} for i, out in results]}) + + def step(self) -> tuple[str, list[tuple[str, str, dict]], dict]: + data = post(self.url, self.headers, {"model": self.model, "max_tokens": MAX_OUTPUT_TOKENS, "system": self.system, "tools": self.tools, "messages": self.messages}) + self.messages.append({"role": "assistant", "content": data["content"]}) + u = data.get("usage") or {} + text = "".join(b.get("text", "") for b in data["content"] if b["type"] == "text") + calls = [(b["id"], b["name"], b["input"]) for b in data["content"] if b["type"] == "tool_use"] + return text, calls, usage_row(u.get("input_tokens") or 0, u.get("output_tokens") or 0, u.get("cache_read_input_tokens") or 0, u.get("cache_creation_input_tokens") or 0, None) + + +class OpenAI: + def __init__(self, model: str, tools: list[dict], system: str, effort: str | None): + self.model, self.effort = model, effort + self.url = os.environ.get("OPENAI_BASE_URL", "https://api.openai.com/v1").rstrip("/") + "/chat/completions" + self.headers = {"authorization": f"Bearer {os.environ['OPENAI_API_KEY']}"} + self.tools = [{"type": "function", "function": {"name": t["name"], "description": t["description"], "parameters": t["schema"]}} for t in tools] + self.messages: list[dict] = [{"role": "system", "content": system}] + + def user(self, text: str) -> None: + self.messages.append({"role": "user", "content": text}) + + def results(self, results: list[tuple[str, str]]) -> None: + self.messages += [{"role": "tool", "tool_call_id": i, "content": out} for i, out in results] + + def step(self) -> tuple[str, list[tuple[str, str, dict]], dict]: + body = {"model": self.model, "tools": self.tools, "messages": self.messages, **({"reasoning_effort": self.effort} if self.effort else {})} + data = post(self.url, self.headers, body) + message = data["choices"][0]["message"] + self.messages.append(message) + u = data.get("usage") or {} + cached = (u.get("prompt_tokens_details") or {}).get("cached_tokens") or 0 + reasoning = (u.get("completion_tokens_details") or {}).get("reasoning_tokens") or 0 + calls = [(c["id"], c["function"]["name"], json.loads(c["function"]["arguments"] or "{}")) for c in message.get("tool_calls") or []] + return message.get("content") or "", calls, usage_row((u.get("prompt_tokens") or 0) - cached, (u.get("completion_tokens") or 0) - reasoning, cached, 0, reasoning) + + +def run_tool(name: str, args: dict, env: dict) -> str: + try: + if name == "bash": + done = subprocess.run(["bash", "--noprofile", "--norc", "-c", args["command"]], capture_output=True, text=True, timeout=COMMAND_SECONDS, env=env, errors="replace") + out = done.stdout + done.stderr + (f"\n[exit {done.returncode}]" if done.returncode else "") + elif name == "fetch": + with urllib.request.urlopen(args["url"], timeout=60) as response: + out = response.read(OUTPUT_CHARS * 2).decode(errors="replace") + else: + return f"unknown tool {name}" + except subprocess.TimeoutExpired: + return f"[timed out after {COMMAND_SECONDS}s]" + except Exception as error: + return f"[{type(error).__name__}: {error}]" + return out if len(out) <= OUTPUT_CHARS else out[:OUTPUT_CHARS] + f"\n[truncated {len(out) - OUTPUT_CHARS} characters]" + + +def skill_listing(skills: list[Path]) -> tuple[str, list[str]]: + lines, names = [], [] + for skill in skills: + target = Path(".agents/skills") / skill.name + shutil.copytree(skill, target) + text = (target / "SKILL.md").read_text() + name = re.search(r"^name:\s*(.+)$", text, re.M) + description = re.search(r"^description:\s*(.+)$", text, re.M) + names.append(name.group(1).strip() if name else skill.name) + lines.append(f"- {names[-1]}: {description.group(1).strip() if description else ''} (read {target}/SKILL.md when relevant)") + return ("\n\nSkills available:\n" + "\n".join(lines)) if lines else "", names + + +def main() -> int: + env = os.environ + provider, _, model = env["BENCH_MODEL"].partition("/") + if provider not in ("anthropic", "openai") or not model: + print("BENCH_MODEL must be anthropic/ or openai/", file=sys.stderr) + return 2 + web = env["BENCH_KNOWLEDGE"] == "web" + tools = [BASH] + ([FETCH] if web else []) + listing, loaded = skill_listing([Path(p) for p in env.get("BENCH_SKILL_DIRS", "").split(os.pathsep) if p]) + effort = env.get("BENCH_THINKING") if provider == "openai" else None + system = SYSTEM + listing + chat = Anthropic(model, tools, system) if provider == "anthropic" else OpenAI(model, tools, system, effort) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({ + "agent": "direct", "model": env["BENCH_MODEL"], "effort": effort, "tools": [t["name"] for t in tools], + "protocol": {"maxTurns": MAX_TURNS, "commandSeconds": COMMAND_SECONDS, "outputChars": OUTPUT_CHARS, "maxOutputTokens": MAX_OUTPUT_TOKENS}}, indent=2)) + Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded})) + child_env = {k: v for k, v in env.items() if k not in ("BENCH_HOST_PATH", "ANTHROPIC_API_KEY", "OPENAI_API_KEY")} + chat.user(sys.stdin.read()) + final, code = "", 0 + with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: + transcript.write(json.dumps({"t": time.time(), "type": "system", "text": system}) + "\n") + for _ in range(MAX_TURNS): + try: + text, calls, row = chat.step() + except ProviderError as error: + print(f"provider error: {error}", file=sys.stderr) + code = INFRA_EXIT if error.transient else 1 + break + usage.write(json.dumps(row) + "\n") + transcript.write(json.dumps({"t": time.time(), "type": "assistant", "text": text, "calls": [{"name": n, "input": a} for _, n, a in calls]}) + "\n") + final = text + if not calls: + break + outputs = [(i, run_tool(n, a, child_env)) for i, n, a in calls] + for (_, n, a), (_, out) in zip(calls, outputs): + transcript.write(json.dumps({"t": time.time(), "type": "tool_result", "name": n, "output": out}) + "\n") + chat.results(outputs) + sys.stdout.write(final) + return code + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 4d608ce1..c86c10bb 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -11,6 +11,7 @@ import platform import random import re +import shlex import shutil import signal import subprocess @@ -443,6 +444,53 @@ def work(job: tuple) -> None: return 3 if state["interrupted"] or state["unattempted"] else 1 if state["failed"] else 0 +ADAPTERS = {"claude-code": "claude_code.py", "pi": "pi.py", "devin": "devin.py", "codex": "codex.py", "agy": "agy.py", "direct": "direct.py"} +CANONICAL_CELL = {"tool": "wright", "skills": ["wright-skill"], "knowledge": "none", "network": "off"} +CONTROL_CELLS = [ + {"tool": "none", "skills": [], "knowledge": "none", "network": "off"}, + {"tool": "wright", "skills": [], "knowledge": "none", "network": "off"}, + CANONICAL_CELL, + {"tool": "overpy", "skills": [], "knowledge": "none", "network": "off"}, + {"tool": "overpy", "skills": ["opy-skill"], "knowledge": "none", "network": "off"}, +] + + +def cmd_evaluate(args: argparse.Namespace) -> int: + """One command from agent and model to data and document: run the matrix, then write report, score cards, and RESULTS.md. + + Needs no agent harness around it: isolation comes from the harness's own scrubbed environment, so it runs the same from a + terminal or from inside another agent's shell.""" + script = Path(__file__).parent / "adapters" / ADAPTERS[args.adapter] + effort = f"BENCH_THINKING={shlex.quote(args.effort)} " if args.effort else "" + agent_id = "-".join(filter(None, (args.adapter, args.model.replace("/", "_"), args.effort))) + cmd = f"BENCH_MODEL={shlex.quote(args.model)} {effort}{shlex.quote(sys.executable)} {shlex.quote(str(script))}" + wanted = [CANONICAL_CELL] if args.cells == "score" else CONTROL_CELLS + cells = [c for c in wanted if all(s in args.skill_dirs for s in c["skills"])] + dropped = [cell_label(normalize_cell(c)) for c in wanted if c not in cells] + if dropped: + print(f"skipped cells without --skill-dir: {', '.join(dropped)}", flush=True) + if not cells: + raise SystemExit("no cell can run: pass --skill-dir wright-skill=DIR") + scenarios = args.scenarios or [s for s in all_scenario_ids() if args.split == "all" or load_scenario(s).get("split") == args.split] + args.out = args.out / args.name + args.out.mkdir(parents=True, exist_ok=True) + config = {"agents": [{"id": agent_id, "cmd": cmd}], "cells": cells, "scenarios": scenarios, "trials": args.trials, "parallel": args.parallel, "seed": args.seed, + "options": {"skill_dirs": {k: str(v) for k, v in args.skill_dirs.items()}, "wiki_dir": str(args.wiki_dir) if args.wiki_dir else None}} + args.config = args.out / "matrix.json" + args.config.write_text(json.dumps(config, indent=2) + "\n") + status = cmd_matrix(args) + if not list(args.out.glob("*/*/*/result.json")): + return status or 1 + bench_report.main([args.out], args.wright, False, load_scenario, bench_report.BASELINE) + languages = ["workshop", "opy"] + expected = {lang: [s for s in all_scenario_ids() if load_scenario(s)["language"] == lang and load_scenario(s).get("split") == "test"] for lang in languages} + bench_score.main([args.out], languages, expected, None) + (args.out / "RESULTS.md").write_text(f"# Agent benchmark results: {args.name}\n\nAgent `{agent_id}`. Generated by `agent_bench.py evaluate`; `matrix.json` reproduces it.\n\n" + f"## Score\n\n```\n{(args.out / 'score.txt').read_text()}```\n\n{(args.out / 'report.md').read_text()}") + print(f"wrote {args.out / 'RESULTS.md'}") + return status + + def cmd_wiki_snapshot(args: argparse.Namespace) -> int: record = bench_wiki.snapshot(args.base, args.dir, tuple(args.categories)) print(f"{len(record['documents'])} document(s) from {record['source']} into {args.dir}\nsnapshotSha256 {record['snapshotSha256']}") @@ -464,11 +512,11 @@ def main() -> int: return bench_trace.shim_main(sys.argv[2:]) parser = argparse.ArgumentParser(description=__doc__) sub = parser.add_subparsers(dest="command", required=True) - for name in ("validate", "run", "matrix"): + for name in ("validate", "run", "matrix", "evaluate"): p = sub.add_parser(name) p.add_argument("--wright", default=str(ROOT / "target/debug/wright"), help="Wright binary under test") p.add_argument("--out", type=Path, default=Path.home() / ".cache/wright-agent-bench", help="outside any repository, so agents cannot discover its instruction files") - for name in ("run", "matrix"): + for name in ("run", "matrix", "evaluate"): p = sub.choices[name] p.add_argument("--skill-dir", action="append", default=[], metavar="NAME=DIR", help=f"pinned skill directory for one of {', '.join(SKILLS)}; repeatable") p.add_argument("--wiki-dir", type=Path, help="pinned wiki snapshot, linked as ./wiki for knowledge 'wiki'; content hashes are verified") @@ -488,6 +536,17 @@ def main() -> int: run.add_argument("--network", choices=("off", "on"), default="off") run.add_argument("--trials", type=int, default=1) sub.choices["matrix"].add_argument("config", type=Path, help="JSON: agents[{id,cmd}], cells[{tool,skills,knowledge,network}], scenarios, trials, parallel, seed, options") + ev = sub.choices["evaluate"] + ev.add_argument("--adapter", choices=sorted(ADAPTERS), required=True, help="agent adapter; `direct` is the built-in loop that needs no agent harness") + ev.add_argument("--model", required=True, help="BENCH_MODEL, in the form the adapter expects") + ev.add_argument("--effort", help="BENCH_THINKING, where the adapter supports it") + ev.add_argument("--name", default=time.strftime("%Y%m%d-%H%M%S"), help="run directory under --out") + ev.add_argument("--cells", choices=("score", "controls"), default="score", help="score: the canonical cell only; controls: also baseline and language-appropriate controls") + ev.add_argument("--split", choices=("test", "train", "all"), default="test") + ev.add_argument("--scenarios", nargs="*", choices=all_scenario_ids()) + ev.add_argument("--trials", type=int, default=3) + ev.add_argument("--parallel", type=int, default=2) + ev.add_argument("--seed", type=int, default=1) sub.add_parser("setup-oracle", help="install the pinned upstream OverPy oracle") skill = sub.add_parser("wiki-skill", help="build the progressive-disclosure workshop-wiki skill from a wiki snapshot") skill.add_argument("--snapshot", type=Path, required=True) @@ -532,7 +591,7 @@ def main() -> int: languages = args.language or ["workshop", "opy"] expected = {lang: [s for s in all_scenario_ids() if load_scenario(s)["language"] == lang and load_scenario(s).get("split") == "test"] for lang in languages} return bench_score.main(args.dirs, languages, expected, None) - return cmd_run(args) if args.command == "run" else cmd_matrix(args) + return {"run": cmd_run, "matrix": cmd_matrix, "evaluate": cmd_evaluate}[args.command](args) if __name__ == "__main__": diff --git a/benchmarks/agent/test_adapters.py b/benchmarks/agent/test_adapters.py index f2e42d8c..8d0b07b4 100644 --- a/benchmarks/agent/test_adapters.py +++ b/benchmarks/agent/test_adapters.py @@ -11,6 +11,11 @@ import pi import codex import agy +import direct +import os +import tempfile +import threading +from http.server import BaseHTTPRequestHandler, HTTPServer class PiAdapterTest(unittest.TestCase): @@ -112,5 +117,67 @@ def test_codex_builtin_skills_are_separate_from_observed_project_skills(self): self.assertEqual(codex.loaded_skills(text), (["wright"], ["openai-docs"])) +class DirectAdapterTest(unittest.TestCase): + def serve(self, replies): + seen = [] + + class Handler(BaseHTTPRequestHandler): + def do_POST(self): + seen.append(json.loads(self.rfile.read(int(self.headers["content-length"])))) + status, body = replies[min(len(seen), len(replies)) - 1] + self.send_response(status) + self.end_headers() + self.wfile.write(json.dumps(body).encode()) + + def log_message(self, *args): + pass + + server = HTTPServer(("127.0.0.1", 0), Handler) + threading.Thread(target=server.serve_forever, daemon=True).start() + self.addCleanup(server.shutdown) + return f"http://127.0.0.1:{server.server_port}", seen + + def run_direct(self, model, base_var, replies): + base, seen = self.serve(replies) + with tempfile.TemporaryDirectory(dir=Path(__file__).parent) as tmp: + tmp = Path(tmp) + skill = tmp / "demo" + skill.mkdir() + (skill / "SKILL.md").write_text("---\nname: demo\ndescription: A demo skill\n---\nbody\n") + env = {"BENCH_MODEL": model, "BENCH_KNOWLEDGE": "none", "BENCH_SKILL_DIRS": str(skill), "ANTHROPIC_API_KEY": "k", "OPENAI_API_KEY": "k", base_var: base, + **{f"BENCH_{n}": str(tmp / n.lower()) for n in ("USAGE", "TRANSCRIPT", "CONTEXT", "AGENT_INFO")}, "PATH": os.environ["PATH"]} + cwd = os.getcwd() + os.makedirs(tmp / "work") + os.chdir(tmp / "work") + out = io.StringIO() + try: + with patch.dict(direct.os.environ, env, clear=True), patch.object(direct.sys, "stdin", io.StringIO("task")), patch.object(direct.sys, "stdout", out): + code = direct.main() + finally: + os.chdir(cwd) + read = lambda n: (tmp / n).read_text() + return code, out.getvalue(), seen, [json.loads(l) for l in read("usage").splitlines()], json.loads(read("context")), json.loads(read("agent_info")) + + def test_anthropic_loop_runs_a_tool_and_records_usage(self): + use = {"content": [{"type": "tool_use", "id": "t1", "name": "bash", "input": {"command": "echo hi"}}], "usage": {"input_tokens": 10, "output_tokens": 2, "cache_read_input_tokens": 5}} + done = {"content": [{"type": "text", "text": "done"}], "usage": {"input_tokens": 20, "output_tokens": 3}} + code, final, seen, usage, context, info = self.run_direct("anthropic/m", "ANTHROPIC_BASE_URL", [(200, use), (200, done)]) + self.assertEqual((code, final, len(seen)), (0, "done", 2)) + self.assertEqual(seen[1]["messages"][-1]["content"][0]["content"].strip(), "hi") + self.assertIn("demo: A demo skill", seen[0]["system"]) + self.assertEqual((usage[0]["input"], usage[0]["cache_read"], usage[0]["context"]), (10, 5, 15)) + self.assertEqual((context["loaded"], info["agent"], [t["name"] for t in seen[0]["tools"]]), (["demo"], "direct", ["bash"])) + + def test_openai_usage_splits_cached_and_reasoning_tokens(self): + reply = {"choices": [{"message": {"role": "assistant", "content": "ok"}}], "usage": {"prompt_tokens": 100, "completion_tokens": 20, "prompt_tokens_details": {"cached_tokens": 60}, "completion_tokens_details": {"reasoning_tokens": 12}}} + code, final, _, usage, _, _ = self.run_direct("openai/m", "OPENAI_BASE_URL", [(200, reply)]) + self.assertEqual((code, final), (0, "ok")) + self.assertEqual((usage[0]["input"], usage[0]["cache_read"], usage[0]["output"], usage[0]["reasoning"]), (40, 60, 8, 12)) + + def test_client_error_is_not_an_infrastructure_failure(self): + code, *_ = self.run_direct("anthropic/m", "ANTHROPIC_BASE_URL", [(400, {"error": "bad"})]) + self.assertEqual(code, 1) + + if __name__ == "__main__": unittest.main() diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index 65989ea6..38df5240 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -290,6 +290,29 @@ model, effort, or protocol, and it is marked provisional with fewer than eight held-out scenarios, missing scenarios, or unequal trials. The card discloses `networkEnforcement` (`declared-only` or `canary-checked`). +## Evaluating without an agent harness + +`agent_bench.py evaluate --adapter ADAPTER --model MODEL --skill-dir wright-skill=DIR` +is the one-command entry: it writes `matrix.json`, runs the cells, then writes +`report.md`, `summary.json`, `score.json`, `score.txt`, and `RESULTS.md` into +`/`. `--cells score` runs the canonical cell only; `--cells controls` +adds the baseline, `wright` without the skill, and, for OverPy scenarios, the +`overpy` controls (cells whose skill has no `--skill-dir` are skipped). It runs the +same from a terminal or from inside another agent's shell, because isolation comes +from the harness's scrubbed environment, not from its parent. It runs locally; CI +does not run it. + +`--adapter direct` is the built-in loop (`adapters/direct.py`) that needs no agent +harness: it calls a model API with one `bash` tool (and `fetch` only for +knowledge `web`), lists the installed skills by name and description, and records +exact usage and a full transcript. `BENCH_MODEL` is `anthropic/` +(`ANTHROPIC_API_KEY`, optional `ANTHROPIC_BASE_URL`) or `openai/` +(`OPENAI_API_KEY`, optional `OPENAI_BASE_URL`, so OpenAI-compatible endpoints work). +Pass the key variables with `--env-pass`. Its turn, time, and output limits are +recorded in `agentInfo.protocol`. It does not sandbox the network, so pair it with +`--canary-cmd`. Scores from `direct` and from product harnesses measure different +things and are not mixed. + ## Cadence The benchmark does not gate pull requests. Run it manually or on a schedule once From 89e96595291dd3cf4de89872c80fc9479c46d9ea Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Fri, 2 Oct 2026 00:32:16 +0800 Subject: [PATCH 02/32] feat(bench): add opencode and grok adapters and report the agent setup actually observed --- benchmarks/agent/adapters/agy.py | 4 +- benchmarks/agent/adapters/claude_code.py | 8 +- benchmarks/agent/adapters/codex.py | 4 +- benchmarks/agent/adapters/common.py | 14 ++++ benchmarks/agent/adapters/devin.py | 4 +- benchmarks/agent/adapters/grok.py | 92 ++++++++++++++++++++++ benchmarks/agent/adapters/opencode.py | 98 ++++++++++++++++++++++++ benchmarks/agent/adapters/pi.py | 4 +- benchmarks/agent/agent_bench.py | 2 +- benchmarks/agent/bench_grade.py | 2 +- benchmarks/agent/bench_report.py | 17 ++++ benchmarks/agent/test_adapters.py | 20 +++++ benchmarks/agent/test_agent_bench.py | 9 +++ docs/agent-benchmark.md | 13 +++- 14 files changed, 283 insertions(+), 8 deletions(-) create mode 100644 benchmarks/agent/adapters/common.py create mode 100644 benchmarks/agent/adapters/grok.py create mode 100644 benchmarks/agent/adapters/opencode.py diff --git a/benchmarks/agent/adapters/agy.py b/benchmarks/agent/adapters/agy.py index eab1d515..1d8b27e1 100644 --- a/benchmarks/agent/adapters/agy.py +++ b/benchmarks/agent/adapters/agy.py @@ -11,6 +11,8 @@ import time from pathlib import Path +from common import cli_version + def usage_row(usage: dict, timestamp: float) -> dict: cached = usage.get("cache_read_tokens") or 0 @@ -63,7 +65,7 @@ def main() -> int: result = event["result"] code = process.wait() Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"installed": installed, "audit": "isolated-home; CLI does not export loaded skill context", "unexpected": sorted(unexpected)})) - Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "agy", "model": env["BENCH_MODEL"], "effort": env["BENCH_THINKING"], "tools": None, "note": "the CLI does not expose its tool list"}, indent=2)) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "agy", "version": cli_version(binary), "model": env["BENCH_MODEL"], "effort": env["BENCH_THINKING"], "tools": None, "note": "the CLI does not expose its tool list"}, indent=2)) (run / "adapter.json").write_text(json.dumps({"agent": "agy", "requestedModel": env["BENCH_MODEL"], "requestedEffort": env["BENCH_THINKING"], "observedModel": init.get("model"), "status": result.get("status"), "usageSource": "stream-step-usage", "providerUsage": result.get("usage")}, indent=2)) sys.stdout.write(result.get("response", "")) stderr_text = (run / "agy-stderr.log").read_text() diff --git a/benchmarks/agent/adapters/claude_code.py b/benchmarks/agent/adapters/claude_code.py index add3a71a..22c016e3 100755 --- a/benchmarks/agent/adapters/claude_code.py +++ b/benchmarks/agent/adapters/claude_code.py @@ -20,6 +20,8 @@ import time from pathlib import Path +from common import cli_version + INFRA_EXIT = 75 TOOLS = ["Bash", "Read", "Edit", "Write", "Glob", "Grep"] WEB_TOOLS = ["WebFetch", "WebSearch"] @@ -47,7 +49,8 @@ def main() -> int: shutil.copytree(skill, plugin / "skills" / skill.name) loaded.append(skill.name) cmd += ["--plugin-dir", str(plugin)] - Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "claude-code", "model": env.get("BENCH_MODEL", "sonnet"), "tools": TOOLS + (WEB_TOOLS if web else [])}, indent=2)) + claude = cmd[0] + init: dict = {} proc = subprocess.Popen( cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env={**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "CLAUDE_CODE_DISABLE_CLAUDE_MDS": "1"}, @@ -62,6 +65,8 @@ def main() -> int: except json.JSONDecodeError: continue transcript.write(json.dumps({"t": time.time(), **event}) + "\n") + if event.get("type") == "system" and event.get("subtype") == "init": + init = event if event.get("type") == "assistant": u = event["message"].get("usage") or {} cached = u.get("cache_read_input_tokens") or 0 @@ -76,6 +81,7 @@ def main() -> int: stderr = proc.stderr.read() code = proc.wait() Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded})) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "claude-code", "version": cli_version(claude), "model": init.get("model") or env.get("BENCH_MODEL", "sonnet"), "tools": init.get("tools") or TOOLS + (WEB_TOOLS if web else []), "toolsSource": "init" if init.get("tools") else "requested", "mcpServers": init.get("mcp_servers")}, indent=2)) sys.stdout.write(final) sys.stderr.write(stderr) if errored or code != 0: diff --git a/benchmarks/agent/adapters/codex.py b/benchmarks/agent/adapters/codex.py index cecc6848..e27f620d 100644 --- a/benchmarks/agent/adapters/codex.py +++ b/benchmarks/agent/adapters/codex.py @@ -13,6 +13,8 @@ from datetime import datetime from pathlib import Path +from common import cli_version + def usage_row(usage: dict, timestamp: float, limit: int | None) -> dict: cached = usage.get("cached_input_tokens") or 0 @@ -111,7 +113,7 @@ def main() -> int: loaded += names builtin += builtins Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": sorted(set(loaded) | {f"unexpected-mcp:{server}" for server in servers}), "builtinSkills": sorted(set(builtin))})) - Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "codex", "model": observed.get("model") or model, "effort": observed.get("effort") or effort, "sandbox": "danger-full-access inside the harness file sandbox", "toolsObserved": sorted(t for t in item_types if t)}, indent=2)) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "codex", "version": cli_version(binary), "model": observed.get("model") or model, "effort": observed.get("effort") or effort, "sandbox": "danger-full-access inside the harness file sandbox", "toolsObserved": sorted(t for t in item_types if t)}, indent=2)) (run / "adapter.json").write_text(json.dumps({"agent": "codex", "requestedModel": model, "requestedEffort": effort, "observed": observed, "usageSource": "session-token-count" if observed else "turn-summary"}, indent=2)) sys.stdout.write(final) stderr_text = (run / "codex-stderr.log").read_text() diff --git a/benchmarks/agent/adapters/common.py b/benchmarks/agent/adapters/common.py new file mode 100644 index 00000000..bb0c3de2 --- /dev/null +++ b/benchmarks/agent/adapters/common.py @@ -0,0 +1,14 @@ +"""Helpers shared by the adapters.""" + +from __future__ import annotations + +import subprocess + + +def cli_version(binary: str, env: dict | None = None) -> str | None: + """First line of ` --version`, or None when the CLI does not answer.""" + try: + done = subprocess.run([binary, "--version"], capture_output=True, text=True, timeout=30, env=env) + except (OSError, subprocess.TimeoutExpired): + return None + return (done.stdout or done.stderr).strip().splitlines()[0] if (done.stdout or done.stderr).strip() else None diff --git a/benchmarks/agent/adapters/devin.py b/benchmarks/agent/adapters/devin.py index 383a7ba1..cefd0fe3 100755 --- a/benchmarks/agent/adapters/devin.py +++ b/benchmarks/agent/adapters/devin.py @@ -22,6 +22,8 @@ from datetime import datetime from pathlib import Path +from common import cli_version + INFRA_EXIT = 75 TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily", "usage limit", "quota", "insufficient credits", "fetch failed", "websocket error", "connection error") WEB_TOOLS = ["WebFetch", "WebSearch", "webfetch", "web_search"] @@ -90,7 +92,7 @@ def main() -> int: Path(env["BENCH_USAGE"]).write_text("".join(json.dumps(r) + "\n" for r in rows)) Path(env["BENCH_TRANSCRIPT"]).write_text("".join(json.dumps(s) + "\n" for s in (json.loads(export.read_text())["steps"] if export.is_file() else []))) exported = json.loads(export.read_text()) if export.is_file() else {} - Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "devin", "model": model, "tools": [t["function"]["name"] for t in exported.get("agent", {}).get("tool_definitions", [])], "denied": MCP_TOOLS + ([] if env["BENCH_KNOWLEDGE"] == "web" else WEB_TOOLS)}, indent=2)) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "devin", "version": cli_version(devin), "model": model, "tools": [t["function"]["name"] for t in exported.get("agent", {}).get("tool_definitions", [])], "denied": MCP_TOOLS + ([] if env["BENCH_KNOWLEDGE"] == "web" else WEB_TOOLS)}, indent=2)) Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded, "ignoredPluginSkills": plugins})) sys.stdout.write(proc.stdout) sys.stderr.write(proc.stderr) diff --git a/benchmarks/agent/adapters/grok.py b/benchmarks/agent/adapters/grok.py new file mode 100644 index 00000000..04d5c85b --- /dev/null +++ b/benchmarks/agent/adapters/grok.py @@ -0,0 +1,92 @@ +#!/usr/bin/env python3 +"""Adapter for the Grok CLI (`grok -p`): run one benchmark trial and report per-turn usage. + +BENCH_MODEL is a model id from `grok models` and must be set in --agent-cmd. The run uses an isolated GROK_HOME that holds only the +login (auth.json) and the benchmark skills, so no global config, rules, skills, or plugins load. The prompt is sent verbatim. +Subagents are disabled so each assistant message is one model call, and web tools are disabled unless knowledge is `web`. +The tools and skills the agent actually had come from the stream's init line. BENCH_THINKING is the reasoning effort. +The network is not sandboxed: pair this with the harness --canary-cmd. Exit 75 marks a provider or infrastructure failure. +""" + +from __future__ import annotations + +import json +import os +import shutil +import subprocess +import sys +import time +from pathlib import Path + +from common import cli_version + +INFRA_EXIT = 75 +TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily", "usage limit", "quota", "insufficient credits", "connection error", "unavailable") + + +def usage_row(usage: dict, limit: int | None, now: float) -> dict: + read, write = usage.get("cache_read_input_tokens") or 0, usage.get("cache_creation_input_tokens") or 0 + return {"t": now, "input": usage.get("input_tokens"), "output": usage.get("output_tokens"), "cache_read": read, "cache_write": write, + "reasoning": None, "context": (usage.get("input_tokens") or 0) + read + write, "context_limit": limit} + + +def main() -> int: + env = os.environ + model = env.get("BENCH_MODEL") or sys.exit("BENCH_MODEL is required: set it in --agent-cmd, for example BENCH_MODEL=grok-4.7 python3 adapters/grok.py") + run_dir, real_home = Path(env["BENCH_RUN_DIR"]), Path(env["HOME"]) + grok_home = run_dir / "grok-home" + (grok_home / "skills").mkdir(parents=True) + shutil.copy(real_home / ".grok/auth.json", grok_home / "auth.json") + for skill in (Path(p) for p in env.get("BENCH_SKILL_DIRS", "").split(os.pathsep) if p): + shutil.copytree(skill, grok_home / "skills" / skill.name) + web = env["BENCH_KNOWLEDGE"] == "web" + grok = shutil.which("grok", path=env.get("BENCH_HOST_PATH")) or "grok" + child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "GROK_HOME": str(grok_home), "HOME": str(run_dir / "grok-user"), "GROK_TELEMETRY_ENABLED": "0", "GROK_DISABLE_AUTOUPDATER": "1"} + (run_dir / "grok-user").mkdir() + prompt = run_dir / "grok-prompt.txt" + prompt.write_text(sys.stdin.read()) + cmd = [grok, "--verbatim", "--output-format", "streaming-messages-json", "--permission-mode", "bypassPermissions", "--no-subagents", "-m", model, "--prompt-file", str(prompt)] + if not web: + cmd.append("--disable-web-search") + if env.get("BENCH_THINKING"): + cmd += ["--reasoning-effort", env["BENCH_THINKING"]] + version = cli_version(grok, child_env) + proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=child_env) + init: dict = {} + pending: list[dict] = [] + final, error, limit = "", "", None + with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: # line-buffered: a killed run keeps its usage + for line in proc.stdout: + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + now = time.time() + transcript.write(json.dumps({"t": now, **event}) + "\n") + if event["type"] == "system" and event.get("subtype") == "init": + init = event + elif event["type"] == "assistant": + pending.append(event["message"].get("usage") or {}) + text = "".join(b.get("text", "") for b in event["message"].get("content", []) if b.get("type") == "text") + final = text or final + elif event["type"] == "result": + final = event.get("result") or final + if event.get("is_error"): + error = json.dumps(event.get("errors") or event.get("result") or "error") + limit = next((m.get("contextWindow") for m in (event.get("modelUsage") or {}).values()), None) + for u in pending: # the context window is only known from the final result line + usage.write(json.dumps(usage_row(u, limit, time.time())) + "\n") + stderr = proc.stderr.read() + code = proc.wait() + Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": init.get("skills") or []})) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "grok", "version": version, "model": init.get("model") or model, "effort": env.get("BENCH_THINKING"), "tools": init.get("tools"), + "mcpServers": init.get("mcp_servers"), "disabled": ["subagents"] + ([] if web else ["web"])}, indent=2)) + sys.stdout.write(final) + sys.stderr.write(stderr) + if error or code != 0: + return INFRA_EXIT if any(s in (error + stderr).lower() for s in TRANSIENT) else (code or 1) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/agent/adapters/opencode.py b/benchmarks/agent/adapters/opencode.py new file mode 100644 index 00000000..b3a7cbf5 --- /dev/null +++ b/benchmarks/agent/adapters/opencode.py @@ -0,0 +1,98 @@ +#!/usr/bin/env python3 +"""Adapter for opencode (`opencode run`): run one benchmark trial and report per-turn usage. + +BENCH_MODEL is `provider/model` as `opencode models` lists it, and must be set in --agent-cmd. The run uses an isolated HOME with +only the opencode credentials copied in, so no global instructions, agents, skills, or plugins load (`--pure`). Skills are installed in the +workspace under `.opencode/skills`; opencode also reads the real home's `~/.claude` and `~/.agents` skills and Claude Code instructions regardless of HOME, +so those scans are disabled by environment variable. Web tools are denied unless knowledge is `web`. BENCH_THINKING is passed as the model variant. +The network is not sandboxed: pair this with the harness --canary-cmd. Exit 75 marks a provider or infrastructure failure. +""" + +from __future__ import annotations + +import json +import os +import shutil +import subprocess +import sys +import time +from pathlib import Path + +from common import cli_version + +INFRA_EXIT = 75 +TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily", "usage limit", "quota", "insufficient credits", "fetch failed", "connection error", "econnreset") +WEB_PERMISSIONS = {"webfetch": "deny", "websearch": "deny"} + + +def usage_row(tokens: dict, now: float) -> dict: + cache = tokens.get("cache") or {} + read, write = cache.get("read") or 0, cache.get("write") or 0 + return {"t": now, "input": tokens.get("input"), "output": tokens.get("output"), "cache_read": read, "cache_write": write, + "reasoning": tokens.get("reasoning"), "context": (tokens.get("input") or 0) + read + write, "context_limit": None} + + +def available_skills(raw: str) -> list[str]: + """Skill names opencode reports, without its built-in ones.""" + try: + return [s["name"] for s in json.loads(raw) if s.get("location") != ""] + except (json.JSONDecodeError, TypeError): + return [] + + +def main() -> int: + env = os.environ + model = env.get("BENCH_MODEL") or sys.exit("BENCH_MODEL is required: set it in --agent-cmd, for example BENCH_MODEL=openai/gpt-6-luna python3 adapters/opencode.py") + run_dir, workspace, real_home = Path(env["BENCH_RUN_DIR"]), Path.cwd(), Path(env["HOME"]) + home = run_dir / "opencode-home" + (home / ".local/share/opencode").mkdir(parents=True) + (home / ".config/opencode").mkdir(parents=True) + shutil.copy(real_home / ".local/share/opencode/auth.json", home / ".local/share/opencode/auth.json") + for skill in (Path(p) for p in env.get("BENCH_SKILL_DIRS", "").split(os.pathsep) if p): + shutil.copytree(skill, workspace / ".opencode/skills" / skill.name) + web = env["BENCH_KNOWLEDGE"] == "web" + opencode = shutil.which("opencode", path=env.get("BENCH_HOST_PATH")) or "opencode" + child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home), "XDG_CONFIG_HOME": str(home / ".config"), "XDG_DATA_HOME": str(home / ".local/share"), + "OPENCODE_DISABLE_CLAUDE_CODE": "1", "OPENCODE_DISABLE_EXTERNAL_SKILLS": "1", + "OPENCODE_CONFIG_CONTENT": json.dumps({"$schema": "https://opencode.ai/config.json", "autoupdate": False, "share": "disabled", **({} if web else {"permission": WEB_PERMISSIONS})})} + version = cli_version(opencode, child_env) + listing = run_dir / "opencode-skills.json" # a file, not a pipe: `debug skill` truncates large output when stdout is a pipe + with open(listing, "w") as out: + subprocess.run([opencode, "debug", "skill"], stdout=out, env=child_env) + loaded = available_skills(listing.read_text()) + cmd = [opencode, "run", "--pure", "--format", "json", "--auto", "-m", model] + if env.get("BENCH_THINKING"): + cmd += ["--variant", env["BENCH_THINKING"]] + proc = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=child_env) + proc.stdin.write(sys.stdin.read()) + proc.stdin.close() + final, error, tools = "", "", set() + with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: # line-buffered: a killed run keeps its usage + for line in proc.stdout: + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + now, part = time.time(), event.get("part") or {} + transcript.write(json.dumps({"t": now, **event}) + "\n") + if event["type"] == "step_finish" and part.get("tokens"): + usage.write(json.dumps(usage_row(part["tokens"], now)) + "\n") + elif event["type"] == "text": + final = part.get("text") or final + elif event["type"] == "tool_use": + tools.add(part.get("tool")) + elif event["type"] == "error": + error = json.dumps(event.get("error")) + stderr = proc.stderr.read() + code = proc.wait() + Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded})) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "opencode", "version": version, "model": model, "effort": env.get("BENCH_THINKING"), "tools": sorted(tools), "toolsNote": "tools the agent used; the CLI does not list its tools", "denied": [] if web else sorted(WEB_PERMISSIONS)}, indent=2)) + sys.stdout.write(final) + sys.stderr.write(stderr) + if error or code != 0: + return INFRA_EXIT if any(s in (error + stderr).lower() for s in TRANSIENT) else (code or 1) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/agent/adapters/pi.py b/benchmarks/agent/adapters/pi.py index 134874af..9f3640cf 100755 --- a/benchmarks/agent/adapters/pi.py +++ b/benchmarks/agent/adapters/pi.py @@ -21,6 +21,8 @@ import time from pathlib import Path +from common import cli_version + INFRA_EXIT = 75 TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily", "usage limit", "quota", "insufficient credits", "fetch failed", "websocket error", "connection error") SCALE = {"K": 1_000, "M": 1_000_000} @@ -80,7 +82,7 @@ def main() -> int: cmd += ["--thinking", env["BENCH_THINKING"]] child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home), "PI_CODING_AGENT_DIR": str(state)} limit = context_limit(pi, model, child_env, extensions) - Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "pi", "model": model, "effort": env.get("BENCH_THINKING"), "tools": ["read", "bash", "edit", "write"], "extensions": [e for e in extensions if e]}, indent=2)) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "pi", "version": cli_version(pi), "model": model, "effort": env.get("BENCH_THINKING"), "tools": ["read", "bash", "edit", "write"], "extensions": [e for e in extensions if e]}, indent=2)) proc = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=child_env) proc.stdin.write(prompt) proc.stdin.close() diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index c86c10bb..ed2240c7 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -444,7 +444,7 @@ def work(job: tuple) -> None: return 3 if state["interrupted"] or state["unattempted"] else 1 if state["failed"] else 0 -ADAPTERS = {"claude-code": "claude_code.py", "pi": "pi.py", "devin": "devin.py", "codex": "codex.py", "agy": "agy.py", "direct": "direct.py"} +ADAPTERS = {"claude-code": "claude_code.py", "pi": "pi.py", "devin": "devin.py", "codex": "codex.py", "agy": "agy.py", "opencode": "opencode.py", "grok": "grok.py", "direct": "direct.py"} CANONICAL_CELL = {"tool": "wright", "skills": ["wright-skill"], "knowledge": "none", "network": "off"} CONTROL_CELLS = [ {"tool": "none", "skills": [], "knowledge": "none", "network": "off"}, diff --git a/benchmarks/agent/bench_grade.py b/benchmarks/agent/bench_grade.py index 6e07a4fc..7b1ef8f4 100644 --- a/benchmarks/agent/bench_grade.py +++ b/benchmarks/agent/bench_grade.py @@ -13,7 +13,7 @@ ORACLE = HERE / "oracle" GRADER_FILES = ("bench_grade.py", "oracle/compile.js", "oracle/package-lock.json") SUITE_VERSION = "v1" -UNSAFE_IGNORED = ("wiki", ".agents", ".devin") # linked wiki and skills installed through the agent's own mechanism +UNSAFE_IGNORED = ("wiki", ".agents", ".devin", ".opencode") # linked wiki and skills installed through the agent's own mechanism def wright_json(wright: str, args: list[str]) -> tuple[int, dict]: diff --git a/benchmarks/agent/bench_report.py b/benchmarks/agent/bench_report.py index 162de014..0aa2afee 100644 --- a/benchmarks/agent/bench_report.py +++ b/benchmarks/agent/bench_report.py @@ -147,6 +147,22 @@ def regrade_notes(runs: list[dict], wright: str, load_scenario) -> list[str]: return [f"GRADER: unstable verdict on {len(unstable)} workspace(s): {unstable}"] if unstable else ["GRADER: consistent on every regraded workspace."] +def setup_rows(runs: list[dict]) -> list[str]: + """What each agent was actually given: CLI version, model, tools, and loaded skills, as the adapters observed them.""" + out = ["", "## Agent setup", "", "What the adapters observed, not what was requested. `not recorded` means the CLI does not expose it.", "", + "| agent | condition | CLI | model | effort | tools | loaded skills |", "| --- | --- | --- | --- | --- | --- | --- |"] + for agent in sorted({r["agent"]["id"] for r in runs}): + for cell in sorted({label(r) for r in runs if r["agent"]["id"] == agent}): + group = [r for r in runs if r["agent"]["id"] == agent and label(r) == cell] + infos = [r.get("agentInfo") or {} for r in group] + first = next((i for i in infos if i), {}) + tools = first.get("tools") or first.get("toolsObserved") + tool_text = "not recorded" if tools is None else f"{len(tools)}: {', '.join(tools[:12])}{' ...' if len(tools) > 12 else ''}" if tools else "none" + loaded = sorted({s for r in group for s in ((r.get("context") or {}).get("loaded") or [])}) + out.append(f"| {agent} | {cell} | {first.get('version') or 'not recorded'} | {first.get('model') or 'not recorded'} | {first.get('effort') or 'not recorded'} | {tool_text} | {', '.join(loaded) or 'none'} |") + return out + + def render(results: list[dict], regrade: list[str] | None = None, reference: str = BASELINE) -> tuple[str, dict]: invalid = [r for r in results if r["status"] == "invalid"] infrastructure = [r for r in results if r["status"] == "provider-interrupted"] @@ -163,6 +179,7 @@ def render(results: list[dict], regrade: list[str] | None = None, reference: str summary["cells"][f"{agent}|{cell}"] = row out.append(f"| {agent} | {cell} | {row['n']} | {rate_runs(group)} | {row['passed']}/{row['n']} | {row['usedTool']}/{row['n']} | " f"{fmt(row['tokens'])} | {fmt(row['tokensPerUsable'])} | {fmt(row['peakContext'])} | {fmt(row['seconds'], 1)} |") + out += setup_rows(runs) out += ["", "## By scenario", "", "| scenario | agent | condition | usable |", "| --- | --- | --- | --- |"] groups: dict[tuple, list[dict]] = defaultdict(list) for r in runs: diff --git a/benchmarks/agent/test_adapters.py b/benchmarks/agent/test_adapters.py index 8d0b07b4..e174d271 100644 --- a/benchmarks/agent/test_adapters.py +++ b/benchmarks/agent/test_adapters.py @@ -12,6 +12,8 @@ import codex import agy import direct +import opencode +import grok import os import tempfile import threading @@ -51,6 +53,7 @@ def test_recovered_provider_error_does_not_fail_completed_task(self): patch.object(pi.Path, "is_file", return_value=False), patch.object(pi.Path, "write_text"), patch.object(pi, "context_limit", return_value=None), + patch.object(pi, "cli_version", return_value="pi 0"), patch.object(pi.subprocess, "Popen", return_value=proc), patch("builtins.open", side_effect=lambda *a, **kw: io.StringIO()), ): @@ -117,6 +120,23 @@ def test_codex_builtin_skills_are_separate_from_observed_project_skills(self): self.assertEqual(codex.loaded_skills(text), (["wright"], ["openai-docs"])) +class GrokAdapterTest(unittest.TestCase): + def test_usage_row_buckets_are_disjoint_and_carry_the_context_limit(self): + row = grok.usage_row({"input_tokens": 12908, "output_tokens": 17, "cache_read_input_tokens": 1536, "cache_creation_input_tokens": 0}, 256000, 1.0) + self.assertEqual((row["input"], row["cache_read"], row["output"], row["context"], row["context_limit"]), (12908, 1536, 17, 14444, 256000)) + + +class OpencodeAdapterTest(unittest.TestCase): + def test_usage_row_keeps_cache_apart_from_input(self): + row = opencode.usage_row({"total": 6679, "input": 1042, "output": 5, "reasoning": 0, "cache": {"write": 0, "read": 5632}}, 1.0) + self.assertEqual((row["input"], row["cache_read"], row["output"], row["context"]), (1042, 5632, 5, 6674)) + + def test_builtin_skills_are_not_loaded_context(self): + raw = json.dumps([{"name": "customize-opencode", "location": ""}, {"name": "wright", "location": "/w/.agents/skills/wright/SKILL.md"}]) + self.assertEqual(opencode.available_skills(raw), ["wright"]) + self.assertEqual(opencode.available_skills("not json"), []) + + class DirectAdapterTest(unittest.TestCase): def serve(self, replies): seen = [] diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 5bd0d966..ec55b76d 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -395,6 +395,15 @@ def test_headroom_and_invalid_runs_are_reported(self): self.assertIn("HEADROOM", text) self.assertIn("INVALID: 1 run(s) excluded", text) + def test_report_shows_what_the_agent_was_actually_given(self): + run = self.result("wright+wright-skill/none/off", 1, True, 100) + run["agentInfo"] = {"version": "tool 1.2", "model": "m-1", "effort": "high", "tools": ["bash", "read"]} + run["context"] = {"loaded": ["wright"]} + bare = self.result("none/none/off", 1, True, 100) + text, _ = bench_report.render([run, bare]) + self.assertIn("| m | wright+wright-skill/none/off | tool 1.2 | m-1 | high | 2: bash, read | wright |", text) + self.assertIn("| m | none/none/off | not recorded | not recorded | not recorded | not recorded | none |", text) + def test_provider_failures_do_not_count_as_agent_failures(self): good = self.result("none/none/off", 1, True, 100) provider_failure = self.result("none/none/off", 2, False, 0) diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index 38df5240..a2f1bdb9 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -10,6 +10,12 @@ complete realistic Workshop work correctly? It measures Wright's discoverable semantic surface; it is not a model leaderboard, and one stochastic run is not evidence of correctness (use `--trials`). +A score is a credible reference, not a measure of an agent's ability. It holds for one +Wright, skill set, suite, agent, model, and protocol, and it is meant to show what +the prompt, skills, and tools did in that setup. Every report therefore lists, from +what the adapter observed, the CLI version, model, tools, and loaded skills each +agent actually had (`Agent setup`). + ## Allowed agent context - The scenario workspace: the seed project only. @@ -120,6 +126,9 @@ keep host configuration out of the run, and it reports what loaded through | [`devin.py`](../benchmarks/agent/adapters/devin.py) | Devin CLI | `BENCH_MODEL` (for example `swe-2-max`). Runs with an isolated `HOME` holding only the Devin credentials, a config that reads no other tool's rules or skills, and MCP tools denied. Managed plugin skills are listed apart from `loaded`. | | [`codex.py`](../benchmarks/agent/adapters/codex.py) | Codex CLI | `BENCH_MODEL` and `BENCH_THINKING`. Isolated `HOME`/`CODEX_HOME`, user config and exec rules ignored, workspace skills installed under `.agents/skills`. Session token events supply per-model-call usage and observed skill context; built-ins are listed apart. Web search, Apps, and plugin discovery are disabled; any observed MCP call invalidates the trial. | | [`agy.py`](../benchmarks/agent/adapters/agy.py) | Antigravity CLI | `BENCH_MODEL` (for example `gemini-3.8-flash-high`) and `BENCH_THINKING`. Isolated `HOME` with only authentication files, workspace skills under `.agents/skills`, MCP/browser access denied and URL reads denied outside `web`. Observed built-in web tools under network `off` stop and invalidate the trial; URL permissions do not cover search. Streaming step usage is recorded. The CLI does not export observed loaded skills or context limits; these remain unreported. | +| [`opencode.py`](../benchmarks/agent/adapters/opencode.py) | opencode | `BENCH_MODEL=provider/model` (as `opencode models` lists it) and `BENCH_THINKING` as the model variant. Isolated `HOME` and XDG directories holding only the credentials; `--pure`; Claude Code instructions and the real home's external skills are disabled by environment variable because opencode reads them regardless of `HOME`; skills installed under `.opencode/skills`; web tools denied outside `web`. Per-step usage comes from `step_finish` events; the tool list is what the agent used. | +| [`grok.py`](../benchmarks/agent/adapters/grok.py) | Grok CLI | `BENCH_MODEL` (a `grok models` id) and `BENCH_THINKING` as reasoning effort. Isolated `GROK_HOME` holding only the login and the skills; prompt sent `--verbatim`; subagents disabled; web search disabled outside `web`. Tools, skills, and the context window come from the stream's init and result lines. | +| [`direct.py`](../benchmarks/agent/adapters/direct.py) | none (built-in loop) | See below. | The pi adapter also uses an isolated `HOME` containing only its authentication files. Explicit provider extensions remain referenced by path, not copied with @@ -300,7 +309,9 @@ adds the baseline, `wright` without the skill, and, for OverPy scenarios, the `overpy` controls (cells whose skill has no `--skill-dir` are skipped). It runs the same from a terminal or from inside another agent's shell, because isolation comes from the harness's scrubbed environment, not from its parent. It runs locally; CI -does not run it. +does not run it. Adapters for CLIs that keep credentials in the home directory (devin, +codex, opencode, grok, agy) need `--env-pass HOME` so they can copy them into their +isolated home. `--adapter direct` is the built-in loop (`adapters/direct.py`) that needs no agent harness: it calls a model API with one `bash` tool (and `fetch` only for From ef1ec04789da74d8a41a21d7c28e7a5af82de5a8 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Fri, 2 Oct 2026 02:34:04 +0800 Subject: [PATCH 03/32] fix(bench): classify a Devin catalog outage as a provider failure, back off retries, and grade a missing entry as an agent failure --- benchmarks/agent/adapters/devin.py | 8 +++++++- benchmarks/agent/agent_bench.py | 2 ++ benchmarks/agent/bench_grade.py | 10 +++++++++- benchmarks/agent/test_adapters.py | 7 +++++++ benchmarks/agent/test_agent_bench.py | 10 ++++++++++ 5 files changed, 35 insertions(+), 2 deletions(-) diff --git a/benchmarks/agent/adapters/devin.py b/benchmarks/agent/adapters/devin.py index cefd0fe3..9e2a9c1a 100755 --- a/benchmarks/agent/adapters/devin.py +++ b/benchmarks/agent/adapters/devin.py @@ -68,6 +68,12 @@ def parse_export(export: dict) -> tuple[list[dict], list[str], list[str]]: return rows, loaded, plugins +def transient(text: str) -> bool: + """A provider or infrastructure failure. An empty model catalog means the catalog could not be fetched, not that the model is wrong.""" + lowered = text.lower() + return any(s in lowered for s in TRANSIENT) or re.search(r"unknown model.*\navailable:\s*$", lowered.strip(), re.S) is not None + + def main() -> int: env = os.environ model = env.get("BENCH_MODEL") or sys.exit("BENCH_MODEL is required: set it in --agent-cmd, for example BENCH_MODEL=swe-2-max python3 adapters/devin.py") @@ -97,7 +103,7 @@ def main() -> int: sys.stdout.write(proc.stdout) sys.stderr.write(proc.stderr) if proc.returncode != 0: - return INFRA_EXIT if any(s in (proc.stdout + proc.stderr).lower() for s in TRANSIENT) else proc.returncode + return INFRA_EXIT if transient(proc.stdout + proc.stderr) else proc.returncode return 0 diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index ed2240c7..3c79ae17 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -282,6 +282,7 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - snaps = snapshots.finish() if agent_exit == INFRA_EXIT and infra_retries < args.infra_retries: infra_retries += 1 + time.sleep(getattr(args, "infra_backoff", 0) * infra_retries) # an outage lasts longer than an immediate retry continue break (out / "agent.log").write_text(f"exit={agent_exit}\n--- stdout ---\n{stdout}\n--- stderr ---\n{stderr}\n") @@ -526,6 +527,7 @@ def main() -> int: p.add_argument("--timeout", type=int, default=1800) p.add_argument("--file-sandbox", action="store_true", help="macOS: restrict agent and descendant file writes to the trial directory") p.add_argument("--infra-retries", type=int, default=2, help="retries when the agent exits 75 (provider or infrastructure failure)") + p.add_argument("--infra-backoff", type=int, default=60, help="seconds before the first retry; each further retry waits one more multiple") run = sub.choices["run"] run.add_argument("scenario", choices=all_scenario_ids()) run.add_argument("--agent-cmd", required=True, help="shell command; the task prompt arrives on stdin, cwd is the workspace, BENCH_* describes the condition") diff --git a/benchmarks/agent/bench_grade.py b/benchmarks/agent/bench_grade.py index 7b1ef8f4..e0b6c0b4 100644 --- a/benchmarks/agent/bench_grade.py +++ b/benchmarks/agent/bench_grade.py @@ -85,6 +85,14 @@ def authorities(wright: str, source: Path, scratch: Path) -> dict: return result +def missing_entry_authorities(entry: Path) -> dict: + """The agent produced no entry file: every authority rejects it, and the oracle is only unavailable when it is not installed.""" + result: dict = {"wrightCompile": {"status": "error", "error": "entry file missing"}} + if entry.suffix == ".opy" and oracle_available(): + result["oracle"] = {"status": "error", "error": "entry file missing"} + return result + + def compiled_text(state: dict, source: str) -> str | None: """Compiled Workshop text from `wright` or the upstream `oracle`, when that authority succeeded.""" auth = state["authorities"] @@ -206,7 +214,7 @@ def grade(scenario: dict, workspace: Path, wright: str, scratch: Path | None = N entry = workspace / scenario["entry"] scratch = scratch or workspace.parent / f"{workspace.name}-grading" shutil.rmtree(scratch, ignore_errors=True) - state: dict = {"scratch": scratch, "authorities": authorities(wright, entry, scratch) if entry.is_file() else {"wrightCompile": {"status": "error"}}} + state: dict = {"scratch": scratch, "authorities": authorities(wright, entry, scratch) if entry.is_file() else missing_entry_authorities(entry)} _, lint = wright_json(wright, ["lint", str(entry)]) state["lint"] = (lint.get("result") or {}).get("findings") or [] checks = [run_check(c, workspace, entry, wright, state) for c in scenario["checks"]] diff --git a/benchmarks/agent/test_adapters.py b/benchmarks/agent/test_adapters.py index e174d271..33c2e233 100644 --- a/benchmarks/agent/test_adapters.py +++ b/benchmarks/agent/test_adapters.py @@ -102,6 +102,13 @@ def test_config_reads_no_other_tools_and_denies_web_unless_asked(self): self.assertNotIn("allow", closed["permissions"]) +class DevinTransientTest(unittest.TestCase): + def test_empty_model_catalog_is_a_provider_failure_but_a_wrong_model_is_not(self): + self.assertTrue(devin.transient("Error: Unknown model: 'swe-2-max'\nAvailable:\n")) + self.assertFalse(devin.transient("Error: Unknown model: 'nope'\nAvailable:\n swe-2-max\n swe-2\n")) + self.assertTrue(devin.transient("429 rate limit")) + + class NativeAdapterUsageTest(unittest.TestCase): def test_codex_inclusive_counts_are_split_without_counting_reasoning_twice(self): row = codex.usage_row({"input_tokens": 100, "cached_input_tokens": 60, "output_tokens": 20, "reasoning_output_tokens": 12}, 1.0, 272000) diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index ec55b76d..ec8d45c0 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -307,6 +307,16 @@ def test_oracle_check_never_passes_silently(self): self.assertFalse(graded["checks"][0]["passed"]) # a raw Workshop entry has no upstream OverPy verdict self.assertRegex(graded["grader"]["hash"], r"^[0-9a-f]{64}$") + @unittest.skipUnless(bench_grade.oracle_available(), "run `agent_bench.py setup-oracle`") + def test_missing_entry_is_an_agent_failure_not_an_unavailable_grader(self): + workspace = self.out / "empty" + workspace.mkdir() + scenario = agent_bench.load_scenario("widow-headshots") + graded = bench_grade.grade(scenario, workspace, WRIGHT) + self.assertEqual(graded["authorities"]["oracle"]["status"], "error") + self.assertNotIn("grader-unavailable", graded["usableReason"]) + self.assertIn("checks-failed", graded["usableReason"]) + @unittest.skipUnless(bench_grade.oracle_available(), "run `agent_bench.py setup-oracle`") def test_oracle_disagreement_is_reported(self): source = self.out / "n.opy" From 1b579b72f7407f67ef6aa27b09a1f59408ee0c48 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Fri, 2 Oct 2026 02:51:40 +0800 Subject: [PATCH 04/32] feat(bench): hide answer keys and host checkouts from the agent with the file sandbox --- benchmarks/agent/agent_bench.py | 29 ++++++++++++++++++++++++++-- benchmarks/agent/test_agent_bench.py | 18 ++++++++++++++++- docs/agent-benchmark.md | 11 +++++++++-- 3 files changed, 53 insertions(+), 5 deletions(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 3c79ae17..b1c79b29 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -207,6 +207,21 @@ def canaries(cell: dict, env: dict, workspace: Path, args: argparse.Namespace) - return None +def read_denials(args: argparse.Namespace, env: dict) -> tuple[list[Path], list[Path]]: + """Paths the agent must not read, and the paths inside them it still needs. + + Agents read the whole machine: the scenario answer keys, other trials' workspaces, the wiki, skills outside the condition, and + sibling checkouts of the repositories under test. Without this the benchmark measures what the agent could find, not what it was given.""" + run_dir = Path(env["BENCH_RUN_DIR"]).resolve() + selected = [name for name in env["BENCH_SKILLS"].split(",") if name] + denied = [args.out.resolve(), SCENARIOS, *(Path(p).expanduser().resolve() for p in getattr(args, "deny_read", []) or [])] + if getattr(args, "wiki_dir", None): + denied.append(Path(args.wiki_dir).resolve()) + denied += [Path(d).resolve() for name, d in (args.skill_dirs or {}).items() if name not in selected] + allowed = [run_dir, HERE, Path(args.wright).resolve().parent, *(Path(args.skill_dirs[name]).resolve() for name in selected)] + return denied, allowed + + def run_agent(args: argparse.Namespace, env: dict, workspace: Path, prompt: str) -> tuple[int | None, str, str]: command = args.agent_cmd if getattr(args, "file_sandbox", False): @@ -216,11 +231,15 @@ def run_agent(args: argparse.Namespace, env: dict, workspace: Path, prompt: str) temporary = run_dir / "tmp" temporary.mkdir(exist_ok=True) env = {**env, "TMPDIR": str(temporary), "PYTHONDONTWRITEBYTECODE": "1"} + denied, allowed = read_denials(args, env) + rules = "".join(f"(deny file-read-data (subpath {json.dumps(str(p))}))\n" for p in denied) + rules += "".join(f"(allow file-read-data (subpath {json.dumps(str(p))}))\n" for p in allowed) + rules += f"(deny file-read-data (subpath {json.dumps(str(SCENARIOS))}))\n" # the answer keys stay hidden even though the harness directory is readable profile = run_dir / "agent.sb" profile.write_text('(version 1)\n(allow default)\n(deny file-write*)\n' f'(allow file-write* (subpath {json.dumps(str(run_dir.resolve()))}) (subpath "/dev"))\n' '(deny file-read-data (require-all (regex "/(AGENTS|CLAUDE|GEMINI)[.]md$") ' - f'(require-not (subpath {json.dumps(str(workspace.resolve()))}))))\n') + f'(require-not (subpath {json.dumps(str(workspace.resolve()))}))))\n' + rules) command = ["sandbox-exec", "-f", str(profile), "/bin/sh", "-c", args.agent_cmd] proc = subprocess.Popen( command, shell=isinstance(command, str), cwd=workspace, env=env, text=True, @@ -290,6 +309,7 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - result["infraRetries"] = infra_retries result["networkEnforcement"] = "canary-checked" if cell["network"] == "off" and args.canary_cmd else "declared-only" result["fileWriteEnforcement"] = "trial-directory-only" if getattr(args, "file_sandbox", False) else "unrestricted" + result["fileReadEnforcement"] = [str(p) for p in read_denials(args, env)[0]] if getattr(args, "file_sandbox", False) else "unrestricted" context = context_report(out, [skill_name(Path(args.skill_dirs[name])) for name in cell["skills"]]) result["context"] = context if context.get("unexpected"): @@ -461,6 +481,7 @@ def cmd_evaluate(args: argparse.Namespace) -> int: Needs no agent harness around it: isolation comes from the harness's own scrubbed environment, so it runs the same from a terminal or from inside another agent's shell.""" + args.file_sandbox = not args.no_file_sandbox # evaluation hides the answer keys and the host's checkouts from the agent by default script = Path(__file__).parent / "adapters" / ADAPTERS[args.adapter] effort = f"BENCH_THINKING={shlex.quote(args.effort)} " if args.effort else "" agent_id = "-".join(filter(None, (args.adapter, args.model.replace("/", "_"), args.effort))) @@ -525,7 +546,8 @@ def main() -> int: p.add_argument("--no-ancestor-check", dest="check_ancestors", action="store_false", help="skip the check for instruction files above the workspace") p.add_argument("--canary-cmd", help="shell command that must fail in the agent environment when the network is 'off'") p.add_argument("--timeout", type=int, default=1800) - p.add_argument("--file-sandbox", action="store_true", help="macOS: restrict agent and descendant file writes to the trial directory") + p.add_argument("--file-sandbox", action="store_true", help="macOS: restrict agent and descendant file writes to the trial directory, and hide the scenarios, other runs, the wiki, unselected skills, and --deny-read paths") + p.add_argument("--deny-read", nargs="*", default=[], metavar="PATH", help="directories the agent must not read, such as the checkouts of the repositories under test; needs --file-sandbox") p.add_argument("--infra-retries", type=int, default=2, help="retries when the agent exits 75 (provider or infrastructure failure)") p.add_argument("--infra-backoff", type=int, default=60, help="seconds before the first retry; each further retry waits one more multiple") run = sub.choices["run"] @@ -549,6 +571,7 @@ def main() -> int: ev.add_argument("--trials", type=int, default=3) ev.add_argument("--parallel", type=int, default=2) ev.add_argument("--seed", type=int, default=1) + ev.add_argument("--no-file-sandbox", action="store_true", help="run without the macOS file sandbox: the agent can then read the scenario answer keys") sub.add_parser("setup-oracle", help="install the pinned upstream OverPy oracle") skill = sub.add_parser("wiki-skill", help="build the progressive-disclosure workshop-wiki skill from a wiki snapshot") skill.add_argument("--snapshot", type=Path, required=True) @@ -579,6 +602,8 @@ def main() -> int: if name not in SKILLS or not directory: raise SystemExit(f"--skill-dir expects NAME=DIR with NAME one of {', '.join(SKILLS)}: {item}") args.skill_dirs[name] = Path(directory) + if getattr(args, "deny_read", None) and not getattr(args, "file_sandbox", False) and args.command != "evaluate": + raise SystemExit("--deny-read needs --file-sandbox") if args.command == "validate": return 0 if validate(args.wright, args.out) else 1 if args.command == "setup-oracle": diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index ec8d45c0..88deef68 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -32,7 +32,7 @@ def setUp(self): def trial(self, agent_cmd: str, tool: str = "wright", skills: tuple = (), knowledge: str = "none", scenario: str = SCENARIO, **options) -> dict: args = argparse.Namespace(**{ "wright": str(Path(WRIGHT).resolve()), "agent_id": "fake", "agent_cmd": agent_cmd, "timeout": 60, "infra_retries": 2, - "env_pass": [], "canary_cmd": None, "skill_dirs": {}, "wiki_dir": None, "check_ancestors": False, **options, + "env_pass": [], "canary_cmd": None, "skill_dirs": {}, "wiki_dir": None, "check_ancestors": False, "out": self.out, **options, }) cell = agent_bench.normalize_cell({"tool": tool, "skills": list(skills), "knowledge": knowledge, "network": "off"}) return agent_bench.run_trial(agent_bench.load_scenario(scenario), cell, args, self.out / f"{scenario}-{tool}") @@ -189,6 +189,22 @@ def test_host_instructions_are_unreadable_but_workspace_instructions_are_allowed self.assertEqual(host.read_text(), "host-only instructions") self.assertEqual((self.out / f"{SCENARIO}-wright/workspace/AGENTS.md").read_text(), "workspace instructions") + @unittest.skipUnless(sys.platform == "darwin" and shutil.which("sandbox-exec"), "macOS file sandbox") + def test_agent_cannot_read_answer_keys_other_runs_or_denied_paths(self): + import shlex + denied = self.out / "checkout" + denied.mkdir() + (denied / "secret.txt").write_text("sibling repository") + answer = agent_bench.SCENARIOS / SCENARIO / "reference" / "mode.ws" + probes = {"answer": answer, "denied": denied / "secret.txt"} + code = ("from pathlib import Path\nimport json\nout = {}\n" + + "".join(f"try:\n Path({str(p)!r}).read_text(); out[{k!r}] = 'read'\nexcept PermissionError:\n out[{k!r}] = 'blocked'\n" for k, p in probes.items()) + + "Path('probe.json').write_text(json.dumps(out))\nPath('own.txt').write_text('ok'); assert Path('own.txt').read_text() == 'ok'\n") + result = self.trial(f'{shlex.quote(sys.executable)} -c {shlex.quote(code)}', file_sandbox=True, deny_read=[str(denied)]) + workspace = self.out / f"{SCENARIO}-wright/workspace" + self.assertEqual(json.loads((workspace / "probe.json").read_text()), {"answer": "blocked", "denied": "blocked"}) + self.assertIn(str(denied.resolve()), result["fileReadEnforcement"]) + def test_tools_differ_only_in_availability(self): agent = f"cp {reference()}/* . && (wright check mode.ws >/dev/null 2>&1 || echo no-wright > missing-wright.txt)" none = self.trial(agent, tool="none") diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index a2f1bdb9..75700669 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -155,8 +155,15 @@ processes: filesystem writes are restricted to that trial's output directory (plus device streams), and `TMPDIR` points inside it. Unsupported hosts fail instead of silently running without protection. This protects host files; it blocks reads of `AGENTS.md`, `CLAUDE.md`, and `GEMINI.md` outside the trial workspace -to prevent tool-path rule discovery from contaminating context. Other host reads -remain possible, and network isolation is not enforced. Model +to prevent tool-path rule discovery from contaminating context. It also hides from +the agent the scenarios (their reference solutions), every other run under `--out`, +the wiki snapshot, skills outside the condition, and any `--deny-read PATH`, such as +the checkouts of the repositories under test. The adapter, harness code, the `wright` +binary directory, and the condition's skills stay readable. The result lists the +denied paths in `fileReadEnforcement`. Without this, agents find the answer keys and the +owner repositories on the host (seen in practice), so `evaluate` turns the sandbox on by +default (`--no-file-sandbox` disables it). Other host reads remain possible, and +network isolation is not enforced. Model account usage, CPU and disk consumption remain shared with the host. Provider failures returned as exit 75 are listed separately and excluded from outcome metrics. The harness never edits the task prompt: network `off` and the workspace From 605594e3d3e45a432eccb0721ecc6a99bad5ef5b Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Fri, 2 Oct 2026 02:57:07 +0800 Subject: [PATCH 05/32] feat(bench): make evaluate runnable by hand with user defaults, a preflight, and a dry run --- benchmarks/agent/agent_bench.py | 54 ++++++++++++++++++++++++++++ benchmarks/agent/test_agent_bench.py | 18 ++++++++++ docs/agent-benchmark.md | 19 ++++++++++ 3 files changed, 91 insertions(+) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index b1c79b29..6c55d4dc 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -476,6 +476,44 @@ def work(job: tuple) -> None: ] +ADAPTER_BINARY = {"claude-code": "claude", "pi": "pi", "devin": "devin", "codex": "codex", "agy": "agy", "opencode": "opencode", "grok": "grok"} +DIRECT_ENV = ("ANTHROPIC_API_KEY", "ANTHROPIC_BASE_URL", "OPENAI_API_KEY", "OPENAI_BASE_URL") +CONFIG_PATH = Path(os.environ.get("WRIGHT_BENCH_CONFIG", Path.home() / ".config/wright-agent-bench/config.json")) + + +def user_defaults() -> dict: + """Per-user defaults for the long options, so a run is `evaluate --adapter A --model M`: keys wright, out, wiki_dir, env_pass, deny_read, skill_dirs {name: dir}.""" + if not CONFIG_PATH.is_file(): + return {} + config = json.loads(CONFIG_PATH.read_text()) + unknown = set(config) - {"wright", "out", "wiki_dir", "env_pass", "deny_read", "skill_dirs"} + if unknown: + raise SystemExit(f"{CONFIG_PATH}: unknown key(s) {sorted(unknown)}") + defaults = {k: v for k, v in config.items() if k not in ("skill_dirs", "out", "wiki_dir")} + defaults.update({k: Path(v).expanduser() for k, v in config.items() if k in ("out", "wiki_dir")}) + if "skill_dirs" in config: + defaults["skill_dir"] = [f"{name}={Path(d).expanduser()}" for name, d in config["skill_dirs"].items()] + return defaults + + +def preflight(args: argparse.Namespace, cells: list[dict]) -> list[str]: + """Problems that would waste a run, found before it starts.""" + problems = [] + if not Path(args.wright).is_file(): + problems.append(f"wright binary not found: {args.wright} (pass --wright or put `wright` on PATH)") + binary = ADAPTER_BINARY.get(args.adapter) + if binary and not shutil.which(binary): + problems.append(f"`{binary}` is not on PATH, which adapter '{args.adapter}' needs") + if args.adapter == "direct" and not any(os.environ.get(k) for k in ("ANTHROPIC_API_KEY", "OPENAI_API_KEY")): + problems.append("adapter 'direct' needs ANTHROPIC_API_KEY or OPENAI_API_KEY in the environment") + for name in sorted({s for c in cells for s in c["skills"]}): + if not (Path(args.skill_dirs.get(name, "/nonexistent")) / "SKILL.md").is_file(): + problems.append(f"skill '{name}' needs --skill-dir {name}=DIR (or skill_dirs in {CONFIG_PATH})") + if any(c["tool"] == "overpy" for c in cells) and not bench_grade.oracle_available(): + problems.append("tool 'overpy' needs the pinned oracle: run `agent_bench.py setup-oracle`") + return problems + + def cmd_evaluate(args: argparse.Namespace) -> int: """One command from agent and model to data and document: run the matrix, then write report, score cards, and RESULTS.md. @@ -493,7 +531,17 @@ def cmd_evaluate(args: argparse.Namespace) -> int: print(f"skipped cells without --skill-dir: {', '.join(dropped)}", flush=True) if not cells: raise SystemExit("no cell can run: pass --skill-dir wright-skill=DIR") + args.env_pass = sorted({*args.env_pass, *(DIRECT_ENV if args.adapter == "direct" else ("HOME",))}) # the credentials the adapter copies or reads + problems = preflight(args, [normalize_cell(c) for c in cells]) + if problems: + raise SystemExit("cannot start:\n " + "\n ".join(problems)) + if args.file_sandbox and not args.deny_read: + print("note: no --deny-read, so the agent can still read the host's repository checkouts (the scenarios are always hidden)", flush=True) scenarios = args.scenarios or [s for s in all_scenario_ids() if args.split == "all" or load_scenario(s).get("split") == args.split] + runnable = sum(args.trials for s in scenarios for c in cells if applicable(load_scenario(s), normalize_cell(c))) + print(f"{args.adapter} {args.model}: {len(scenarios)} scenario(s), cells {', '.join(cell_label(normalize_cell(c)) for c in cells)}, {runnable} trial(s) into {args.out / args.name}", flush=True) + if args.dry_run: + return 0 args.out = args.out / args.name args.out.mkdir(parents=True, exist_ok=True) config = {"agents": [{"id": agent_id, "cmd": cmd}], "cells": cells, "scenarios": scenarios, "trials": args.trials, "parallel": args.parallel, "seed": args.seed, @@ -571,6 +619,7 @@ def main() -> int: ev.add_argument("--trials", type=int, default=3) ev.add_argument("--parallel", type=int, default=2) ev.add_argument("--seed", type=int, default=1) + ev.add_argument("--dry-run", action="store_true", help="check the setup and print what would run, without running it") ev.add_argument("--no-file-sandbox", action="store_true", help="run without the macOS file sandbox: the agent can then read the scenario answer keys") sub.add_parser("setup-oracle", help="install the pinned upstream OverPy oracle") skill = sub.add_parser("wiki-skill", help="build the progressive-disclosure workshop-wiki skill from a wiki snapshot") @@ -590,6 +639,11 @@ def main() -> int: score = sub.add_parser("score", help="compute the Wright Agent Score card of each language track from canonical test runs") score.add_argument("dirs", nargs="+", type=Path) score.add_argument("--language", choices=("workshop", "opy"), action="append", help="track to score; both when omitted") + defaults = user_defaults() + for choice in sub.choices.values(): + known = {a.dest for a in choice._actions} + choice.set_defaults(**{k: v for k, v in defaults.items() if k in known}) + sub.choices["evaluate"].set_defaults(wright=defaults.get("wright") or shutil.which("wright") or str(ROOT / "target/debug/wright")) args = parser.parse_args() if hasattr(args, "wright"): args.wright = str(Path(args.wright).resolve()) diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 88deef68..2bd8db46 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -6,6 +6,7 @@ import sys import tempfile import unittest +from unittest.mock import patch from pathlib import Path import agent_bench @@ -205,6 +206,23 @@ def test_agent_cannot_read_answer_keys_other_runs_or_denied_paths(self): self.assertEqual(json.loads((workspace / "probe.json").read_text()), {"answer": "blocked", "denied": "blocked"}) self.assertIn(str(denied.resolve()), result["fileReadEnforcement"]) + def test_user_defaults_are_read_from_the_config_file(self): + config = self.out / "config.json" + config.write_text(json.dumps({"skill_dirs": {"wright-skill": "/s/wright"}, "deny_read": ["~/Repos"], "wright": "/bin/wright"})) + with patch.object(agent_bench, "CONFIG_PATH", config): + defaults = agent_bench.user_defaults() + self.assertEqual((defaults["skill_dir"], defaults["deny_read"], defaults["wright"]), (["wright-skill=/s/wright"], ["~/Repos"], "/bin/wright")) + config.write_text(json.dumps({"skil_dirs": {}})) + with patch.object(agent_bench, "CONFIG_PATH", config), self.assertRaises(SystemExit): + agent_bench.user_defaults() + + def test_preflight_names_what_is_missing_before_a_run_starts(self): + args = argparse.Namespace(wright=str(self.out / "nope"), adapter="devin", skill_dirs={}) + with patch.object(agent_bench.shutil, "which", return_value=None): + problems = agent_bench.preflight(args, [agent_bench.normalize_cell({"tool": "wright", "skills": ["wright-skill"], "knowledge": "none", "network": "off"})]) + self.assertEqual(len(problems), 3) + self.assertTrue(any("wright binary not found" in p for p in problems) and any("`devin` is not on PATH" in p for p in problems) and any("--skill-dir wright-skill" in p for p in problems)) + def test_tools_differ_only_in_availability(self): agent = f"cp {reference()}/* . && (wright check mode.ws >/dev/null 2>&1 || echo no-wright > missing-wright.txt)" none = self.trial(agent, tool="none") diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index 75700669..9051b0c3 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -308,6 +308,25 @@ held-out scenarios, missing scenarios, or unequal trials. The card discloses ## Evaluating without an agent harness +Quick start. Put your paths in `~/.config/wright-agent-bench/config.json` once +(`WRIGHT_BENCH_CONFIG` overrides the location): + +```json +{"skill_dirs": {"wright-skill": "~/skills/skills/wright", "opy-skill": "~/skills/skills/overpy"}, + "deny_read": ["~/Repos", "~/.agents", "~/.claude"], + "wiki_dir": "~/.local/share/wright-agent-bench/wiki"} +``` + +then run one command per agent. `--dry-run` checks the setup (wright binary, the +agent CLI, credentials, skill directories, the oracle) and prints the plan without +running anything; `wright` is taken from `PATH` unless `--wright` or the config says +otherwise, and the credentials each adapter needs are passed through automatically. + +```sh +python3 benchmarks/agent/agent_bench.py evaluate --adapter devin --model swe-2-max --dry-run +python3 benchmarks/agent/agent_bench.py evaluate --adapter devin --model swe-2-max +``` + `agent_bench.py evaluate --adapter ADAPTER --model MODEL --skill-dir wright-skill=DIR` is the one-command entry: it writes `matrix.json`, runs the cells, then writes `report.md`, `summary.json`, `score.json`, `score.txt`, and `RESULTS.md` into From 35e30e6192c15c597bbc7c0467c0d0039c8befcf Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Fri, 2 Oct 2026 03:06:01 +0800 Subject: [PATCH 06/32] feat(bench): compare score cards from separate evaluation runs --- benchmarks/agent/agent_bench.py | 5 +++++ benchmarks/agent/bench_score.py | 24 ++++++++++++++++++++++++ benchmarks/agent/test_score.py | 16 ++++++++++++++++ docs/agent-benchmark.md | 17 +++++++++++++++++ 4 files changed, 62 insertions(+) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 6c55d4dc..c1ab2526 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -636,6 +636,8 @@ def main() -> int: report.add_argument("--regrade", action="store_true", help="re-grade stored workspaces twice and flag unstable graders") report.add_argument("--wright", default=str(ROOT / "target/debug/wright")) report.add_argument("--reference", default=bench_report.BASELINE, help="condition label the paired comparison is made against") + compare = sub.add_parser("compare", help="one table from the score.json of several evaluation runs, warning when they are not comparable") + compare.add_argument("dirs", nargs="+", type=Path) score = sub.add_parser("score", help="compute the Wright Agent Score card of each language track from canonical test runs") score.add_argument("dirs", nargs="+", type=Path) score.add_argument("--language", choices=("workshop", "opy"), action="append", help="track to score; both when omitted") @@ -668,6 +670,9 @@ def main() -> int: return cmd_wiki_skill(args) if args.command == "report": return bench_report.main(args.dirs, args.wright, args.regrade, lambda s: load_scenario(s), args.reference) + if args.command == "compare": + print(bench_score.compare(args.dirs)) + return 0 if args.command == "score": languages = args.language or ["workshop", "opy"] expected = {lang: [s for s in all_scenario_ids() if load_scenario(s)["language"] == lang and load_scenario(s).get("split") == "test"] for lang in languages} diff --git a/benchmarks/agent/bench_score.py b/benchmarks/agent/bench_score.py index 7e1e140b..1868519c 100644 --- a/benchmarks/agent/bench_score.py +++ b/benchmarks/agent/bench_score.py @@ -137,6 +137,30 @@ def render(c: dict) -> str: return "\n".join(lines) + "\n" +COMPARABLE = ("wrightSha256", "skills", "suite") # what must match for two cards to be read side by side; agent, model, and effort are what is being compared + + +def compare(dirs: list[Path]) -> str: + """One table from the score cards of several evaluation runs, with a warning when they were not made against the same Wright, skills, and suite.""" + rows, identities = [], {} + for directory in dirs: + for card_ in json.loads((directory / "score.json").read_text())["cards"]: + if "refused" in card_: + rows.append((card_["track"], directory.name, "no score", card_["refused"])) + continue + ident = card_["identity"] + identities.setdefault(card_["track"], []).append((directory.name, {k: ident[k] for k in COMPARABLE})) + label = " ".join(filter(None, (ident["agent"], ident["effort"] and f"effort {ident['effort']}"))) + note = "provisional: " + "; ".join(card_["provisional"]) if card_["provisional"] else "" + rows.append((card_["track"], directory.name, f"{card_['score']} [{card_['ci95'][0]}-{card_['ci95'][1]}]", f"{label}; {card_['trialsPerScenario']} trial(s) x {card_['scenarios']} scenarios; excluded {card_['exclusions'] or 'none'}. {note}".strip())) + lines = ["| track | run | score [95% CI] | agent and notes |", "| --- | --- | --- | --- |", *(f"| {t} | {d} | {s} | {n} |" for t, d, s, n in sorted(rows))] + for track, entries in identities.items(): + differing = sorted({k for _, a in entries for _, b in entries for k in COMPARABLE if a[k] != b[k]}) + if differing: + lines.append(f"\nWARNING {track}: runs differ in {', '.join(differing)}, so these scores are not directly comparable.") + return "\n".join(lines) + "\n" + + def main(dirs: list[Path], languages: list[str], expected_by_language: dict[str, list[str]], out: Path | None) -> int: from bench_report import load results = load(dirs) diff --git a/benchmarks/agent/test_score.py b/benchmarks/agent/test_score.py index cb9ec03c..1883710b 100644 --- a/benchmarks/agent/test_score.py +++ b/benchmarks/agent/test_score.py @@ -101,6 +101,22 @@ def test_main_writes_the_machine_readable_card_and_exit_status(self): self.assertEqual(card["cards"][0]["score"], 100.0) self.assertEqual(bench_score.main([root], ["workshop"], {"workshop": SCENARIOS}, None), 2) + def test_compare_tables_runs_and_warns_when_the_environment_differs(self): + root = Path(tempfile.mkdtemp(dir=Path(__file__).resolve().parents[2] / "target")) + self.addCleanup(shutil.rmtree, root, True) + for name, sha, model in (("codex-run", "a" * 64, "luna"), ("pi-run", "d" * 64, "luna2")): + directory = root / name + directory.mkdir() + card = bench_score.card(runs(SCENARIOS[:6], sha=sha, model=model), "opy", SCENARIOS) + (directory / "score.json").write_text(json.dumps({"contract": bench_score.CONTRACT, "cards": [card]})) + table = bench_score.compare([root / "codex-run", root / "pi-run"]) + self.assertIn("codex-run", table) + self.assertIn("pi-run", table) + self.assertIn("effort high", table) + self.assertIn("runs differ in wrightSha256", table) + same = bench_score.compare([root / "codex-run"]) + self.assertNotIn("WARNING", same) + if __name__ == "__main__": unittest.main() diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index 9051b0c3..ade78d66 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -350,6 +350,23 @@ recorded in `agentInfo.protocol`. It does not sandbox the network, so pair it wi `--canary-cmd`. Scores from `direct` and from product harnesses measure different things and are not mixed. +### Running different models at different times + +Each `evaluate` writes its own directory and scores only its own runs, so models +and agents can be run whenever quota allows, in any order. Put the runs side by side +with + +```sh +python3 benchmarks/agent/agent_bench.py compare ~/.cache/wright-agent-bench/{devin-swe2-stage1,codex-luna-xhigh,pi-luna-xhigh} +``` + +which prints one table of scores, intervals, trials, and exclusions, and warns when +the runs differ in the Wright binary, skill contents, or suite (scenarios, grader, +oracle lock). Scores are comparable when it prints no warning. Keep the Wright +binary, skills, and scenarios unchanged between runs; changes to the rest of the +harness are disclosed in each card's `Harness` line and do not block comparison. +An interrupted run resumes by repeating the same command with the same `--name`. + ## Cadence The benchmark does not gate pull requests. Run it manually or on a schedule once From bd7755a62ebe19b4e59b02503dc484ac0ed9f8fb Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Fri, 2 Oct 2026 13:11:02 +0800 Subject: [PATCH 07/32] fix(bench): hide withheld Devin tools from the agent and deny reading sibling runs and harness data --- benchmarks/agent/adapters/devin.py | 2 ++ benchmarks/agent/agent_bench.py | 6 +++++- benchmarks/agent/test_adapters.py | 3 +++ benchmarks/agent/test_agent_bench.py | 8 ++++++++ 4 files changed, 18 insertions(+), 1 deletion(-) diff --git a/benchmarks/agent/adapters/devin.py b/benchmarks/agent/adapters/devin.py index 9e2a9c1a..395b5ab6 100755 --- a/benchmarks/agent/adapters/devin.py +++ b/benchmarks/agent/adapters/devin.py @@ -35,6 +35,8 @@ def isolated_config(user_config: dict, model: str, web: bool) -> dict: config = copy.deepcopy(user_config) config["read_config_from"] = NO_TOOL_CONFIG config["permissions"] = {"deny": MCP_TOOLS + ([] if web else WEB_TOOLS)} + # A denied call ends a headless session, so the tools are also hidden from the agent: it never reaches for what the condition withholds. + config["disabled_tools"] = MCP_TOOLS + ([] if web else [t for t in WEB_TOOLS if t.islower()]) config.setdefault("agent", {})["model"] = model return config diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index c1ab2526..ee033b19 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -207,6 +207,9 @@ def canaries(cell: dict, env: dict, workspace: Path, args: argparse.Namespace) - return None +DATA_ROOTS = (Path.home() / ".cache/wright-agent-bench", Path.home() / ".local/share/wright-agent-bench") # other runs, wikis, and pinned skills live here by default + + def read_denials(args: argparse.Namespace, env: dict) -> tuple[list[Path], list[Path]]: """Paths the agent must not read, and the paths inside them it still needs. @@ -214,7 +217,7 @@ def read_denials(args: argparse.Namespace, env: dict) -> tuple[list[Path], list[ sibling checkouts of the repositories under test. Without this the benchmark measures what the agent could find, not what it was given.""" run_dir = Path(env["BENCH_RUN_DIR"]).resolve() selected = [name for name in env["BENCH_SKILLS"].split(",") if name] - denied = [args.out.resolve(), SCENARIOS, *(Path(p).expanduser().resolve() for p in getattr(args, "deny_read", []) or [])] + denied = [args.out.resolve(), getattr(args, "out_root", args.out).resolve(), *DATA_ROOTS, SCENARIOS, *(Path(p).expanduser().resolve() for p in getattr(args, "deny_read", []) or [])] if getattr(args, "wiki_dir", None): denied.append(Path(args.wiki_dir).resolve()) denied += [Path(d).resolve() for name, d in (args.skill_dirs or {}).items() if name not in selected] @@ -542,6 +545,7 @@ def cmd_evaluate(args: argparse.Namespace) -> int: print(f"{args.adapter} {args.model}: {len(scenarios)} scenario(s), cells {', '.join(cell_label(normalize_cell(c)) for c in cells)}, {runnable} trial(s) into {args.out / args.name}", flush=True) if args.dry_run: return 0 + args.out_root = args.out # sibling evaluation runs must stay unreadable too args.out = args.out / args.name args.out.mkdir(parents=True, exist_ok=True) config = {"agents": [{"id": agent_id, "cmd": cmd}], "cells": cells, "scenarios": scenarios, "trials": args.trials, "parallel": args.parallel, "seed": args.seed, diff --git a/benchmarks/agent/test_adapters.py b/benchmarks/agent/test_adapters.py index 33c2e233..25e5e3f8 100644 --- a/benchmarks/agent/test_adapters.py +++ b/benchmarks/agent/test_adapters.py @@ -97,6 +97,9 @@ def test_config_reads_no_other_tools_and_denies_web_unless_asked(self): self.assertFalse(any(closed["read_config_from"].values())) self.assertEqual(closed["agent"]["model"], "swe-2-max") self.assertIn("web_search", closed["permissions"]["deny"]) + self.assertIn("web_search", closed["disabled_tools"]) + self.assertIn("mcp_call_tool", closed["disabled_tools"]) + self.assertNotIn("web_search", opened["disabled_tools"]) self.assertNotIn("web_search", opened["permissions"]["deny"]) self.assertIn("mcp_call_tool", opened["permissions"]["deny"]) self.assertNotIn("allow", closed["permissions"]) diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 2bd8db46..ca3d07fd 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -223,6 +223,14 @@ def test_preflight_names_what_is_missing_before_a_run_starts(self): self.assertEqual(len(problems), 3) self.assertTrue(any("wright binary not found" in p for p in problems) and any("`devin` is not on PATH" in p for p in problems) and any("--skill-dir wright-skill" in p for p in problems)) + def test_sibling_runs_and_harness_data_roots_are_unreadable(self): + root = self.out + args = argparse.Namespace(out=root / "run-a", out_root=root, wright=WRIGHT, skill_dirs={}, wiki_dir=None, deny_read=[]) + denied, allowed = agent_bench.read_denials(args, {"BENCH_RUN_DIR": str(root / "run-a" / "t"), "BENCH_SKILLS": ""}) + self.assertIn(root.resolve(), denied) + self.assertTrue(all(d in denied for d in agent_bench.DATA_ROOTS)) + self.assertIn((root / "run-a" / "t").resolve(), allowed) + def test_tools_differ_only_in_availability(self): agent = f"cp {reference()}/* . && (wright check mode.ws >/dev/null 2>&1 || echo no-wright > missing-wright.txt)" none = self.trial(agent, tool="none") From b430e29f5abf9d8262b6e118383c55a5b177d5b7 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Fri, 2 Oct 2026 13:37:23 +0800 Subject: [PATCH 08/32] fix(bench): clear adapter homes between infrastructure retries --- benchmarks/agent/agent_bench.py | 3 ++- benchmarks/agent/test_agent_bench.py | 6 ++++++ 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index ee033b19..156e4846 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -283,7 +283,8 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - infra_retries = 0 while True: shutil.rmtree(workspace, ignore_errors=True) - for stale in ("home", "bin", "real", "tool-trace.jsonl", "tool-calls", "usage.jsonl", "transcript.jsonl", "context.json", "agent-info.json", "snapshots"): + for stale in ("home", "bin", "real", "tool-trace.jsonl", "tool-calls", "usage.jsonl", "transcript.jsonl", "context.json", "agent-info.json", "snapshots", "tmp", + *(child.name for child in out.iterdir() if child.name.endswith(("-home", "-user")))): # adapters keep their isolated homes here; a retry starts from none target = out / stale shutil.rmtree(target, ignore_errors=True) if target.is_dir() else target.unlink(missing_ok=True) materialize(scenario, workspace) diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index ca3d07fd..d25fd77f 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -313,6 +313,12 @@ def test_adapter_usage_reaches_result(self): self.assertEqual(result["usage"]["totalTokens"], 6) self.assertIn("unexpected loaded context", result["invalid"]) + def test_a_retry_starts_without_the_previous_attempts_adapter_home(self): + agent = ('if [ -f "$BENCH_RUN_DIR/tried" ]; then test ! -e "$BENCH_RUN_DIR/devin-home" || exit 1; exit 0; ' + 'else touch "$BENCH_RUN_DIR/tried"; mkdir -p "$BENCH_RUN_DIR/devin-home/.local"; exit 75; fi') + result = self.trial(agent) + self.assertEqual((result["agent"]["exit"], result["infraRetries"]), (0, 1)) + def test_infrastructure_failures_are_retried(self): agent = 'if [ -f "$BENCH_RUN_DIR/tried" ]; then exit 0; else touch "$BENCH_RUN_DIR/tried"; exit 75; fi' self.assertEqual(self.trial(agent)["infraRetries"], 1) From 2b014fc089fc0eb90fe7ca4ee7a150d28b71389a Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sat, 3 Oct 2026 01:39:58 +0800 Subject: [PATCH 09/32] feat(bench): allow-list file reads, record Devin effort from the model id, run sequentially by default, and write refreshed logins back --- benchmarks/agent/adapters/devin.py | 13 ++- benchmarks/agent/agent_bench.py | 129 ++++++++++++++++++++++----- benchmarks/agent/bench_trace.py | 4 + benchmarks/agent/test_adapters.py | 7 ++ benchmarks/agent/test_agent_bench.py | 63 +++++++++++-- 5 files changed, 185 insertions(+), 31 deletions(-) diff --git a/benchmarks/agent/adapters/devin.py b/benchmarks/agent/adapters/devin.py index 395b5ab6..c6d6644b 100755 --- a/benchmarks/agent/adapters/devin.py +++ b/benchmarks/agent/adapters/devin.py @@ -31,6 +31,17 @@ NO_TOOL_CONFIG = {"claude": False, "cursor": False, "windsurf": False, "codex": False} +EFFORTS = ("low", "medium", "high", "xhigh", "max") + + +def split_effort(model_id: str) -> tuple[str, str | None]: + """Devin names the effort in the model id (`swe-2-max` is model `swe-2` at effort `max`), so the CLI never reports it separately.""" + for effort in EFFORTS: + if model_id.endswith(f"-{effort}"): + return model_id[: -len(effort) - 1], effort + return model_id, None + + def isolated_config(user_config: dict, model: str, web: bool) -> dict: config = copy.deepcopy(user_config) config["read_config_from"] = NO_TOOL_CONFIG @@ -100,7 +111,7 @@ def main() -> int: Path(env["BENCH_USAGE"]).write_text("".join(json.dumps(r) + "\n" for r in rows)) Path(env["BENCH_TRANSCRIPT"]).write_text("".join(json.dumps(s) + "\n" for s in (json.loads(export.read_text())["steps"] if export.is_file() else []))) exported = json.loads(export.read_text()) if export.is_file() else {} - Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "devin", "version": cli_version(devin), "model": model, "tools": [t["function"]["name"] for t in exported.get("agent", {}).get("tool_definitions", [])], "denied": MCP_TOOLS + ([] if env["BENCH_KNOWLEDGE"] == "web" else WEB_TOOLS)}, indent=2)) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "devin", "version": cli_version(devin), "model": split_effort(model)[0], "effort": split_effort(model)[1], "modelId": model, "tools": [t["function"]["name"] for t in exported.get("agent", {}).get("tool_definitions", [])], "denied": MCP_TOOLS + ([] if env["BENCH_KNOWLEDGE"] == "web" else WEB_TOOLS)}, indent=2)) Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded, "ignoredPluginSkills": plugins})) sys.stdout.write(proc.stdout) sys.stderr.write(proc.stderr) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 156e4846..1b77c7be 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -169,7 +169,7 @@ def build_env(cell: dict, args: argparse.Namespace, out: Path, workspace: Path) shim_dir = out / "bin" shim_dir.mkdir() shim = shim_dir / cell["tool"] - shim.write_text(f'#!/bin/sh\nexec "{sys.executable}" "{Path(__file__).resolve()}" shim {cell["tool"]} "$@"\n') + shim.write_text(f'#!/bin/sh\nexec "{sys.executable}" "{(HERE / "bench_trace.py").resolve()}" shim {cell["tool"]} "$@"\n') shim.chmod(0o755) real = args.wright if cell["tool"] == "wright" else str(overpy_launcher(out, os.environ["PATH"])) env.update({f"BENCH_TOOL_REAL_{cell['tool'].upper()}": real, "BENCH_TOOL_TRACE": str(out / "tool-trace.jsonl"), "BENCH_TOOL_SIDECAR": str(out / "tool-calls")}) @@ -208,21 +208,101 @@ def canaries(cell: dict, env: dict, workspace: Path, args: argparse.Namespace) - DATA_ROOTS = (Path.home() / ".cache/wright-agent-bench", Path.home() / ".local/share/wright-agent-bench") # other runs, wikis, and pinned skills live here by default +HIDDEN_ROOTS = (Path("/Users"), Path("/Volumes"), Path.home()) # the host's home directories and external drives are hidden unless listed below +ADAPTER_READS = { # what each adapter reads from the real home before the agent starts: credentials, configuration, installation (relative to the home directory) + "devin": [".local/share/devin", ".config/devin"], "pi": [".pi"], "codex": [".codex/auth.json"], "agy": [".gemini/antigravity-cli"], + "opencode": [".local/share/opencode/auth.json"], "grok": [".grok/auth.json"], "claude-code": [".claude.json", ".claude/.credentials.json"], "direct": [], +} + + +CREDENTIALS = { # (real file relative to the home directory, its isolated copy relative to the run directory) per adapter + "codex": [(".codex/auth.json", "codex-home/.codex/auth.json")], + "pi": [(".pi/agent/auth.json", "pi-home/.pi/agent/auth.json"), (".pi/agent/antigravity-accounts.json", "pi-home/.pi/agent/antigravity-accounts.json")], + "devin": [(".local/share/devin/credentials.toml", "devin-home/.local/share/devin/credentials.toml")], + "agy": [(".gemini/antigravity-cli/antigravity-oauth-token", "agy-home/.gemini/antigravity-cli/antigravity-oauth-token")], + "opencode": [(".local/share/opencode/auth.json", "opencode-home/.local/share/opencode/auth.json")], + "grok": [(".grok/auth.json", "grok-home/auth.json")], +} + + +def sync_credentials_back(pairs: list[tuple[str, str]], run_dir: Path, home: Path | None = None) -> list[str]: + """Copy a login the agent's CLI refreshed inside its isolated home back to the real one. + + Adapters hand the CLI a copy of the user's OAuth login. Refresh tokens rotate, so a refresh in the copy that is then discarded + leaves the user's own login holding a dead token. Only a changed, newer copy is written back, atomically.""" + home = home or Path.home() + updated = [] + for real_rel, isolated_rel in pairs: + real, isolated = home / real_rel, run_dir / isolated_rel + if not isolated.is_file() or not real.is_file() or isolated.read_bytes() == real.read_bytes() or isolated.stat().st_mtime <= real.stat().st_mtime: + continue + temporary = real.with_name(f".{real.name}.bench-sync") + temporary.write_bytes(isolated.read_bytes()) + temporary.chmod(real.stat().st_mode & 0o777) + os.replace(temporary, real) + updated.append(real_rel) + return updated + + +CREDENTIALS = { # (real file relative to the home directory, its isolated copy relative to the run directory) per adapter + "codex": [(".codex/auth.json", "codex-home/.codex/auth.json")], + "pi": [(".pi/agent/auth.json", "pi-home/.pi/agent/auth.json"), (".pi/agent/antigravity-accounts.json", "pi-home/.pi/agent/antigravity-accounts.json")], + "devin": [(".local/share/devin/credentials.toml", "devin-home/.local/share/devin/credentials.toml")], + "agy": [(".gemini/antigravity-cli/antigravity-oauth-token", "agy-home/.gemini/antigravity-cli/antigravity-oauth-token")], + "opencode": [(".local/share/opencode/auth.json", "opencode-home/.local/share/opencode/auth.json")], + "grok": [(".grok/auth.json", "grok-home/auth.json")], +} + + +def sync_credentials_back(pairs: list[tuple[str, str]], run_dir: Path, home: Path | None = None) -> list[str]: + """Copy a login the agent's CLI refreshed inside its isolated home back to the real one. + + Adapters hand the CLI a copy of the user's OAuth login. Refresh tokens rotate, so a refresh in the copy that is then discarded + leaves the user's own login holding a dead token. Only a changed, newer copy is written back, atomically.""" + home = home or Path.home() + updated = [] + for real_rel, isolated_rel in pairs: + real, isolated = home / real_rel, run_dir / isolated_rel + if not isolated.is_file() or not real.is_file() or isolated.read_bytes() == real.read_bytes() or isolated.stat().st_mtime <= real.stat().st_mtime: + continue + temporary = real.with_name(f".{real.name}.bench-sync") + temporary.write_bytes(isolated.read_bytes()) + temporary.chmod(real.stat().st_mode & 0o777) + os.replace(temporary, real) + updated.append(real_rel) + return updated -def read_denials(args: argparse.Namespace, env: dict) -> tuple[list[Path], list[Path]]: - """Paths the agent must not read, and the paths inside them it still needs. +def read_policy(args: argparse.Namespace, env: dict) -> tuple[list[Path], list[Path]]: + """What the agent may not read, and the exceptions it needs. - Agents read the whole machine: the scenario answer keys, other trials' workspaces, the wiki, skills outside the condition, and - sibling checkouts of the repositories under test. Without this the benchmark measures what the agent could find, not what it was given.""" + An allow-list: the host's home directories and drives are hidden, so the agent sees the machine as a clean one holding only its run + directory, the condition's skills, the tool and runtime binaries, and what its adapter needs. Agents otherwise read the whole machine: + the answer keys, other runs, the grader, installed copies of the OverPy source, and checkouts of the repositories under test, so the + benchmark would measure what the agent could find instead of what it was given.""" run_dir = Path(env["BENCH_RUN_DIR"]).resolve() selected = [name for name in env["BENCH_SKILLS"].split(",") if name] - denied = [args.out.resolve(), getattr(args, "out_root", args.out).resolve(), *DATA_ROOTS, SCENARIOS, *(Path(p).expanduser().resolve() for p in getattr(args, "deny_read", []) or [])] + hidden = [*HIDDEN_ROOTS, args.out, getattr(args, "out_root", args.out), *DATA_ROOTS, HERE, *(Path(p).expanduser() for p in getattr(args, "deny_read", []) or [])] if getattr(args, "wiki_dir", None): - denied.append(Path(args.wiki_dir).resolve()) - denied += [Path(d).resolve() for name, d in (args.skill_dirs or {}).items() if name not in selected] - allowed = [run_dir, HERE, Path(args.wright).resolve().parent, *(Path(args.skill_dirs[name]).resolve() for name in selected)] - return denied, allowed + hidden.append(Path(args.wiki_dir)) + hidden += [Path(d) for name, d in (args.skill_dirs or {}).items() if name not in selected] + runtime = [Path(args.wright), Path(sys.executable)] + for binary in (shutil.which("node"), shutil.which(ADAPTER_BINARY.get(getattr(args, "adapter", ""), ""))): + if binary: + runtime.append(Path(binary)) + allowed = [run_dir, HERE / "adapters", HERE / "bench_trace.py", HERE / "oracle", Path(sys.prefix), Path(sys.base_prefix), + *(Path(args.skill_dirs[name]) for name in selected), *(Path(p).expanduser() for p in getattr(args, "allow_read", []) or [])] + for binary in runtime: # the directory of the binary and of the file its symlink resolves to + allowed += [binary.parent, binary.resolve().parent] + unique = lambda paths: list(dict.fromkeys(p.resolve() for p in paths)) + return unique(hidden), unique(allowed) + + +def sandbox_read_rules(hidden: list[Path], allowed: list[Path]) -> str: + def rule(action: str, path: Path) -> str: + return f"({action} file-read-data ({'literal' if path.is_file() else 'subpath'} {json.dumps(str(path))}))\n" + # the allow-list overrides the hidden roots, and the answer keys are hidden again whatever else is allowed + return "".join(rule("deny", p) for p in hidden) + "".join(rule("allow", p) for p in allowed) + rule("deny", SCENARIOS.resolve()) def run_agent(args: argparse.Namespace, env: dict, workspace: Path, prompt: str) -> tuple[int | None, str, str]: @@ -234,10 +314,7 @@ def run_agent(args: argparse.Namespace, env: dict, workspace: Path, prompt: str) temporary = run_dir / "tmp" temporary.mkdir(exist_ok=True) env = {**env, "TMPDIR": str(temporary), "PYTHONDONTWRITEBYTECODE": "1"} - denied, allowed = read_denials(args, env) - rules = "".join(f"(deny file-read-data (subpath {json.dumps(str(p))}))\n" for p in denied) - rules += "".join(f"(allow file-read-data (subpath {json.dumps(str(p))}))\n" for p in allowed) - rules += f"(deny file-read-data (subpath {json.dumps(str(SCENARIOS))}))\n" # the answer keys stay hidden even though the harness directory is readable + rules = sandbox_read_rules(*read_policy(args, env)) profile = run_dir / "agent.sb" profile.write_text('(version 1)\n(allow default)\n(deny file-write*)\n' f'(allow file-write* (subpath {json.dumps(str(run_dir.resolve()))}) (subpath "/dev"))\n' @@ -301,6 +378,8 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - snapshots.start() start = time.monotonic() agent_exit, stdout, stderr = run_agent(args, env, workspace, prompt) + sync_credentials_back(getattr(args, "credentials", []), out) + sync_credentials_back(getattr(args, "credentials", []), out) seconds = round(time.monotonic() - start, 1) snaps = snapshots.finish() if agent_exit == INFRA_EXIT and infra_retries < args.infra_retries: @@ -313,7 +392,11 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - result["infraRetries"] = infra_retries result["networkEnforcement"] = "canary-checked" if cell["network"] == "off" and args.canary_cmd else "declared-only" result["fileWriteEnforcement"] = "trial-directory-only" if getattr(args, "file_sandbox", False) else "unrestricted" - result["fileReadEnforcement"] = [str(p) for p in read_denials(args, env)[0]] if getattr(args, "file_sandbox", False) else "unrestricted" + if getattr(args, "file_sandbox", False): + hidden, allowed = read_policy(args, env) + result["fileReadEnforcement"] = {"mode": "allow-list", "hidden": [str(p) for p in hidden], "allowed": [str(p) for p in allowed]} + else: + result["fileReadEnforcement"] = "unrestricted" context = context_report(out, [skill_name(Path(args.skill_dirs[name])) for name in cell["skills"]]) result["context"] = context if context.get("unexpected"): @@ -490,7 +573,7 @@ def user_defaults() -> dict: if not CONFIG_PATH.is_file(): return {} config = json.loads(CONFIG_PATH.read_text()) - unknown = set(config) - {"wright", "out", "wiki_dir", "env_pass", "deny_read", "skill_dirs"} + unknown = set(config) - {"wright", "out", "wiki_dir", "env_pass", "deny_read", "allow_read", "skill_dirs"} if unknown: raise SystemExit(f"{CONFIG_PATH}: unknown key(s) {sorted(unknown)}") defaults = {k: v for k, v in config.items() if k not in ("skill_dirs", "out", "wiki_dir")} @@ -523,7 +606,10 @@ def cmd_evaluate(args: argparse.Namespace) -> int: Needs no agent harness around it: isolation comes from the harness's own scrubbed environment, so it runs the same from a terminal or from inside another agent's shell.""" - args.file_sandbox = not args.no_file_sandbox # evaluation hides the answer keys and the host's checkouts from the agent by default + args.file_sandbox = not args.no_file_sandbox # evaluation hides the answer keys and the rest of the host from the agent by default + args.credentials = CREDENTIALS.get(args.adapter, []) + args.credentials = CREDENTIALS.get(args.adapter, []) + args.allow_read = [*(str(Path.home() / rel) for rel in ADAPTER_READS[args.adapter]), *args.allow_read] script = Path(__file__).parent / "adapters" / ADAPTERS[args.adapter] effort = f"BENCH_THINKING={shlex.quote(args.effort)} " if args.effort else "" agent_id = "-".join(filter(None, (args.adapter, args.model.replace("/", "_"), args.effort))) @@ -539,8 +625,6 @@ def cmd_evaluate(args: argparse.Namespace) -> int: problems = preflight(args, [normalize_cell(c) for c in cells]) if problems: raise SystemExit("cannot start:\n " + "\n ".join(problems)) - if args.file_sandbox and not args.deny_read: - print("note: no --deny-read, so the agent can still read the host's repository checkouts (the scenarios are always hidden)", flush=True) scenarios = args.scenarios or [s for s in all_scenario_ids() if args.split == "all" or load_scenario(s).get("split") == args.split] runnable = sum(args.trials for s in scenarios for c in cells if applicable(load_scenario(s), normalize_cell(c))) print(f"{args.adapter} {args.model}: {len(scenarios)} scenario(s), cells {', '.join(cell_label(normalize_cell(c)) for c in cells)}, {runnable} trial(s) into {args.out / args.name}", flush=True) @@ -583,8 +667,6 @@ def cmd_setup_oracle(_: argparse.Namespace) -> int: def main() -> int: - if len(sys.argv) > 1 and sys.argv[1] == "shim": - return bench_trace.shim_main(sys.argv[2:]) parser = argparse.ArgumentParser(description=__doc__) sub = parser.add_subparsers(dest="command", required=True) for name in ("validate", "run", "matrix", "evaluate"): @@ -600,7 +682,8 @@ def main() -> int: p.add_argument("--canary-cmd", help="shell command that must fail in the agent environment when the network is 'off'") p.add_argument("--timeout", type=int, default=1800) p.add_argument("--file-sandbox", action="store_true", help="macOS: restrict agent and descendant file writes to the trial directory, and hide the scenarios, other runs, the wiki, unselected skills, and --deny-read paths") - p.add_argument("--deny-read", nargs="*", default=[], metavar="PATH", help="directories the agent must not read, such as the checkouts of the repositories under test; needs --file-sandbox") + p.add_argument("--deny-read", nargs="*", default=[], metavar="PATH", help="extra directories hidden from the agent (the home directories and drives already are); needs --file-sandbox") + p.add_argument("--allow-read", nargs="*", default=[], metavar="PATH", help="paths the agent's CLI needs inside the hidden home directories (credentials, installation); `evaluate` adds its adapter's") p.add_argument("--infra-retries", type=int, default=2, help="retries when the agent exits 75 (provider or infrastructure failure)") p.add_argument("--infra-backoff", type=int, default=60, help="seconds before the first retry; each further retry waits one more multiple") run = sub.choices["run"] @@ -622,7 +705,7 @@ def main() -> int: ev.add_argument("--split", choices=("test", "train", "all"), default="test") ev.add_argument("--scenarios", nargs="*", choices=all_scenario_ids()) ev.add_argument("--trials", type=int, default=3) - ev.add_argument("--parallel", type=int, default=2) + ev.add_argument("--parallel", type=int, default=1, help="trials at a time; sequential by default so provider limits are not hit, and a run can continue across sessions") ev.add_argument("--seed", type=int, default=1) ev.add_argument("--dry-run", action="store_true", help="check the setup and print what would run, without running it") ev.add_argument("--no-file-sandbox", action="store_true", help="run without the macOS file sandbox: the agent can then read the scenario answer keys") diff --git a/benchmarks/agent/bench_trace.py b/benchmarks/agent/bench_trace.py index a46c3f8e..9a50f8af 100644 --- a/benchmarks/agent/bench_trace.py +++ b/benchmarks/agent/bench_trace.py @@ -298,3 +298,7 @@ def total(row: dict) -> int: "peakContextShare": round(peak["context"] / limit, 4) if limit and peak.get("context") else None, "toFirstValid": to_first, } + + +if __name__ == "__main__": + sys.exit(shim_main(sys.argv[2:])) # the tool shims run this file directly: `bench_trace.py shim args...` diff --git a/benchmarks/agent/test_adapters.py b/benchmarks/agent/test_adapters.py index 25e5e3f8..d842ad7f 100644 --- a/benchmarks/agent/test_adapters.py +++ b/benchmarks/agent/test_adapters.py @@ -112,6 +112,13 @@ def test_empty_model_catalog_is_a_provider_failure_but_a_wrong_model_is_not(self self.assertTrue(devin.transient("429 rate limit")) +class DevinEffortTest(unittest.TestCase): + def test_effort_is_read_from_the_model_id(self): + self.assertEqual(devin.split_effort("swe-2-max"), ("swe-2", "max")) + self.assertEqual(devin.split_effort("claude-opus-5-5-medium"), ("claude-opus-5-5", "medium")) + self.assertEqual(devin.split_effort("swe-2"), ("swe-2", None)) + + class NativeAdapterUsageTest(unittest.TestCase): def test_codex_inclusive_counts_are_split_without_counting_reasoning_twice(self): row = codex.usage_row({"input_tokens": 100, "cached_input_tokens": 60, "output_tokens": 20, "reasoning_output_tokens": 12}, 1.0, 272000) diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index d25fd77f..099916dc 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -204,7 +204,7 @@ def test_agent_cannot_read_answer_keys_other_runs_or_denied_paths(self): result = self.trial(f'{shlex.quote(sys.executable)} -c {shlex.quote(code)}', file_sandbox=True, deny_read=[str(denied)]) workspace = self.out / f"{SCENARIO}-wright/workspace" self.assertEqual(json.loads((workspace / "probe.json").read_text()), {"answer": "blocked", "denied": "blocked"}) - self.assertIn(str(denied.resolve()), result["fileReadEnforcement"]) + self.assertIn(str(denied.resolve()), result["fileReadEnforcement"]["hidden"]) def test_user_defaults_are_read_from_the_config_file(self): config = self.out / "config.json" @@ -223,13 +223,62 @@ def test_preflight_names_what_is_missing_before_a_run_starts(self): self.assertEqual(len(problems), 3) self.assertTrue(any("wright binary not found" in p for p in problems) and any("`devin` is not on PATH" in p for p in problems) and any("--skill-dir wright-skill" in p for p in problems)) - def test_sibling_runs_and_harness_data_roots_are_unreadable(self): + def test_read_policy_hides_the_host_and_allows_only_what_the_run_needs(self): root = self.out - args = argparse.Namespace(out=root / "run-a", out_root=root, wright=WRIGHT, skill_dirs={}, wiki_dir=None, deny_read=[]) - denied, allowed = agent_bench.read_denials(args, {"BENCH_RUN_DIR": str(root / "run-a" / "t"), "BENCH_SKILLS": ""}) - self.assertIn(root.resolve(), denied) - self.assertTrue(all(d in denied for d in agent_bench.DATA_ROOTS)) - self.assertIn((root / "run-a" / "t").resolve(), allowed) + skills = {"wright-skill": root / "s1", "opy-skill": root / "s2"} + args = argparse.Namespace(out=root / "run-a", out_root=root, wright=WRIGHT, skill_dirs=skills, wiki_dir=None, deny_read=[], allow_read=[str(root / "creds")], adapter="devin") + hidden, allowed = agent_bench.read_policy(args, {"BENCH_RUN_DIR": str(root / "run-a" / "t"), "BENCH_SKILLS": "wright-skill"}) + self.assertTrue(all(p in hidden for p in (Path("/Users"), root.resolve(), agent_bench.HERE, agent_bench.HERE))) + self.assertTrue(all(d.resolve() in hidden for d in agent_bench.DATA_ROOTS)) + self.assertIn(skills["opy-skill"].resolve(), hidden) + self.assertNotIn(skills["wright-skill"].resolve(), hidden) + for needed in (root / "run-a" / "t", skills["wright-skill"], root / "creds", agent_bench.HERE / "adapters", agent_bench.HERE / "bench_trace.py", agent_bench.HERE / "oracle", Path(WRIGHT).resolve().parent): + self.assertIn(needed.resolve(), allowed) + self.assertNotIn(agent_bench.HERE / "bench_grade.py", allowed) + + @unittest.skipUnless(sys.platform == "darwin" and shutil.which("sandbox-exec"), "macOS file sandbox") + def test_the_agent_sees_only_its_run_the_allowed_paths_and_the_shim(self): + import shlex + other = self.out / "elsewhere" + other.mkdir() + (other / "note.txt").write_text("a file the agent was not given") + granted = self.out / "granted" + granted.mkdir() + (granted / "note.txt").write_text("a file the adapter needs") + probes = {"unlisted": other / "note.txt", "granted": granted / "note.txt", "grader": agent_bench.HERE / "bench_grade.py", "shim": agent_bench.HERE / "bench_trace.py", + "answer": agent_bench.SCENARIOS / SCENARIO / "reference" / "mode.ws", "oracle": agent_bench.HERE / "oracle" / "compile.js"} + code = ("import json\nfrom pathlib import Path\nout = {}\n" + + "".join(f"try:\n Path({str(p)!r}).read_text(); out[{k!r}] = 'read'\nexcept PermissionError:\n out[{k!r}] = 'blocked'\n" for k, p in probes.items()) + + "Path('probe.json').write_text(json.dumps(out))\n") + result = self.trial(f'{shlex.quote(sys.executable)} -c {shlex.quote(code)}', file_sandbox=True, allow_read=[str(granted)]) + seen = json.loads((self.out / f"{SCENARIO}-wright/workspace/probe.json").read_text()) + self.assertEqual(seen, {"unlisted": "blocked", "granted": "read", "grader": "blocked", "shim": "read", "answer": "blocked", "oracle": "read"}) + self.assertEqual(result["fileReadEnforcement"]["mode"], "allow-list") + + def test_the_tool_shim_runs_without_the_grader(self): + shim = (agent_bench.HERE / "bench_trace.py").read_text() + self.assertNotIn("import bench_grade", shim) + self.assertIn("shim_main(sys.argv[2:])", shim) + + def test_a_refreshed_login_is_written_back_to_the_real_one_only_when_newer(self): + home, run = self.out / "home", self.out / "run" + (home / ".grok").mkdir(parents=True) + run.mkdir() + real, isolated = home / ".grok/auth.json", run / "grok-home/auth.json" + real.write_text("old-token") + os.chmod(real, 0o600) + isolated.parent.mkdir() + isolated.write_text("old-token") + pairs = [(".grok/auth.json", "grok-home/auth.json")] + self.assertEqual(agent_bench.sync_credentials_back(pairs, run, home), []) # unchanged + isolated.write_text("refreshed-token") + os.utime(isolated, (real.stat().st_mtime + 10, real.stat().st_mtime + 10)) + self.assertEqual(agent_bench.sync_credentials_back(pairs, run, home), [".grok/auth.json"]) + self.assertEqual((real.read_text(), oct(real.stat().st_mode & 0o777)), ("refreshed-token", "0o600")) + real.write_text("newer-real-token") # the user logged in again meanwhile: never overwrite a newer real login + os.utime(isolated, (real.stat().st_mtime - 10, real.stat().st_mtime - 10)) + self.assertEqual(agent_bench.sync_credentials_back(pairs, run, home), []) + self.assertEqual(real.read_text(), "newer-real-token") def test_tools_differ_only_in_availability(self): agent = f"cp {reference()}/* . && (wright check mode.ws >/dev/null 2>&1 || echo no-wright > missing-wright.txt)" From 24f187fc84cbb7bec07c61a998ee93e602a3be7e Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sat, 3 Oct 2026 01:45:01 +0800 Subject: [PATCH 10/32] feat(bench): add the suite command, a publishable results page, and a shell how-to --- benchmarks/agent/agent_bench.py | 62 ++++++++-- benchmarks/agent/bench_leaderboard.py | 164 ++++++++++++++++++++++++++ benchmarks/agent/test_agent_bench.py | 23 ++++ benchmarks/agent/test_leaderboard.py | 63 ++++++++++ docs/agent-benchmark-howto.md | 93 +++++++++++++++ docs/agent-benchmark.md | 1 + 6 files changed, 399 insertions(+), 7 deletions(-) create mode 100644 benchmarks/agent/bench_leaderboard.py create mode 100644 benchmarks/agent/test_leaderboard.py create mode 100644 docs/agent-benchmark-howto.md diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 1b77c7be..d6d56698 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -23,6 +23,7 @@ from pathlib import Path import bench_grade +import bench_leaderboard import bench_report import bench_score import bench_trace @@ -573,7 +574,7 @@ def user_defaults() -> dict: if not CONFIG_PATH.is_file(): return {} config = json.loads(CONFIG_PATH.read_text()) - unknown = set(config) - {"wright", "out", "wiki_dir", "env_pass", "deny_read", "allow_read", "skill_dirs"} + unknown = set(config) - {"wright", "out", "wiki_dir", "env_pass", "deny_read", "allow_read", "skill_dirs", "models"} if unknown: raise SystemExit(f"{CONFIG_PATH}: unknown key(s) {sorted(unknown)}") defaults = {k: v for k, v in config.items() if k not in ("skill_dirs", "out", "wiki_dir")} @@ -650,6 +651,33 @@ def cmd_evaluate(args: argparse.Namespace) -> int: return status +def model_slug(entry: dict) -> str: + return "-".join(filter(None, (entry["adapter"], entry["model"].replace("/", "_"), entry.get("effort")))) + + +def cmd_suite(args: argparse.Namespace) -> int: + """Evaluate every model of the user's list one after another, then write the publishable results page. + + Safe to repeat: finished runs are skipped, so quota or time limits only postpone the rest. Exit 3 when something is waiting for a rerun.""" + models = [m for m in (args.models or []) if not args.only or any(f"{m['adapter']}:{m['model']}".startswith(o) for o in args.only)] + if not models: + raise SystemExit(f'no models: add "models": [{{"adapter": "devin", "model": "swe-2-max"}}, ...] to {CONFIG_PATH}') + root = args.out / args.suite_name + outcome = {} + for entry in models: + slug = model_slug(entry) + print(f"\n=== {slug}", flush=True) + sub = argparse.Namespace(**{**vars(args), "adapter": entry["adapter"], "model": entry["model"], "effort": entry.get("effort"), "name": slug, "out": root}) + try: + outcome[slug] = {0: "done", 3: "waiting: provider limit or outage, rerun later"}.get(cmd_evaluate(sub), "finished with errors") + except SystemExit as stop: + outcome[slug] = f"skipped: {stop.code}" + print("\n" + "\n".join(f"{slug}: {state}" for slug, state in outcome.items())) + if not args.dry_run: + bench_leaderboard.main(sorted(root.glob("*/")), root / "leaderboard") + return 3 if any(state.startswith("waiting") for state in outcome.values()) else 0 + + def cmd_wiki_snapshot(args: argparse.Namespace) -> int: record = bench_wiki.snapshot(args.base, args.dir, tuple(args.categories)) print(f"{len(record['documents'])} document(s) from {record['source']} into {args.dir}\nsnapshotSha256 {record['snapshotSha256']}") @@ -669,11 +697,12 @@ def cmd_setup_oracle(_: argparse.Namespace) -> int: def main() -> int: parser = argparse.ArgumentParser(description=__doc__) sub = parser.add_subparsers(dest="command", required=True) - for name in ("validate", "run", "matrix", "evaluate"): - p = sub.add_parser(name) + sub.add_parser("suite", help="evaluate every model listed in the user config in turn, then write the results page") + for name in ("validate", "run", "matrix", "evaluate", "suite"): + p = sub.choices[name] if name == "suite" else sub.add_parser(name) p.add_argument("--wright", default=str(ROOT / "target/debug/wright"), help="Wright binary under test") p.add_argument("--out", type=Path, default=Path.home() / ".cache/wright-agent-bench", help="outside any repository, so agents cannot discover its instruction files") - for name in ("run", "matrix", "evaluate"): + for name in ("run", "matrix", "evaluate", "suite"): p = sub.choices[name] p.add_argument("--skill-dir", action="append", default=[], metavar="NAME=DIR", help=f"pinned skill directory for one of {', '.join(SKILLS)}; repeatable") p.add_argument("--wiki-dir", type=Path, help="pinned wiki snapshot, linked as ./wiki for knowledge 'wiki'; content hashes are verified") @@ -701,6 +730,21 @@ def main() -> int: ev.add_argument("--model", required=True, help="BENCH_MODEL, in the form the adapter expects") ev.add_argument("--effort", help="BENCH_THINKING, where the adapter supports it") ev.add_argument("--name", default=time.strftime("%Y%m%d-%H%M%S"), help="run directory under --out") + su = sub.choices["suite"] + su.add_argument("--suite-name", default="results", help="directory under --out holding every model's run and the results page") + su.add_argument("--only", nargs="*", metavar="ADAPTER[:MODEL]", help="evaluate only these entries of the models list") + for shared in (su,): + shared.add_argument("--cells", choices=("score", "controls"), default="score") + shared.add_argument("--split", choices=("test", "train", "all"), default="test") + shared.add_argument("--scenarios", nargs="*", choices=all_scenario_ids()) + shared.add_argument("--trials", type=int, default=3) + shared.add_argument("--parallel", type=int, default=1) + shared.add_argument("--seed", type=int, default=1) + shared.add_argument("--dry-run", action="store_true") + shared.add_argument("--no-file-sandbox", action="store_true") + lb = sub.add_parser("leaderboard", help="write the publishable results page (Markdown, HTML, JSON) from evaluation run directories") + lb.add_argument("dirs", nargs="+", type=Path) + lb.add_argument("--page-out", type=Path, help="directory for the page; `leaderboard` inside the first directory's parent by default") ev.add_argument("--cells", choices=("score", "controls"), default="score", help="score: the canonical cell only; controls: also baseline and language-appropriate controls") ev.add_argument("--split", choices=("test", "train", "all"), default="test") ev.add_argument("--scenarios", nargs="*", choices=all_scenario_ids()) @@ -733,7 +777,9 @@ def main() -> int: for choice in sub.choices.values(): known = {a.dest for a in choice._actions} choice.set_defaults(**{k: v for k, v in defaults.items() if k in known}) - sub.choices["evaluate"].set_defaults(wright=defaults.get("wright") or shutil.which("wright") or str(ROOT / "target/debug/wright")) + for name in ("evaluate", "suite"): + sub.choices[name].set_defaults(wright=defaults.get("wright") or shutil.which("wright") or str(ROOT / "target/debug/wright")) + sub.choices["suite"].set_defaults(models=defaults.get("models")) args = parser.parse_args() if hasattr(args, "wright"): args.wright = str(Path(args.wright).resolve()) @@ -746,7 +792,7 @@ def main() -> int: if name not in SKILLS or not directory: raise SystemExit(f"--skill-dir expects NAME=DIR with NAME one of {', '.join(SKILLS)}: {item}") args.skill_dirs[name] = Path(directory) - if getattr(args, "deny_read", None) and not getattr(args, "file_sandbox", False) and args.command != "evaluate": + if getattr(args, "deny_read", None) and not getattr(args, "file_sandbox", False) and args.command not in ("evaluate", "suite"): raise SystemExit("--deny-read needs --file-sandbox") if args.command == "validate": return 0 if validate(args.wright, args.out) else 1 @@ -758,6 +804,8 @@ def main() -> int: return cmd_wiki_skill(args) if args.command == "report": return bench_report.main(args.dirs, args.wright, args.regrade, lambda s: load_scenario(s), args.reference) + if args.command == "leaderboard": + return bench_leaderboard.main(args.dirs, args.page_out or args.dirs[0].parent / "leaderboard") if args.command == "compare": print(bench_score.compare(args.dirs)) return 0 @@ -765,7 +813,7 @@ def main() -> int: languages = args.language or ["workshop", "opy"] expected = {lang: [s for s in all_scenario_ids() if load_scenario(s)["language"] == lang and load_scenario(s).get("split") == "test"] for lang in languages} return bench_score.main(args.dirs, languages, expected, None) - return {"run": cmd_run, "matrix": cmd_matrix, "evaluate": cmd_evaluate}[args.command](args) + return {"run": cmd_run, "matrix": cmd_matrix, "evaluate": cmd_evaluate, "suite": cmd_suite}[args.command](args) if __name__ == "__main__": diff --git a/benchmarks/agent/bench_leaderboard.py b/benchmarks/agent/bench_leaderboard.py new file mode 100644 index 00000000..e28ad5f1 --- /dev/null +++ b/benchmarks/agent/bench_leaderboard.py @@ -0,0 +1,164 @@ +"""Publishable results page (#467): one table of Wright Agent Scores for every evaluated agent, as Markdown, a self-contained HTML page, and JSON.""" + +from __future__ import annotations + +import html +import json +import re +from collections import Counter +from datetime import date +from pathlib import Path + +import bench_report + +TRACKS = (("workshop", "Workshop"), ("opy", "OverPy")) +READING = ( + "The score is the share of tasks an agent completed with a valid, safe result, averaged over 8 realistic Workshop tasks per language, each tried 3 times.", + "The bar shows the score; the range in brackets is the 95% confidence interval. With 8 tasks the range is wide: agents marked **tied with top** cannot be told apart from the first row.", + "It is a reference for how an agent behaves with Wright and its guide, not a measure of general ability, and a model's score depends on the agent program that runs it.", +) +LIMITS = ( + "Network access was switched off by instruction only; the harness did not block it.", + "A run that hit the time limit counts as a failure. Runs cut off by provider outages are retried, not counted.", + "The suite has 8 test tasks per language, so differences of a few points mean nothing.", +) + + +def load_entries(dirs: list[Path]) -> list[dict]: + """One entry per evaluation run directory that has a score.json.""" + entries = [] + for directory in dirs: + path = directory / "score.json" + if not path.is_file(): + continue + cards = {c["language"]: c for c in json.loads(path.read_text())["cards"] if "refused" not in c} + if not cards: + continue + first = next(iter(cards.values())) + ident = first["identity"] + info = next((r.get("agentInfo") or {} for r in bench_report.load([directory]) if r.get("agentInfo")), {}) + entries.append({ + "run": directory.name, + "harness": info.get("agent") or ident["agent"], "harnessVersion": info.get("version"), + "model": info.get("model") or ident.get("model"), "effort": info.get("effort") or ident.get("effort"), + "tracks": {lang: {"score": c["score"], "ci": c["ci95"], "trials": c["trialsPerScenario"], "scenarios": c["scenarios"], "passPowK": c["passPowK"], "provisional": c["provisional"], "exclusions": c["exclusions"]} for lang, c in cards.items()}, + "environment": {"wrightSha256": ident["wrightSha256"], "wright": ident["wright"], "skills": ident["skills"], "suite": first["suite"].get("hash")}, + "network": first["networkEnforcement"], "harnessCommit": first["harness"], + }) + return entries + + +def mean_score(entry: dict) -> float: + scores = [t["score"] for t in entry["tracks"].values()] + return sum(scores) / len(scores) + + +def overlaps(a: list[float], b: list[float]) -> bool: + return a[1] >= b[0] and a[0] <= b[1] + + +def standing(entry: dict, leader: dict) -> str: + """`top`, `tied with top` when every track's interval overlaps the leader's, else `below top`.""" + if entry is leader: + return "top" + shared = [lang for lang in entry["tracks"] if lang in leader["tracks"]] + return "tied with top" if shared and all(overlaps(entry["tracks"][l]["ci"], leader["tracks"][l]["ci"]) for l in shared) else "below top" + + +def build(dirs: list[Path]) -> dict: + entries = load_entries(dirs) + if not entries: + return {"entries": [], "excluded": [], "environment": None, "date": date.today().isoformat()} + key = lambda e: json.dumps(e["environment"], sort_keys=True) + main_key = Counter(key(e) for e in entries).most_common(1)[0][0] + ranked = sorted([e for e in entries if key(e) == main_key], key=mean_score, reverse=True) + for entry in ranked: + entry["standing"] = standing(entry, ranked[0]) + other = [{"run": e["run"], "reason": "made against a different Wright binary, skills, or task suite"} for e in entries if key(e) != main_key] + return {"entries": ranked, "excluded": other, "environment": ranked[0]["environment"], "date": date.today().isoformat()} + + +def label(entry: dict) -> str: + version = re.sub(r" \([0-9a-f]{7,}\)$", "", entry["harnessVersion"] or "") # the build hash adds nothing for a reader + return version if version.startswith(entry["harness"]) else " ".join(filter(None, (entry["harness"], version))) + + +def bar(score: float, width: int = 10) -> str: + filled = round(score / 100 * width) + return "█" * filled + "░" * (width - filled) + + +def markdown(data: dict) -> str: + env = data["environment"] + lines = ["# Wright Agent Score", "", f"How well coding agents work on real Overwatch Workshop projects with Wright. Results of {data['date']}.", ""] + if not data["entries"]: + return "\n".join(lines + ["No results yet.", ""]) + lines += ["| # | Agent | Model | Effort | " + " | ".join(f"{n} score" for _, n in TRACKS) + " | Against the top |", "| --- | --- | --- | --- | " + " | ".join("---" for _ in TRACKS) + " | --- |"] + for rank, e in enumerate(data["entries"], 1): + cells = [] + for lang, _ in TRACKS: + t = e["tracks"].get(lang) + cells.append(f"`{bar(t['score'])}` **{t['score']:.0f}** ({t['ci'][0]:.0f}–{t['ci'][1]:.0f})" + (" ⚠️" if t["provisional"] else "") if t else "n/a") + lines.append(f"| {rank} | {label(e)} | {e['model']} | {e['effort'] or 'default'} | " + " | ".join(cells) + f" | {e['standing']} |") + lines += ["", "## How to read this", "", *(f"- {s}" for s in READING), "", "## Limits", "", *(f"- {s}" for s in LIMITS)] + if any(t["provisional"] for e in data["entries"] for t in e["tracks"].values()): + lines.append("- ⚠️ marks a score that is provisional because the run did not cover every task or had unequal trials.") + skills = ", ".join(f"{k} `{v[:8]}`" for k, v in env["skills"].items()) or "none" + commits = ", ".join(f"`{c[:8]}`" for c in sorted({c for e in data["entries"] for c in e["harnessCommit"]})) or "not recorded" + lines += ["", "## What was run", "", f"- Wright: {env['wright']} (sha256 `{env['wrightSha256'][:12]}`)", f"- Skills: {skills}", f"- Task suite: `{env['suite'][:12]}`", + f"- Harness commit: {commits}", "- Scores come from the benchmark in `benchmarks/agent`; the run directories hold every result.json."] + if data["excluded"]: + lines += ["", "## Not comparable", "", *(f"- `{o['run']}`: {o['reason']}." for o in data["excluded"])] + return "\n".join(lines) + "\n" + + +def page(data: dict) -> str: + esc = html.escape + rows = [] + for rank, e in enumerate(data["entries"], 1): + cells = [] + for lang, _ in TRACKS: + t = e["tracks"].get(lang) + if not t: + cells.append("n/a") + continue + lo, hi = t["ci"] + cells.append(f'
' + f'{t["score"]:.0f} {lo:.0f}–{hi:.0f}{" ⚠️" if t["provisional"] else ""}') + rows.append(f'{rank}{esc(label(e))}{esc(str(e["model"]))}{esc(e["effort"] or "default")}{"".join(cells)}{esc(e["standing"])}') + env = data["environment"] or {} + skills = ", ".join(f"{k} {v[:8]}" for k, v in (env.get("skills") or {}).items()) or "none" + head = "".join(f"{n} score" for _, n in TRACKS) + body = (f"{head}{''.join(rows)}
#AgentModelEffortAgainst the top
" + if rows else "

No results yet.

") + li = lambda items: "".join(f"
  • {esc(s.replace('**', ''))}
  • " for s in items) + return f""" +Wright Agent Score +
    +

    Wright Agent Score

    How well coding agents work on real Overwatch Workshop projects with Wright. Results of {esc(data.get("date", ""))}.

    +{body} +

    How to read this

      {li(READING)}
    +

    Limits

      {li(LIMITS)}
    +

    What was run

    Wright {esc(str(env.get("wright", "")))} · sha256 {esc(str(env.get("wrightSha256", ""))[:12])} · skills {esc(skills)} · task suite {esc(str(env.get("suite", ""))[:12])}

    +
    +""" + + +def main(dirs: list[Path], out: Path) -> int: + data = build(dirs) + out.mkdir(parents=True, exist_ok=True) + (out / "LEADERBOARD.md").write_text(markdown(data)) + (out / "leaderboard.html").write_text(page(data)) + (out / "leaderboard.json").write_text(json.dumps(data, indent=2) + "\n") + print(markdown(data)) + print(f"wrote {out / 'LEADERBOARD.md'}, {out / 'leaderboard.html'}, {out / 'leaderboard.json'}") + return 0 if data["entries"] else 1 diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 099916dc..2875d6e0 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -280,6 +280,29 @@ def test_a_refreshed_login_is_written_back_to_the_real_one_only_when_newer(self) self.assertEqual(agent_bench.sync_credentials_back(pairs, run, home), []) self.assertEqual(real.read_text(), "newer-real-token") + def test_suite_runs_each_model_in_turn_continues_past_a_wait_or_a_skip_and_writes_the_page(self): + models = [{"adapter": "devin", "model": "swe-2-max"}, {"adapter": "codex", "model": "gpt-6-luna", "effort": "xhigh"}, {"adapter": "pi", "model": "m"}] + args = argparse.Namespace(models=models, only=None, out=self.out, suite_name="s", dry_run=False) + calls = [] + + def evaluate(sub): + calls.append((sub.adapter, sub.name, sub.effort, sub.out)) + if sub.adapter == "codex": + return 3 + if sub.adapter == "pi": + raise SystemExit("cannot start: pi missing") + return 0 + + with patch.object(agent_bench, "cmd_evaluate", evaluate), patch.object(agent_bench.bench_leaderboard, "main") as page: + status = agent_bench.cmd_suite(args) + self.assertEqual(status, 3) # codex is waiting for a rerun + self.assertEqual([c[1] for c in calls], ["devin-swe-2-max", "codex-gpt-6-luna-xhigh", "pi-m"]) + self.assertTrue(all(c[3] == self.out / "s" for c in calls)) + page.assert_called_once() + with patch.object(agent_bench, "cmd_evaluate", evaluate): + only = agent_bench.cmd_suite(argparse.Namespace(models=models, only=["devin"], out=self.out, suite_name="s2", dry_run=True)) + self.assertEqual(only, 0) + def test_tools_differ_only_in_availability(self): agent = f"cp {reference()}/* . && (wright check mode.ws >/dev/null 2>&1 || echo no-wright > missing-wright.txt)" none = self.trial(agent, tool="none") diff --git a/benchmarks/agent/test_leaderboard.py b/benchmarks/agent/test_leaderboard.py new file mode 100644 index 00000000..56c6aa9e --- /dev/null +++ b/benchmarks/agent/test_leaderboard.py @@ -0,0 +1,63 @@ +import json +import shutil +import tempfile +import unittest +from pathlib import Path + +import bench_leaderboard +import bench_score +from test_score import SCENARIOS, runs + +ROOT = Path(__file__).resolve().parents[2] + + +class LeaderboardTest(unittest.TestCase): + def setUp(self): + self.root = Path(tempfile.mkdtemp(dir=ROOT / "target")) + self.addCleanup(shutil.rmtree, self.root, True) + + def write(self, name, usable, sha="a" * 64, model="m", tracks=("workshop", "opy")): + directory = self.root / name + directory.mkdir() + cards = [bench_score.card([{**r, "language": lang} for r in runs(SCENARIOS[:usable], sha=sha, model=model)], lang, SCENARIOS) for lang in tracks] + (directory / "score.json").write_text(json.dumps({"contract": bench_score.CONTRACT, "cards": cards})) + return directory + + def test_entries_are_ranked_and_tied_with_the_top_when_intervals_overlap(self): + data = bench_leaderboard.build([self.write("low", 1), self.write("high", 8), self.write("close", 7)]) + names = [e["run"] for e in data["entries"]] + self.assertEqual(names[0], "high") + standing = {e["run"]: e["standing"] for e in data["entries"]} + self.assertEqual((standing["high"], standing["close"], standing["low"]), ("top", "tied with top", "below top")) + + def test_runs_against_a_different_wright_are_listed_as_not_comparable(self): + data = bench_leaderboard.build([self.write("a", 6), self.write("b", 5), self.write("other", 6, sha="d" * 64)]) + self.assertEqual({e["run"] for e in data["entries"]}, {"a", "b"}) + self.assertEqual([o["run"] for o in data["excluded"]], ["other"]) + self.assertIn("Not comparable", bench_leaderboard.markdown(data)) + + def test_markdown_and_page_give_the_headline_a_reader_needs(self): + data = bench_leaderboard.build([self.write("only", 6)]) + text, page = bench_leaderboard.markdown(data), bench_leaderboard.page(data) + for needed in ("Wright Agent Score", "Workshop score", "OverPy score", "top", "How to read this", "Limits"): + self.assertIn(needed, text) + self.assertIn("█", text) + self.assertIn('class="bar"', page) + self.assertIn("width:75.0%", page) # 6 of 8 scenarios + self.assertNotIn("` or `openai/` | `openai/` only | needs an API key in the environment | + +The agent program runs the model, so the same model scores differently under different programs. The page shows both. + +## What the page tells you + +- The score is the share of tasks an agent finished with a valid, safe result. A bar shows it; the bracketed range is the 95% interval. +- With 8 tasks per language the range is wide. A row marked "tied with top" cannot be told apart from the first. +- It is a reference for how an agent behaves with Wright and its guide, not a measure of general ability. Network access is off by instruction only, not blocked. + +## When something goes wrong + +| What you see | Meaning | +| --- | --- | +| Exit code 3, "waiting" | quota or outage; run the same command later | +| `cannot start: ...` | `--dry-run` names what is missing (binary, login, skill directory, oracle) | +| An agent exits at once with an auth error | its login expired; sign in again with that program | +| An agent cannot start under the file sandbox | add the paths it needs: `--allow-read PATH` or `allow_read` in the config | + +The agent only sees its own run directory, the condition's skills, the tool binaries, and what its program needs to start; everything else on the machine is hidden from it, so it cannot find the answers or other runs. diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index ade78d66..b940df10 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -2,6 +2,7 @@ - Contracts: `wright-agent-bench/v3` (a run result) and `wright-agent-score/v1` (a score card) - Harness: [`benchmarks/agent/agent_bench.py`](../benchmarks/agent/agent_bench.py) +- Run it from a shell: [agent-benchmark-howto.md](agent-benchmark-howto.md) - Design and requirements: [`SPEC-414`](specs/SPEC-414-agent-benchmark-comparison.md) The benchmark answers one product question: can a general coding agent, with no From 42d60acb178d1dfe7d96bc19036af52cd8ef0a4c Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sat, 3 Oct 2026 01:53:16 +0800 Subject: [PATCH 11/32] refactor(bench): keep runs, wikis, and pinned skills under one directory --- benchmarks/agent/agent_bench.py | 7 ++++--- docs/agent-benchmark-howto.md | 18 +++++++++++++++--- docs/agent-benchmark.md | 4 ++-- 3 files changed, 21 insertions(+), 8 deletions(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index d6d56698..53ce29c0 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -208,7 +208,8 @@ def canaries(cell: dict, env: dict, workspace: Path, args: argparse.Namespace) - return None -DATA_ROOTS = (Path.home() / ".cache/wright-agent-bench", Path.home() / ".local/share/wright-agent-bench") # other runs, wikis, and pinned skills live here by default +BENCH_HOME = Path(os.environ.get("WRIGHT_BENCH_HOME", Path.home() / ".local/share/wright-agent-bench")) # the one place for runs (`runs/`), wiki snapshots, and pinned skills +DATA_ROOTS = (BENCH_HOME, Path.home() / ".cache/wright-agent-bench") # hidden from agents; the second is where earlier versions wrote runs HIDDEN_ROOTS = (Path("/Users"), Path("/Volumes"), Path.home()) # the host's home directories and external drives are hidden unless listed below ADAPTER_READS = { # what each adapter reads from the real home before the agent starts: credentials, configuration, installation (relative to the home directory) "devin": [".local/share/devin", ".config/devin"], "pi": [".pi"], "codex": [".codex/auth.json"], "agy": [".gemini/antigravity-cli"], @@ -701,7 +702,7 @@ def main() -> int: for name in ("validate", "run", "matrix", "evaluate", "suite"): p = sub.choices[name] if name == "suite" else sub.add_parser(name) p.add_argument("--wright", default=str(ROOT / "target/debug/wright"), help="Wright binary under test") - p.add_argument("--out", type=Path, default=Path.home() / ".cache/wright-agent-bench", help="outside any repository, so agents cannot discover its instruction files") + p.add_argument("--out", type=Path, default=BENCH_HOME / "runs", help="outside any repository, so agents cannot discover its instruction files") for name in ("run", "matrix", "evaluate", "suite"): p = sub.choices[name] p.add_argument("--skill-dir", action="append", default=[], metavar="NAME=DIR", help=f"pinned skill directory for one of {', '.join(SKILLS)}; repeatable") @@ -760,7 +761,7 @@ def main() -> int: skill.add_argument("--catalog", type=Path, required=True, help="workshop-rs catalog.json, for Workshop names and ids") skill.add_argument("--opy-manifest", type=Path, required=True, help="opy-rs manifest.json, for upstream OverPy spellings") wiki = sub.add_parser("wiki-snapshot", help="fetch the Workshop wiki Markdown mirror into a pinned local snapshot") - wiki.add_argument("--dir", type=Path, default=Path.home() / ".cache/wright-agent-bench-wiki") + wiki.add_argument("--dir", type=Path, default=BENCH_HOME / "wiki") wiki.add_argument("--base", default=bench_wiki.BASE) wiki.add_argument("--categories", nargs="+", default=list(bench_wiki.CATEGORIES), help="wiki categories to crawl (add tutorials for the second tier)") report = sub.add_parser("report", help="summarize result.json files") diff --git a/docs/agent-benchmark-howto.md b/docs/agent-benchmark-howto.md index 9f35bb58..48edee9f 100644 --- a/docs/agent-benchmark-howto.md +++ b/docs/agent-benchmark-howto.md @@ -26,6 +26,18 @@ Run it from a shell, one command for every model, and get a results page you can ``` `skill_dirs` points at the `wrightkit/skills` checkout. Keep the Wright binary, the skills, and the scenarios unchanged for as long as you want results to be comparable. +## Where things live + +Everything is under one directory, `~/.local/share/wright-agent-bench` (the `WRIGHT_BENCH_HOME` variable moves it): + +| Path | Holds | +| --- | --- | +| `runs/` | every evaluation run and the results page (`runs/results/leaderboard/`) | +| `wiki/` | local wiki snapshots (not published) | +| `skills-*/` | pinned copies of the skills under test | + +Runs made by earlier versions are in `~/.cache/wright-agent-bench`; nothing writes there any more. + ## Run everything ```sh @@ -35,7 +47,7 @@ python3 benchmarks/agent/agent_bench.py suite It evaluates the models one after another, each on 16 tasks tried 3 times, one trial at a time so provider limits are not hit. It is safe to stop and repeat: finished trials are skipped, and a model that hits a quota or an outage waits for the next run (the command ends with exit code 3 and says which). Run it again later, on another day if needed, and the same command carries on. -The results are in `~/.cache/wright-agent-bench/results/leaderboard/`: +The results are in `~/.local/share/wright-agent-bench/runs/results/leaderboard/`: | File | For | | --- | --- | @@ -54,8 +66,8 @@ python3 benchmarks/agent/agent_bench.py evaluate --adapter codex --model gpt-6-l Put several runs side by side, or rebuild the page from chosen runs: ```sh -python3 benchmarks/agent/agent_bench.py compare ~/.cache/wright-agent-bench/{run-a,run-b} -python3 benchmarks/agent/agent_bench.py leaderboard ~/.cache/wright-agent-bench/{run-a,run-b} +python3 benchmarks/agent/agent_bench.py compare ~/.local/share/wright-agent-bench/runs/{run-a,run-b} +python3 benchmarks/agent/agent_bench.py leaderboard ~/.local/share/wright-agent-bench/runs/{run-a,run-b} ``` Runs made against a different Wright binary, skills, or task suite are listed as not comparable instead of being ranked. diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index b940df10..93a2f430 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -145,7 +145,7 @@ Pass `--env-pass HOME` when the agent authenticates from the real home directory Two checks protect the context. The workspace must not sit below a directory that holds instruction files (`AGENTS.md`, `CLAUDE.md`, and similar), because agents discover them by walking up; the default `--out` is -`~/.cache/wright-agent-bench` for that reason, and a violation marks the run +`~/.local/share/wright-agent-bench/runs` for that reason, and a violation marks the run `invalid` (`--no-ancestor-check` disables it). Network `off` is enforced only when `--canary-cmd` is given and fails inside the agent environment; without it the result records `networkEnforcement: declared-only`, which is what the shell @@ -358,7 +358,7 @@ and agents can be run whenever quota allows, in any order. Put the runs side by with ```sh -python3 benchmarks/agent/agent_bench.py compare ~/.cache/wright-agent-bench/{devin-swe2-stage1,codex-luna-xhigh,pi-luna-xhigh} +python3 benchmarks/agent/agent_bench.py compare ~/.local/share/wright-agent-bench/runs/{devin-swe2-stage1,codex-luna-xhigh,pi-luna-xhigh} ``` which prints one table of scores, intervals, trials, and exclusions, and warns when From 7e688beb5b173774a9c25618d6dc6146813aa3fd Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sat, 3 Oct 2026 13:07:50 +0800 Subject: [PATCH 12/32] fix(bench): retry provider-interrupted trials when a run is repeated --- benchmarks/agent/agent_bench.py | 3 ++- benchmarks/agent/test_agent_bench.py | 7 +++++++ 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 53ce29c0..2b92683f 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -532,7 +532,8 @@ def cmd_matrix(args: argparse.Namespace) -> int: def work(job: tuple) -> None: scenario_id, agent, cell, trial = job out = trial_dir(args.out, scenario_id, agent["id"], cell, trial) - if (out / "result.json").is_file(): + finished = out / "result.json" + if finished.is_file() and json.loads(finished.read_text()).get("status") != "provider-interrupted": # an interrupted trial is retried on the next run return with lock: if state["streak"] >= STOP_AFTER_INTERRUPTIONS: diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 2875d6e0..c1471657 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -159,6 +159,13 @@ def test_matrix_skips_inapplicable_cells_and_stops_after_repeated_interruptions( self.assertIn("not applicable: repair-runaway-loop", printed.getvalue()) self.assertEqual(len(list((self.out / "m").rglob("result.json"))), 2) # the third and fourth jobs were left unattempted self.assertIn("left unattempted", printed.getvalue()) + interrupted = {p for p in (self.out / "m").rglob("result.json") if json.loads(p.read_text())["status"] == "provider-interrupted"} + self.assertEqual(len(interrupted), 2) + config.write_text(config.read_text().replace('"cmd": "exit 75"', '"cmd": "exit 0"')) # the provider is back + with contextlib.redirect_stdout(io.StringIO()): + agent_bench.cmd_matrix(args) + self.assertTrue(all(json.loads(p.read_text())["status"] != "provider-interrupted" for p in (self.out / "m").rglob("result.json"))) + self.assertEqual(len(list((self.out / "m").rglob("result.json"))), 4) # the interrupted two were rerun and the rest ran def test_scenarios_are_solvable_and_not_vacuous(self): self.assertTrue(agent_bench.validate(WRIGHT, self.out / "validate")) From 338687bad889acdb91b7e060cd68599b6d514c9a Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sat, 3 Oct 2026 13:57:38 +0800 Subject: [PATCH 13/32] fix(bench): refuse to continue a run with a different Wright binary --- benchmarks/agent/agent_bench.py | 17 +++++++++++++++++ benchmarks/agent/test_agent_bench.py | 14 ++++++++++++++ 2 files changed, 31 insertions(+) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 2b92683f..8975c8ec 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -604,6 +604,21 @@ def preflight(args: argparse.Namespace, cells: list[dict]) -> list[str]: return problems +def wright_mismatch(run_dir: Path, wright: str) -> str | None: + """Why a repeated run must not continue with this Wright binary, or None. + + A run is repeated to finish it later, possibly after `wright` on PATH was upgraded. Trials made with two binaries cannot be scored + together, so the mismatch is refused up front instead of being found when the score card is refused.""" + current = file_sha256(Path(wright)) + for path in sorted(run_dir.glob("*/*/*/result.json")): + env = json.loads(path.read_text()).get("environment", {}) + recorded = env.get("wrightSha256") + if recorded and recorded != current: + return (f"{run_dir.name} already has trials made with {env.get('wright')} (sha256 {recorded[:12]}), but {wright} is a different binary. " + f"Pass --wright with the binary that made them, or start a new run with --name.") + return None + + def cmd_evaluate(args: argparse.Namespace) -> int: """One command from agent and model to data and document: run the matrix, then write report, score cards, and RESULTS.md. @@ -626,6 +641,8 @@ def cmd_evaluate(args: argparse.Namespace) -> int: raise SystemExit("no cell can run: pass --skill-dir wright-skill=DIR") args.env_pass = sorted({*args.env_pass, *(DIRECT_ENV if args.adapter == "direct" else ("HOME",))}) # the credentials the adapter copies or reads problems = preflight(args, [normalize_cell(c) for c in cells]) + if (problem := wright_mismatch(args.out / args.name, args.wright)): + problems.append(problem) if problems: raise SystemExit("cannot start:\n " + "\n ".join(problems)) scenarios = args.scenarios or [s for s in all_scenario_ids() if args.split == "all" or load_scenario(s).get("split") == args.split] diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index c1471657..e5de3845 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -310,6 +310,20 @@ def evaluate(sub): only = agent_bench.cmd_suite(argparse.Namespace(models=models, only=["devin"], out=self.out, suite_name="s2", dry_run=True)) self.assertEqual(only, 0) + def test_a_repeated_run_refuses_a_different_wright_binary(self): + run = self.out / "r" + trial = run / "s" / "agent" / "cell-1" + trial.mkdir(parents=True) + (trial / "result.json").write_text(json.dumps({"environment": {"wright": "wright 0.7.0", "wrightSha256": "a" * 64}})) + other = self.out / "wright-other" + other.write_text("a different binary") + message = agent_bench.wright_mismatch(run, str(other)) + self.assertIn("wright 0.7.0", message) + self.assertIn("--wright", message) + self.assertIsNone(agent_bench.wright_mismatch(self.out / "fresh", str(other))) # a new run has nothing to disagree with + (trial / "result.json").write_text(json.dumps({"environment": {"wright": "x", "wrightSha256": agent_bench.file_sha256(other)}})) + self.assertIsNone(agent_bench.wright_mismatch(run, str(other))) + def test_tools_differ_only_in_availability(self): agent = f"cp {reference()}/* . && (wright check mode.ws >/dev/null 2>&1 || echo no-wright > missing-wright.txt)" none = self.trial(agent, tool="none") From 4b7d548ba413c55ea1ea7b7a9d0f774cc0ec9172 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sat, 3 Oct 2026 20:30:25 +0800 Subject: [PATCH 14/32] fix(bench): harden score edge cases and cut adapter duplication Independent review of the v3 score branch found a verbatim duplicate of the credential block, per-adapter copies of the transient-failure classification, crashes when --effort is omitted, preflight that missed unusable sandboxes and missing logins, a KeyError on results without a status, and a leaderboard that could rank a one-track run above full two-track runs. - share INFRA_EXIT and TRANSIENT through adapters/common.py - tolerate a missing BENCH_THINKING in the agy and codex adapters - preflight the file sandbox and the adapter's primary login - classify any non-completed, non-timeout status as excluded - rank only runs covering every language track; partial runs are listed as not comparable - derive leaderboard prose and network claims from the actual runs - exclude BUILD.json from its own skill hash so verification is stable - fail the suite exit code when a model finishes with errors --- benchmarks/agent/adapters/agy.py | 18 ++++--- benchmarks/agent/adapters/claude_code.py | 10 ++-- benchmarks/agent/adapters/codex.py | 18 ++++--- benchmarks/agent/adapters/common.py | 5 ++ benchmarks/agent/adapters/devin.py | 5 +- benchmarks/agent/adapters/direct.py | 5 +- benchmarks/agent/adapters/grok.py | 5 +- benchmarks/agent/adapters/opencode.py | 9 ++-- benchmarks/agent/adapters/pi.py | 10 ++-- benchmarks/agent/agent_bench.py | 68 +++++++++--------------- benchmarks/agent/bench_leaderboard.py | 66 +++++++++++++++-------- benchmarks/agent/bench_score.py | 10 ++-- benchmarks/agent/test_agent_bench.py | 14 ++++- benchmarks/agent/test_leaderboard.py | 6 +++ benchmarks/agent/wiki_skill.py | 3 +- docs/README.md | 2 + docs/agent-benchmark-howto.md | 2 +- docs/agent-benchmark.md | 6 +-- 18 files changed, 151 insertions(+), 111 deletions(-) diff --git a/benchmarks/agent/adapters/agy.py b/benchmarks/agent/adapters/agy.py index 1d8b27e1..17775d4e 100644 --- a/benchmarks/agent/adapters/agy.py +++ b/benchmarks/agent/adapters/agy.py @@ -11,7 +11,7 @@ import time from pathlib import Path -from common import cli_version +from common import cli_version, INFRA_EXIT, TRANSIENT def usage_row(usage: dict, timestamp: float) -> dict: @@ -40,14 +40,18 @@ def main() -> int: shutil.copytree(skill, Path.cwd() / ".agents/skills" / skill.name) installed.append(skill.name) binary = shutil.which("agy", path=env.get("BENCH_HOST_PATH")) or "agy" - command = [binary, "--model", env["BENCH_MODEL"], "--effort", env["BENCH_THINKING"], + effort = env.get("BENCH_THINKING") + command = [binary, "--model", env["BENCH_MODEL"], *(["--effort", effort] if effort else []), "--output-format", "stream-json", "--print-timeout", "0", "--print", sys.stdin.read()] child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home)} seen, result, init, unexpected = set(), {}, {}, set() with (run / "agy-stderr.log").open("w") as stderr, open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: process = subprocess.Popen(command, env=child_env, stdout=subprocess.PIPE, stderr=stderr, text=True) for line in process.stdout: - event = json.loads(line) + try: + event = json.loads(line) + except json.JSONDecodeError: + continue now = time.time() transcript.write(json.dumps({"t": now, **event}) + "\n") if event["event"] == "init": @@ -65,16 +69,16 @@ def main() -> int: result = event["result"] code = process.wait() Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"installed": installed, "audit": "isolated-home; CLI does not export loaded skill context", "unexpected": sorted(unexpected)})) - Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "agy", "version": cli_version(binary), "model": env["BENCH_MODEL"], "effort": env["BENCH_THINKING"], "tools": None, "note": "the CLI does not expose its tool list"}, indent=2)) - (run / "adapter.json").write_text(json.dumps({"agent": "agy", "requestedModel": env["BENCH_MODEL"], "requestedEffort": env["BENCH_THINKING"], "observedModel": init.get("model"), "status": result.get("status"), "usageSource": "stream-step-usage", "providerUsage": result.get("usage")}, indent=2)) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "agy", "version": cli_version(binary), "model": env["BENCH_MODEL"], "effort": effort, "tools": None, "note": "the CLI does not expose its tool list"}, indent=2)) + (run / "adapter.json").write_text(json.dumps({"agent": "agy", "requestedModel": env["BENCH_MODEL"], "requestedEffort": effort, "observedModel": init.get("model"), "status": result.get("status"), "usageSource": "stream-step-usage", "providerUsage": result.get("usage")}, indent=2)) sys.stdout.write(result.get("response", "")) stderr_text = (run / "agy-stderr.log").read_text() sys.stderr.write(stderr_text) error = str(result.get("error", "")) + stderr_text if "no output produced" in stderr_text and "headless" in stderr_text: - return 75 + return INFRA_EXIT if code or result.get("status") != "SUCCESS": - return 75 if any(s in error.lower() for s in ("quota", "rate limit", "429", "temporarily", "503", "credits")) else (code or 1) + return INFRA_EXIT if any(s in error.lower() for s in TRANSIENT) else (code or 1) return 0 diff --git a/benchmarks/agent/adapters/claude_code.py b/benchmarks/agent/adapters/claude_code.py index 22c016e3..8d83afb0 100755 --- a/benchmarks/agent/adapters/claude_code.py +++ b/benchmarks/agent/adapters/claude_code.py @@ -13,6 +13,7 @@ import json import os +import re import shutil import subprocess import sys @@ -20,9 +21,8 @@ import time from pathlib import Path -from common import cli_version +from common import cli_version, INFRA_EXIT, TRANSIENT -INFRA_EXIT = 75 TOOLS = ["Bash", "Read", "Edit", "Write", "Glob", "Grep"] WEB_TOOLS = ["WebFetch", "WebSearch"] @@ -47,7 +47,8 @@ def main() -> int: (plugin / ".claude-plugin/plugin.json").write_text(json.dumps({"name": "bench", "version": "0.0.0", "description": "benchmark skills"})) for skill in skills: shutil.copytree(skill, plugin / "skills" / skill.name) - loaded.append(skill.name) + match = re.search(r"^name:\s*(\S+)", (skill / "SKILL.md").read_text(), re.M) # the name the harness checks, not the directory name + loaded.append(match.group(1) if match else skill.name) cmd += ["--plugin-dir", str(plugin)] claude = cmd[0] init: dict = {} @@ -85,8 +86,7 @@ def main() -> int: sys.stdout.write(final) sys.stderr.write(stderr) if errored or code != 0: - transient = any(s in stderr.lower() + final.lower() for s in ("overloaded", "rate limit", "529", "timed out")) - return INFRA_EXIT if transient else (code or 1) + return INFRA_EXIT if any(s in stderr.lower() + final.lower() for s in TRANSIENT) else (code or 1) return 0 diff --git a/benchmarks/agent/adapters/codex.py b/benchmarks/agent/adapters/codex.py index e27f620d..3c054749 100644 --- a/benchmarks/agent/adapters/codex.py +++ b/benchmarks/agent/adapters/codex.py @@ -13,7 +13,7 @@ from datetime import datetime from pathlib import Path -from common import cli_version +from common import cli_version, INFRA_EXIT, TRANSIENT def usage_row(usage: dict, timestamp: float, limit: int | None) -> dict: @@ -51,7 +51,7 @@ def session_usage(state: Path): def main() -> int: env = os.environ model = env["BENCH_MODEL"] - effort = env["BENCH_THINKING"] + effort = env.get("BENCH_THINKING") run = Path(env["BENCH_RUN_DIR"]) home = run / "codex-home" state = home / ".codex" @@ -66,7 +66,7 @@ def main() -> int: "--disable", "apps", "--disable", "plugins", "--disable", "remote_plugin", "--disable", "skill_mcp_dependency_install", "--sandbox", "danger-full-access", "-c", 'approval_policy="never"', - "-c", 'web_search="disabled"', "-c", f'model_reasoning_effort="{effort}"', "-m", model, "-"] + "-c", 'web_search="disabled"', *(["-c", f'model_reasoning_effort="{effort}"'] if effort else []), "-m", model, "-"] child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home), "CODEX_HOME": str(state)} prompt = sys.stdin.read() final, errors, summary, seen, servers, item_types = "", [], None, set(), set(), set() @@ -75,7 +75,10 @@ def main() -> int: process.stdin.write(prompt) process.stdin.close() for line in process.stdout: - event = json.loads(line) + try: + event = json.loads(line) + except json.JSONDecodeError: + continue transcript.write(json.dumps({"t": time.time(), **event}) + "\n") item = event.get("item") or {} item_types.add(item.get("type")) @@ -103,7 +106,10 @@ def main() -> int: loaded, builtin, observed = [], [], {} for path in state.glob("sessions/**/*.jsonl"): for line in path.read_text().splitlines(): - event = json.loads(line) + try: + event = json.loads(line) + except json.JSONDecodeError: + continue payload = event.get("payload") or {} if event["type"] == "turn_context": observed = {k: payload.get(k) for k in ("model", "effort")} @@ -120,7 +126,7 @@ def main() -> int: sys.stderr.write(stderr_text) failure = " ".join(errors) + stderr_text if code or errors: - return 75 if any(s in failure.lower() for s in ("rate limit", "usage limit", "429", "quota", "temporarily", "503")) else (code or 1) + return INFRA_EXIT if any(s in failure.lower() for s in TRANSIENT) else (code or 1) return 0 diff --git a/benchmarks/agent/adapters/common.py b/benchmarks/agent/adapters/common.py index bb0c3de2..74cd7268 100644 --- a/benchmarks/agent/adapters/common.py +++ b/benchmarks/agent/adapters/common.py @@ -4,6 +4,11 @@ import subprocess +INFRA_EXIT = 75 # EX_TEMPFAIL: a provider or infrastructure failure, not an agent failure; the harness retries the trial later +# Error text that means the provider, not the agent, failed. Shared so the classification cannot drift between adapters. +TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily", "usage limit", "quota", + "credits", "fetch failed", "websocket error", "connection error", "econnreset", "unavailable") + def cli_version(binary: str, env: dict | None = None) -> str | None: """First line of ` --version`, or None when the CLI does not answer.""" diff --git a/benchmarks/agent/adapters/devin.py b/benchmarks/agent/adapters/devin.py index c6d6644b..bdbd55f2 100755 --- a/benchmarks/agent/adapters/devin.py +++ b/benchmarks/agent/adapters/devin.py @@ -22,10 +22,7 @@ from datetime import datetime from pathlib import Path -from common import cli_version - -INFRA_EXIT = 75 -TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily", "usage limit", "quota", "insufficient credits", "fetch failed", "websocket error", "connection error") +from common import cli_version, INFRA_EXIT, TRANSIENT WEB_TOOLS = ["WebFetch", "WebSearch", "webfetch", "web_search"] MCP_TOOLS = ["mcp_call_tool", "mcp_list_tools", "mcp_list_servers", "mcp_read_resource"] # org-managed plugins install MCP servers NO_TOOL_CONFIG = {"claude": False, "cursor": False, "windsurf": False, "codex": False} diff --git a/benchmarks/agent/adapters/direct.py b/benchmarks/agent/adapters/direct.py index a0d1ddd0..dca4e840 100644 --- a/benchmarks/agent/adapters/direct.py +++ b/benchmarks/agent/adapters/direct.py @@ -5,7 +5,7 @@ ANTHROPIC_BASE_URL) or `openai/` (OPENAI_API_KEY, optional OPENAI_BASE_URL, so any OpenAI-compatible endpoint works). The model gets one `bash` tool that runs in the workspace on the shimmed PATH, plus `fetch` only when knowledge is `web`. The system prompt lists the installed skills by name and description, and the model reads their files itself. -The loop, its limits, and the tool set are fixed here so every model faces the same protocol; they are part of the result identity. +The loop, its limits, and the tool set are fixed here so every model faces the same protocol; the recorded harness commit identifies them. It does not sandbox the network: pair it with the harness --canary-cmd. Exit 75 marks a provider or infrastructure failure. """ @@ -22,7 +22,8 @@ import urllib.request from pathlib import Path -INFRA_EXIT = 75 +from common import INFRA_EXIT + MAX_TURNS = 60 COMMAND_SECONDS = 120 OUTPUT_CHARS = 20_000 diff --git a/benchmarks/agent/adapters/grok.py b/benchmarks/agent/adapters/grok.py index 04d5c85b..a216f7bb 100644 --- a/benchmarks/agent/adapters/grok.py +++ b/benchmarks/agent/adapters/grok.py @@ -18,10 +18,7 @@ import time from pathlib import Path -from common import cli_version - -INFRA_EXIT = 75 -TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily", "usage limit", "quota", "insufficient credits", "connection error", "unavailable") +from common import cli_version, INFRA_EXIT, TRANSIENT def usage_row(usage: dict, limit: int | None, now: float) -> dict: diff --git a/benchmarks/agent/adapters/opencode.py b/benchmarks/agent/adapters/opencode.py index b3a7cbf5..62c71683 100644 --- a/benchmarks/agent/adapters/opencode.py +++ b/benchmarks/agent/adapters/opencode.py @@ -18,10 +18,7 @@ import time from pathlib import Path -from common import cli_version - -INFRA_EXIT = 75 -TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily", "usage limit", "quota", "insufficient credits", "fetch failed", "connection error", "econnreset") +from common import cli_version, INFRA_EXIT, TRANSIENT WEB_PERMISSIONS = {"webfetch": "deny", "websearch": "deny"} @@ -79,8 +76,8 @@ def main() -> int: usage.write(json.dumps(usage_row(part["tokens"], now)) + "\n") elif event["type"] == "text": final = part.get("text") or final - elif event["type"] == "tool_use": - tools.add(part.get("tool")) + elif event["type"] == "tool_use" and part.get("tool"): + tools.add(part["tool"]) elif event["type"] == "error": error = json.dumps(event.get("error")) stderr = proc.stderr.read() diff --git a/benchmarks/agent/adapters/pi.py b/benchmarks/agent/adapters/pi.py index 9f3640cf..26545dce 100755 --- a/benchmarks/agent/adapters/pi.py +++ b/benchmarks/agent/adapters/pi.py @@ -21,10 +21,7 @@ import time from pathlib import Path -from common import cli_version - -INFRA_EXIT = 75 -TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily", "usage limit", "quota", "insufficient credits", "fetch failed", "websocket error", "connection error") +from common import cli_version, INFRA_EXIT, TRANSIENT SCALE = {"K": 1_000, "M": 1_000_000} @@ -33,7 +30,10 @@ def context_limit(pi: str, model: str, env: dict, extensions: list[str]) -> int command = [pi, "--no-extensions"] for extension in filter(None, extensions): command += ["-e", extension] - listing = subprocess.run([*command, "--list-models", model.split("/")[-1]], env=env, capture_output=True, text=True).stdout + try: + listing = subprocess.run([*command, "--list-models", model.split("/")[-1]], env=env, capture_output=True, text=True, timeout=30).stdout + except (OSError, subprocess.TimeoutExpired): + return None for line in listing.splitlines(): cols = line.split() if len(cols) > 2 and cols[0] == model.split("/")[0] and cols[1] == model.split("/")[-1]: diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 8975c8ec..f5d35686 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -4,7 +4,6 @@ from __future__ import annotations import argparse -import functools import hashlib import json import os @@ -227,35 +226,6 @@ def canaries(cell: dict, env: dict, workspace: Path, args: argparse.Namespace) - } -def sync_credentials_back(pairs: list[tuple[str, str]], run_dir: Path, home: Path | None = None) -> list[str]: - """Copy a login the agent's CLI refreshed inside its isolated home back to the real one. - - Adapters hand the CLI a copy of the user's OAuth login. Refresh tokens rotate, so a refresh in the copy that is then discarded - leaves the user's own login holding a dead token. Only a changed, newer copy is written back, atomically.""" - home = home or Path.home() - updated = [] - for real_rel, isolated_rel in pairs: - real, isolated = home / real_rel, run_dir / isolated_rel - if not isolated.is_file() or not real.is_file() or isolated.read_bytes() == real.read_bytes() or isolated.stat().st_mtime <= real.stat().st_mtime: - continue - temporary = real.with_name(f".{real.name}.bench-sync") - temporary.write_bytes(isolated.read_bytes()) - temporary.chmod(real.stat().st_mode & 0o777) - os.replace(temporary, real) - updated.append(real_rel) - return updated - - -CREDENTIALS = { # (real file relative to the home directory, its isolated copy relative to the run directory) per adapter - "codex": [(".codex/auth.json", "codex-home/.codex/auth.json")], - "pi": [(".pi/agent/auth.json", "pi-home/.pi/agent/auth.json"), (".pi/agent/antigravity-accounts.json", "pi-home/.pi/agent/antigravity-accounts.json")], - "devin": [(".local/share/devin/credentials.toml", "devin-home/.local/share/devin/credentials.toml")], - "agy": [(".gemini/antigravity-cli/antigravity-oauth-token", "agy-home/.gemini/antigravity-cli/antigravity-oauth-token")], - "opencode": [(".local/share/opencode/auth.json", "opencode-home/.local/share/opencode/auth.json")], - "grok": [(".grok/auth.json", "grok-home/auth.json")], -} - - def sync_credentials_back(pairs: list[tuple[str, str]], run_dir: Path, home: Path | None = None) -> list[str]: """Copy a login the agent's CLI refreshed inside its isolated home back to the real one. @@ -335,7 +305,12 @@ def run_agent(args: argparse.Namespace, env: dict, workspace: Path, prompt: str) os.killpg(proc.pid, signal.SIGKILL) except ProcessLookupError: pass - stdout, stderr = proc.communicate() + try: + stdout, stderr = proc.communicate(timeout=10) + except subprocess.TimeoutExpired: # a detached grandchild still holds the pipes; the adapter's own transcript has the record + proc.stdout.close() + proc.stderr.close() + stdout, stderr = "", "" return None, stdout, f"{stderr}\ntimeout" if stderr else "timeout" @@ -381,7 +356,6 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - start = time.monotonic() agent_exit, stdout, stderr = run_agent(args, env, workspace, prompt) sync_credentials_back(getattr(args, "credentials", []), out) - sync_credentials_back(getattr(args, "credentials", []), out) seconds = round(time.monotonic() - start, 1) snaps = snapshots.finish() if agent_exit == INFRA_EXIT and infra_retries < args.infra_retries: @@ -444,20 +418,19 @@ def snapshot_validity(scenario: dict, snaps: list[dict], wright: str, out: Path) return {"count": len(series), "firstValidIndex": next((s["i"] for s in series if s["valid"]), None), "regressions": regressions, "series": series} -@functools.lru_cache(maxsize=None) def harness_commit() -> str: proc = subprocess.run(["git", "-C", str(HERE), "rev-parse", "HEAD"], capture_output=True, text=True) dirty = subprocess.run(["git", "-C", str(HERE), "status", "--porcelain", "--", "."], capture_output=True, text=True).stdout.strip() return proc.stdout.strip() + ("+dirty" if dirty else "") -@functools.lru_cache(maxsize=None) def file_sha256(path: str) -> str: + """Re-hashed per call: a suite resumes for days in one process and the binary may be rebuilt between trials.""" return hashlib.sha256(Path(path).read_bytes()).hexdigest() -@functools.lru_cache(maxsize=None) -def cached_suite(scenarios: str) -> tuple: +def suite_identity(scenarios: str) -> tuple: + """Recomputed per call: the suite may change between trials of a long-running suite.""" return tuple(bench_grade.suite_identity(Path(scenarios)).items()) @@ -476,7 +449,7 @@ def base_result(scenario: dict, cell: dict, args: argparse.Namespace, out: Path, "wright": subprocess.run([args.wright, "--version"], capture_output=True, text=True).stdout.strip(), "wrightSha256": file_sha256(args.wright), "harness": harness_commit(), - "suite": dict(cached_suite(str(SCENARIOS))), + "suite": dict(suite_identity(str(SCENARIOS))), "timestamp": datetime.now(timezone.utc).isoformat(timespec="seconds"), "skills": {name: skill_identity(name, Path(args.skill_dirs[name])) for name in cell["skills"]}, **({"wiki": bench_wiki.identity(Path(args.wiki_dir))} if cell["knowledge"] == "wiki" else {}), @@ -579,6 +552,9 @@ def user_defaults() -> dict: unknown = set(config) - {"wright", "out", "wiki_dir", "env_pass", "deny_read", "allow_read", "skill_dirs", "models"} if unknown: raise SystemExit(f"{CONFIG_PATH}: unknown key(s) {sorted(unknown)}") + for i, entry in enumerate(config.get("models", [])): + if entry.get("adapter") not in ADAPTERS or not entry.get("model"): + raise SystemExit(f"{CONFIG_PATH}: models[{i}] needs an adapter in {sorted(ADAPTERS)} and a model") defaults = {k: v for k, v in config.items() if k not in ("skill_dirs", "out", "wiki_dir")} defaults.update({k: Path(v).expanduser() for k, v in config.items() if k in ("out", "wiki_dir")}) if "skill_dirs" in config: @@ -601,6 +577,10 @@ def preflight(args: argparse.Namespace, cells: list[dict]) -> list[str]: problems.append(f"skill '{name}' needs --skill-dir {name}=DIR (or skill_dirs in {CONFIG_PATH})") if any(c["tool"] == "overpy" for c in cells) and not bench_grade.oracle_available(): problems.append("tool 'overpy' needs the pinned oracle: run `agent_bench.py setup-oracle`") + if getattr(args, "file_sandbox", False) and (sys.platform != "darwin" or not shutil.which("sandbox-exec")): + problems.append("the file sandbox needs macOS sandbox-exec; pass --no-file-sandbox for an unprotected run") + if (credentials := getattr(args, "credentials", [])) and not (Path.home() / credentials[0][0]).is_file(): + problems.append(f"adapter '{args.adapter}' needs its login at ~/{credentials[0][0]} (sign in with that program first)") return problems @@ -625,12 +605,13 @@ def cmd_evaluate(args: argparse.Namespace) -> int: Needs no agent harness around it: isolation comes from the harness's own scrubbed environment, so it runs the same from a terminal or from inside another agent's shell.""" args.file_sandbox = not args.no_file_sandbox # evaluation hides the answer keys and the rest of the host from the agent by default - args.credentials = CREDENTIALS.get(args.adapter, []) + if args.deny_read and not args.file_sandbox: + raise SystemExit("--deny-read does nothing without the file sandbox") args.credentials = CREDENTIALS.get(args.adapter, []) args.allow_read = [*(str(Path.home() / rel) for rel in ADAPTER_READS[args.adapter]), *args.allow_read] script = Path(__file__).parent / "adapters" / ADAPTERS[args.adapter] effort = f"BENCH_THINKING={shlex.quote(args.effort)} " if args.effort else "" - agent_id = "-".join(filter(None, (args.adapter, args.model.replace("/", "_"), args.effort))) + agent_id = model_slug({"adapter": args.adapter, "model": args.model, "effort": args.effort}) cmd = f"BENCH_MODEL={shlex.quote(args.model)} {effort}{shlex.quote(sys.executable)} {shlex.quote(str(script))}" wanted = [CANONICAL_CELL] if args.cells == "score" else CONTROL_CELLS cells = [c for c in wanted if all(s in args.skill_dirs for s in c["skills"])] @@ -694,7 +675,9 @@ def cmd_suite(args: argparse.Namespace) -> int: print("\n" + "\n".join(f"{slug}: {state}" for slug, state in outcome.items())) if not args.dry_run: bench_leaderboard.main(sorted(root.glob("*/")), root / "leaderboard") - return 3 if any(state.startswith("waiting") for state in outcome.values()) else 0 + if any(state.startswith("waiting") for state in outcome.values()): + return 3 + return 0 if all(state == "done" for state in outcome.values()) else 1 def cmd_wiki_snapshot(args: argparse.Namespace) -> int: @@ -729,11 +712,12 @@ def main() -> int: p.add_argument("--no-ancestor-check", dest="check_ancestors", action="store_false", help="skip the check for instruction files above the workspace") p.add_argument("--canary-cmd", help="shell command that must fail in the agent environment when the network is 'off'") p.add_argument("--timeout", type=int, default=1800) - p.add_argument("--file-sandbox", action="store_true", help="macOS: restrict agent and descendant file writes to the trial directory, and hide the scenarios, other runs, the wiki, unselected skills, and --deny-read paths") - p.add_argument("--deny-read", nargs="*", default=[], metavar="PATH", help="extra directories hidden from the agent (the home directories and drives already are); needs --file-sandbox") + p.add_argument("--deny-read", nargs="*", default=[], metavar="PATH", help="extra directories hidden from the agent (the home directories and drives already are); needs the file sandbox") p.add_argument("--allow-read", nargs="*", default=[], metavar="PATH", help="paths the agent's CLI needs inside the hidden home directories (credentials, installation); `evaluate` adds its adapter's") p.add_argument("--infra-retries", type=int, default=2, help="retries when the agent exits 75 (provider or infrastructure failure)") p.add_argument("--infra-backoff", type=int, default=60, help="seconds before the first retry; each further retry waits one more multiple") + for p in (sub.choices["run"], sub.choices["matrix"]): + p.add_argument("--file-sandbox", action="store_true", help="macOS: restrict agent and descendant file writes to the trial directory, and hide the scenarios, other runs, the wiki, unselected skills, and --deny-read paths") run = sub.choices["run"] run.add_argument("scenario", choices=all_scenario_ids()) run.add_argument("--agent-cmd", required=True, help="shell command; the task prompt arrives on stdin, cwd is the workspace, BENCH_* describes the condition") diff --git a/benchmarks/agent/bench_leaderboard.py b/benchmarks/agent/bench_leaderboard.py index e28ad5f1..9eaacca7 100644 --- a/benchmarks/agent/bench_leaderboard.py +++ b/benchmarks/agent/bench_leaderboard.py @@ -12,16 +12,35 @@ import bench_report TRACKS = (("workshop", "Workshop"), ("opy", "OverPy")) -READING = ( - "The score is the share of tasks an agent completed with a valid, safe result, averaged over 8 realistic Workshop tasks per language, each tried 3 times.", - "The bar shows the score; the range in brackets is the 95% confidence interval. With 8 tasks the range is wide: agents marked **tied with top** cannot be told apart from the first row.", - "It is a reference for how an agent behaves with Wright and its guide, not a measure of general ability, and a model's score depends on the agent program that runs it.", -) -LIMITS = ( - "Network access was switched off by instruction only; the harness did not block it.", - "A run that hit the time limit counts as a failure. Runs cut off by provider outages are retried, not counted.", - "The suite has 8 test tasks per language, so differences of a few points mean nothing.", -) + + +def suite_shape(data: dict) -> tuple[int | None, int | None]: + """(scenarios per language, trials per scenario) when every run used the same suite shape, else (None, None).""" + counts = {(t["scenarios"], t["trials"]) for e in data["entries"] for t in e["tracks"].values()} + return next(iter(counts)) if len(counts) == 1 else (None, None) + + +def reading(data: dict) -> list[str]: + scenarios, trials = suite_shape(data) + tasks = f"{scenarios} realistic Workshop tasks" if scenarios else "the held-out tasks" + times = f"{trials} times" if trials else "a fixed number of times" + return [ + f"The score is the share of tasks an agent completed with a valid, safe result, averaged over {tasks} per language, each tried {times}.", + f"The bar shows the score; the range in brackets is the 95% confidence interval. With {scenarios or 'few'} tasks the range is wide: agents marked **tied with top** cannot be told apart from the first row.", + "It is a reference for how an agent behaves with Wright and its guide, not a measure of general ability, and a model's score depends on the agent program that runs it.", + ] + + +def limits(data: dict) -> list[str]: + scenarios, _ = suite_shape(data) + networks = {n for e in data["entries"] for n in e["network"]} + network = ("The harness verified that the network was unreachable on every run." if networks == {"canary-checked"} + else "Network access was switched off by instruction only for at least some runs; the harness did not block it.") + return [ + network, + "A run that hit the time limit counts as a failure. Runs cut off by provider outages are retried, not counted.", + f"The suite has {scenarios or 'few'} test tasks per language, so differences of a few points mean nothing.", + ] def load_entries(dirs: list[Path]) -> list[dict]: @@ -69,12 +88,15 @@ def build(dirs: list[Path]) -> dict: entries = load_entries(dirs) if not entries: return {"entries": [], "excluded": [], "environment": None, "date": date.today().isoformat()} + coverage = max(len(e["tracks"]) for e in entries) + covered, partial = [e for e in entries if len(e["tracks"]) == coverage], [e for e in entries if len(e["tracks"]) < coverage] key = lambda e: json.dumps(e["environment"], sort_keys=True) - main_key = Counter(key(e) for e in entries).most_common(1)[0][0] - ranked = sorted([e for e in entries if key(e) == main_key], key=mean_score, reverse=True) + main_key = Counter(key(e) for e in covered).most_common(1)[0][0] + ranked = sorted([e for e in covered if key(e) == main_key], key=mean_score, reverse=True) for entry in ranked: entry["standing"] = standing(entry, ranked[0]) - other = [{"run": e["run"], "reason": "made against a different Wright binary, skills, or task suite"} for e in entries if key(e) != main_key] + other = [{"run": e["run"], "reason": "made against a different Wright binary, skills, or task suite"} for e in covered if key(e) != main_key] + other += [{"run": e["run"], "reason": f"covers {len(e['tracks'])} of {coverage} language tracks"} for e in partial] return {"entries": ranked, "excluded": other, "environment": ranked[0]["environment"], "date": date.today().isoformat()} @@ -94,18 +116,19 @@ def markdown(data: dict) -> str: if not data["entries"]: return "\n".join(lines + ["No results yet.", ""]) lines += ["| # | Agent | Model | Effort | " + " | ".join(f"{n} score" for _, n in TRACKS) + " | Against the top |", "| --- | --- | --- | --- | " + " | ".join("---" for _ in TRACKS) + " | --- |"] + cell = lambda s: str(s if s is not None else "not recorded").replace("|", "\\|") # a raw | would break the table row for rank, e in enumerate(data["entries"], 1): cells = [] for lang, _ in TRACKS: t = e["tracks"].get(lang) cells.append(f"`{bar(t['score'])}` **{t['score']:.0f}** ({t['ci'][0]:.0f}–{t['ci'][1]:.0f})" + (" ⚠️" if t["provisional"] else "") if t else "n/a") - lines.append(f"| {rank} | {label(e)} | {e['model']} | {e['effort'] or 'default'} | " + " | ".join(cells) + f" | {e['standing']} |") - lines += ["", "## How to read this", "", *(f"- {s}" for s in READING), "", "## Limits", "", *(f"- {s}" for s in LIMITS)] + lines.append(f"| {rank} | {cell(label(e))} | {cell(e['model'])} | {cell(e['effort'] or 'default')} | " + " | ".join(cells) + f" | {e['standing']} |") + lines += ["", "## How to read this", "", *(f"- {s}" for s in reading(data)), "", "## Limits", "", *(f"- {s}" for s in limits(data))] if any(t["provisional"] for e in data["entries"] for t in e["tracks"].values()): lines.append("- ⚠️ marks a score that is provisional because the run did not cover every task or had unequal trials.") - skills = ", ".join(f"{k} `{v[:8]}`" for k, v in env["skills"].items()) or "none" + skills = ", ".join(f"{k} `{v[:8]}`" for k, v in (env["skills"] or {}).items()) or "none" commits = ", ".join(f"`{c[:8]}`" for c in sorted({c for e in data["entries"] for c in e["harnessCommit"]})) or "not recorded" - lines += ["", "## What was run", "", f"- Wright: {env['wright']} (sha256 `{env['wrightSha256'][:12]}`)", f"- Skills: {skills}", f"- Task suite: `{env['suite'][:12]}`", + lines += ["", "## What was run", "", f"- Wright: {env['wright'] or 'not recorded'} (sha256 `{(env['wrightSha256'] or 'not recorded')[:12]}`)", f"- Skills: {skills}", f"- Task suite: `{(env['suite'] or 'not recorded')[:12]}`", f"- Harness commit: {commits}", "- Scores come from the benchmark in `benchmarks/agent`; the run directories hold every result.json."] if data["excluded"]: lines += ["", "## Not comparable", "", *(f"- `{o['run']}`: {o['reason']}." for o in data["excluded"])] @@ -125,13 +148,14 @@ def page(data: dict) -> str: lo, hi = t["ci"] cells.append(f'
    ' f'{t["score"]:.0f} {lo:.0f}–{hi:.0f}{" ⚠️" if t["provisional"] else ""}') - rows.append(f'{rank}{esc(label(e))}{esc(str(e["model"]))}{esc(e["effort"] or "default")}{"".join(cells)}{esc(e["standing"])}') + rows.append(f'{rank}{esc(label(e))}{esc(str(e["model"] or "not recorded"))}{esc(e["effort"] or "default")}{"".join(cells)}{esc(e["standing"])}') env = data["environment"] or {} skills = ", ".join(f"{k} {v[:8]}" for k, v in (env.get("skills") or {}).items()) or "none" head = "".join(f"{n} score" for _, n in TRACKS) body = (f"{head}{''.join(rows)}
    #AgentModelEffortAgainst the top
    " if rows else "

    No results yet.

    ") li = lambda items: "".join(f"
  • {esc(s.replace('**', ''))}
  • " for s in items) + sha12 = lambda v: esc(str(v or "not recorded")[:12]) return f""" Wright Agent Score

    Wright Agent Score

    How well coding agents work on real Overwatch Workshop projects with Wright. Results of {esc(data.get("date", ""))}.

    {body} -

    How to read this

      {li(READING)}
    -

    Limits

      {li(LIMITS)}
    -

    What was run

    Wright {esc(str(env.get("wright", "")))} · sha256 {esc(str(env.get("wrightSha256", ""))[:12])} · skills {esc(skills)} · task suite {esc(str(env.get("suite", ""))[:12])}

    +

    How to read this

      {li(reading(data))}
    +

    Limits

      {li(limits(data))}
    +

    What was run

    Wright {esc(str(env.get("wright") or "not recorded"))} · sha256 {sha12(env.get("wrightSha256"))} · skills {esc(skills)} · task suite {sha12(env.get("suite"))}

    """ diff --git a/benchmarks/agent/bench_score.py b/benchmarks/agent/bench_score.py index 1868519c..b6054bf4 100644 --- a/benchmarks/agent/bench_score.py +++ b/benchmarks/agent/bench_score.py @@ -14,7 +14,7 @@ BOOTSTRAP_DRAWS = 10_000 BOOTSTRAP_SEED = 467 METHOD = f"two-stage percentile bootstrap (scenarios, then trials within a scenario), {BOOTSTRAP_DRAWS} draws, seed {BOOTSTRAP_SEED}" -EXCLUDED = ("provider-interrupted", "invalid", "agent-error") # reported separately; a timeout is the agent's own outcome and counts +VALID = ("completed", "timeout") # a timeout is the agent's own outcome and counts; every other status is excluded and reported separately TRACKS = {"workshop": "Wright Workshop Agent Score", "opy": "Wright OPY Agent Score"} @@ -56,7 +56,7 @@ def card(results: list[dict], language: str, expected: list[str]) -> dict: excluded = defaultdict(list) valid = [] for run in track: - (excluded[run["status"]] if run["status"] in EXCLUDED else valid).append(run) + (valid if run.get("status") in VALID else excluded[run.get("status") or "missing-status"]).append(run) if not valid: return {"contract": CONTRACT, "track": TRACKS[language], "refused": "no valid canonical test runs for this language"} identities = {json.dumps(identity_of(r), sort_keys=True) for r in valid} @@ -144,7 +144,11 @@ def compare(dirs: list[Path]) -> str: """One table from the score cards of several evaluation runs, with a warning when they were not made against the same Wright, skills, and suite.""" rows, identities = [], {} for directory in dirs: - for card_ in json.loads((directory / "score.json").read_text())["cards"]: + path = directory / "score.json" + if not path.is_file(): + rows.append(("", directory.name, "no score", "no score.json in this directory")) + continue + for card_ in json.loads(path.read_text())["cards"]: if "refused" in card_: rows.append((card_["track"], directory.name, "no score", card_["refused"])) continue diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index e5de3845..17fc252e 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -230,12 +230,21 @@ def test_preflight_names_what_is_missing_before_a_run_starts(self): self.assertEqual(len(problems), 3) self.assertTrue(any("wright binary not found" in p for p in problems) and any("`devin` is not on PATH" in p for p in problems) and any("--skill-dir wright-skill" in p for p in problems)) + def test_preflight_names_a_missing_adapter_login_and_an_unusable_sandbox(self): + args = argparse.Namespace(wright=str(Path(WRIGHT).resolve()), adapter="devin", skill_dirs={}, file_sandbox=True, + credentials=[(".missing-bench-cred/auth.json", "x"), (".also-missing/secondary.json", "y")]) + with patch.object(agent_bench.shutil, "which", side_effect=lambda b: f"/bin/{b}" if b != "sandbox-exec" else None): + problems = agent_bench.preflight(args, [agent_bench.normalize_cell({"tool": "wright", "skills": [], "knowledge": "none", "network": "off"})]) + self.assertTrue(any("sandbox-exec" in p for p in problems)) + self.assertTrue(any("~/.missing-bench-cred/auth.json" in p for p in problems)) + self.assertFalse(any("secondary.json" in p for p in problems)) # only the primary login is required + def test_read_policy_hides_the_host_and_allows_only_what_the_run_needs(self): root = self.out skills = {"wright-skill": root / "s1", "opy-skill": root / "s2"} args = argparse.Namespace(out=root / "run-a", out_root=root, wright=WRIGHT, skill_dirs=skills, wiki_dir=None, deny_read=[], allow_read=[str(root / "creds")], adapter="devin") hidden, allowed = agent_bench.read_policy(args, {"BENCH_RUN_DIR": str(root / "run-a" / "t"), "BENCH_SKILLS": "wright-skill"}) - self.assertTrue(all(p in hidden for p in (Path("/Users"), root.resolve(), agent_bench.HERE, agent_bench.HERE))) + self.assertTrue(all(p in hidden for p in (Path("/Users"), root.resolve(), agent_bench.HERE))) self.assertTrue(all(d.resolve() in hidden for d in agent_bench.DATA_ROOTS)) self.assertIn(skills["opy-skill"].resolve(), hidden) self.assertNotIn(skills["wright-skill"].resolve(), hidden) @@ -309,6 +318,9 @@ def evaluate(sub): with patch.object(agent_bench, "cmd_evaluate", evaluate): only = agent_bench.cmd_suite(argparse.Namespace(models=models, only=["devin"], out=self.out, suite_name="s2", dry_run=True)) self.assertEqual(only, 0) + with patch.object(agent_bench, "cmd_evaluate", return_value=1): + failed = agent_bench.cmd_suite(argparse.Namespace(models=[{"adapter": "devin", "model": "m"}], only=None, out=self.out, suite_name="s3", dry_run=True)) + self.assertEqual(failed, 1) # a model that finished with errors fails the suite, it does not pass silently def test_a_repeated_run_refuses_a_different_wright_binary(self): run = self.out / "r" diff --git a/benchmarks/agent/test_leaderboard.py b/benchmarks/agent/test_leaderboard.py index 56c6aa9e..f80ccbe8 100644 --- a/benchmarks/agent/test_leaderboard.py +++ b/benchmarks/agent/test_leaderboard.py @@ -36,6 +36,12 @@ def test_runs_against_a_different_wright_are_listed_as_not_comparable(self): self.assertEqual([o["run"] for o in data["excluded"]], ["other"]) self.assertIn("Not comparable", bench_leaderboard.markdown(data)) + def test_a_run_covering_fewer_tracks_is_not_ranked(self): + data = bench_leaderboard.build([self.write("full", 5), self.write("half", 8, tracks=("workshop",))]) + self.assertEqual([e["run"] for e in data["entries"]], ["full"]) + self.assertEqual([o["run"] for o in data["excluded"]], ["half"]) + self.assertIn("1 of 2 language tracks", bench_leaderboard.markdown(data)) + def test_markdown_and_page_give_the_headline_a_reader_needs(self): data = bench_leaderboard.build([self.write("only", 6)]) text, page = bench_leaderboard.markdown(data), bench_leaderboard.page(data) diff --git a/benchmarks/agent/wiki_skill.py b/benchmarks/agent/wiki_skill.py index 34689e4c..eeac33e7 100644 --- a/benchmarks/agent/wiki_skill.py +++ b/benchmarks/agent/wiki_skill.py @@ -146,7 +146,8 @@ def build(snapshot: Path, out: Path, catalog: dict, manifest: dict, upstream_sou def content_hash(skill_dir: Path) -> str: - return hashlib.sha256("".join(f"{p.relative_to(skill_dir)}{hashlib.sha256(p.read_bytes()).hexdigest()}" for p in sorted(skill_dir.rglob("*.md"))).encode()).hexdigest() + return hashlib.sha256("".join(f"{p.relative_to(skill_dir)}{hashlib.sha256(p.read_bytes()).hexdigest()}" for p in sorted(skill_dir.rglob("*")) + if p.is_file() and p.name != "BUILD.json").encode()).hexdigest() # the build record is written after the hash is taken def identity(skill_dir: Path) -> dict: diff --git a/docs/README.md b/docs/README.md index 24446a57..3e52e09e 100644 --- a/docs/README.md +++ b/docs/README.md @@ -41,6 +41,8 @@ Issue contract. transport mappings for coding agents and embedding consumers. - [Agent benchmark](agent-benchmark.md): product-level benchmark contract for general coding agents working with Wright. +- [Agent benchmark how-to](agent-benchmark-howto.md): run the benchmark and + publish the results page. - [Agent benchmark comparison spec](specs/SPEC-414-agent-benchmark-comparison.md): proposed multi-condition, multi-model benchmark and tool-call analysis. - [Language services & LSP](language-services.md): editor-neutral language diff --git a/docs/agent-benchmark-howto.md b/docs/agent-benchmark-howto.md index 48edee9f..52c10dad 100644 --- a/docs/agent-benchmark-howto.md +++ b/docs/agent-benchmark-howto.md @@ -102,4 +102,4 @@ The agent program runs the model, so the same model scores differently under dif | An agent exits at once with an auth error | its login expired; sign in again with that program | | An agent cannot start under the file sandbox | add the paths it needs: `--allow-read PATH` or `allow_read` in the config | -The agent only sees its own run directory, the condition's skills, the tool binaries, and what its program needs to start; everything else on the machine is hidden from it, so it cannot find the answers or other runs. +The agent only sees its own run directory, the condition's skills, the tool binaries, and what its program needs to start; the home directories, drives, and benchmark data are hidden from it, so it cannot find the answers or other runs. diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index 93a2f430..bf3cacc4 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -336,9 +336,9 @@ adds the baseline, `wright` without the skill, and, for OverPy scenarios, the `overpy` controls (cells whose skill has no `--skill-dir` are skipped). It runs the same from a terminal or from inside another agent's shell, because isolation comes from the harness's scrubbed environment, not from its parent. It runs locally; CI -does not run it. Adapters for CLIs that keep credentials in the home directory (devin, -codex, opencode, grok, agy) need `--env-pass HOME` so they can copy them into their -isolated home. +does not run it. `evaluate` passes `HOME` through and the adapter copies the +credentials it needs into an isolated home; preflight names the missing login when +one is absent. `--adapter direct` is the built-in loop (`adapters/direct.py`) that needs no agent harness: it calls a model API with one `bash` tool (and `fetch` only for From 13cf1fe6a2042d0df8f8eb4d844fb80a63e1143a Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sat, 3 Oct 2026 20:53:05 +0800 Subject: [PATCH 15/32] fix(bench): keep agent text out of the provider-failure classification The second review found that TRANSIENT was matched against the agent's own output in claude_code, devin, and grok: a failed run whose message mentioned "timeout" or "quota" would exit 75, be recorded as provider-interrupted, and be excluded from the score, inflating it. Only provider channels (stderr, structured error fields) now classify. - devin.transient takes stdout and stderr separately; the anchored model-catalog error still counts on either stream - grok and pi classify on the structured error field only - claude_code checks stderr only and names its plugin dir predictably - every file an adapter unconditionally reads is in CREDENTIALS, so preflight names it; pi's optional antigravity login moves to OPTIONAL_CREDENTIALS - stream parsers skip non-dict JSON and use .get() throughout; codex throttles its per-line session rescan - matrix config options can no longer inject a silent deny_read - result loading drops records missing the fields renderers need; score cards sort mixed recorded/unrecorded enforcement values - the HTML page lists non-comparable runs like the Markdown does --- benchmarks/agent/adapters/agy.py | 10 ++++--- benchmarks/agent/adapters/claude_code.py | 12 ++++---- benchmarks/agent/adapters/codex.py | 38 ++++++++++++++++-------- benchmarks/agent/adapters/common.py | 4 +-- benchmarks/agent/adapters/devin.py | 10 +++---- benchmarks/agent/adapters/direct.py | 2 +- benchmarks/agent/adapters/grok.py | 15 ++++++---- benchmarks/agent/adapters/opencode.py | 12 ++++---- benchmarks/agent/adapters/pi.py | 12 ++++---- benchmarks/agent/agent_bench.py | 35 ++++++++++++---------- benchmarks/agent/bench_leaderboard.py | 5 ++-- benchmarks/agent/bench_report.py | 2 +- benchmarks/agent/bench_score.py | 6 ++-- benchmarks/agent/test_adapters.py | 9 ++++-- benchmarks/agent/test_agent_bench.py | 8 ++--- 15 files changed, 106 insertions(+), 74 deletions(-) diff --git a/benchmarks/agent/adapters/agy.py b/benchmarks/agent/adapters/agy.py index 17775d4e..644a9c79 100644 --- a/benchmarks/agent/adapters/agy.py +++ b/benchmarks/agent/adapters/agy.py @@ -52,10 +52,12 @@ def main() -> int: event = json.loads(line) except json.JSONDecodeError: continue + if not isinstance(event, dict): + continue now = time.time() transcript.write(json.dumps({"t": now, **event}) + "\n") - if event["event"] == "init": - init = event["init"] + if event.get("event") == "init": + init = event.get("init") or {} step = event.get("step_update") or {} tool = step.get("tool_name", "") if env["BENCH_NETWORK"] == "off" and tool in {"search_web", "read_url_content", "browser_subagent", "open_browser_url"}: @@ -65,8 +67,8 @@ def main() -> int: if step.get("state") == "DONE" and step.get("usage") and step["step_index"] not in seen: seen.add(step["step_index"]) usage.write(json.dumps(usage_row(step["usage"], now)) + "\n") - if event["event"] == "result": - result = event["result"] + if event.get("event") == "result": + result = event.get("result") or {} code = process.wait() Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"installed": installed, "audit": "isolated-home; CLI does not export loaded skill context", "unexpected": sorted(unexpected)})) Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "agy", "version": cli_version(binary), "model": env["BENCH_MODEL"], "effort": effort, "tools": None, "note": "the CLI does not expose its tool list"}, indent=2)) diff --git a/benchmarks/agent/adapters/claude_code.py b/benchmarks/agent/adapters/claude_code.py index 8d83afb0..ce3189e6 100755 --- a/benchmarks/agent/adapters/claude_code.py +++ b/benchmarks/agent/adapters/claude_code.py @@ -17,7 +17,6 @@ import shutil import subprocess import sys -import tempfile import time from pathlib import Path @@ -42,8 +41,9 @@ def main() -> int: loaded: list[str] = [] skills = [Path(p) for p in env.get("BENCH_SKILL_DIRS", "").split(os.pathsep) if p] if skills: - plugin = Path(tempfile.mkdtemp(dir=env["BENCH_RUN_DIR"])) - (plugin / ".claude-plugin").mkdir() + plugin = Path(env["BENCH_RUN_DIR"]) / "claude-plugin" # a fixed name: an infra retry would else orphan a mkdtemp dir + shutil.rmtree(plugin, ignore_errors=True) + (plugin / ".claude-plugin").mkdir(parents=True) (plugin / ".claude-plugin/plugin.json").write_text(json.dumps({"name": "bench", "version": "0.0.0", "description": "benchmark skills"})) for skill in skills: shutil.copytree(skill, plugin / "skills" / skill.name) @@ -65,11 +65,13 @@ def main() -> int: event = json.loads(line) except json.JSONDecodeError: continue + if not isinstance(event, dict): + continue transcript.write(json.dumps({"t": time.time(), **event}) + "\n") if event.get("type") == "system" and event.get("subtype") == "init": init = event if event.get("type") == "assistant": - u = event["message"].get("usage") or {} + u = (event.get("message") or {}).get("usage") or {} cached = u.get("cache_read_input_tokens") or 0 written = u.get("cache_creation_input_tokens") or 0 usage.write(json.dumps({ @@ -86,7 +88,7 @@ def main() -> int: sys.stdout.write(final) sys.stderr.write(stderr) if errored or code != 0: - return INFRA_EXIT if any(s in stderr.lower() + final.lower() for s in TRANSIENT) else (code or 1) + return INFRA_EXIT if any(s in stderr.lower() for s in TRANSIENT) else (code or 1) # `final` is the agent's own text; only provider channels classify return 0 diff --git a/benchmarks/agent/adapters/codex.py b/benchmarks/agent/adapters/codex.py index 3c054749..6c0557fb 100644 --- a/benchmarks/agent/adapters/codex.py +++ b/benchmarks/agent/adapters/codex.py @@ -42,10 +42,16 @@ def session_usage(state: Path): event = json.loads(line) except json.JSONDecodeError: continue + if not isinstance(event, dict): + continue payload = event.get("payload") or {} - info = payload.get("info") - if payload.get("type") == "token_count" and info: - yield json.dumps(info["total_token_usage"], sort_keys=True), usage_row(info["last_token_usage"], datetime.fromisoformat(event["timestamp"]).timestamp(), info.get("model_context_window")) + info = payload.get("info") or {} + if payload.get("type") == "token_count" and info.get("total_token_usage") and info.get("last_token_usage"): + try: + stamp = datetime.fromisoformat(event.get("timestamp") or "").timestamp() + except ValueError: + stamp = time.time() + yield json.dumps(info["total_token_usage"], sort_keys=True), usage_row(info["last_token_usage"], stamp, info.get("model_context_window")) def main() -> int: @@ -69,7 +75,7 @@ def main() -> int: "-c", 'web_search="disabled"', *(["-c", f'model_reasoning_effort="{effort}"'] if effort else []), "-m", model, "-"] child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home), "CODEX_HOME": str(state)} prompt = sys.stdin.read() - final, errors, summary, seen, servers, item_types = "", [], None, set(), set(), set() + final, errors, summary, seen, servers, item_types, scanned = "", [], None, set(), set(), set(), 0.0 with (run / "codex-stderr.log").open("w") as stderr, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript, open(env["BENCH_USAGE"], "w", buffering=1) as usage: process = subprocess.Popen(command, env=child_env, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=stderr, text=True) process.stdin.write(prompt) @@ -79,21 +85,25 @@ def main() -> int: event = json.loads(line) except json.JSONDecodeError: continue + if not isinstance(event, dict): + continue transcript.write(json.dumps({"t": time.time(), **event}) + "\n") - item = event.get("item") or {} + item, etype = event.get("item") or {}, event.get("type") item_types.add(item.get("type")) if item.get("type") == "mcp_tool_call": servers.add(item.get("server", "unknown")) if item.get("type") == "agent_message": final = item.get("text", final) - if event["type"] in ("error", "turn.failed"): + if etype in ("error", "turn.failed"): errors.append(str(event.get("message") or event.get("error"))) - if event["type"] == "turn.completed": + if etype == "turn.completed": summary = event.get("usage") - for key, row in session_usage(state): - if key not in seen: - seen.add(key) - usage.write(json.dumps(row) + "\n") + if time.time() - scanned > 1: # re-globbing every session file per line is quadratic on long sessions + scanned = time.time() + for key, row in session_usage(state): + if key not in seen: + seen.add(key) + usage.write(json.dumps(row) + "\n") code = process.wait() for key, row in session_usage(state): if key not in seen: @@ -110,10 +120,12 @@ def main() -> int: event = json.loads(line) except json.JSONDecodeError: continue + if not isinstance(event, dict): + continue payload = event.get("payload") or {} - if event["type"] == "turn_context": + if event.get("type") == "turn_context": observed = {k: payload.get(k) for k in ("model", "effort")} - if event["type"] == "response_item" and payload.get("role") == "developer": + if event.get("type") == "response_item" and payload.get("role") == "developer": for part in payload.get("content", []): names, builtins = loaded_skills(part.get("text", "")) loaded += names diff --git a/benchmarks/agent/adapters/common.py b/benchmarks/agent/adapters/common.py index 74cd7268..fc100810 100644 --- a/benchmarks/agent/adapters/common.py +++ b/benchmarks/agent/adapters/common.py @@ -6,8 +6,8 @@ INFRA_EXIT = 75 # EX_TEMPFAIL: a provider or infrastructure failure, not an agent failure; the harness retries the trial later # Error text that means the provider, not the agent, failed. Shared so the classification cannot drift between adapters. -TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily", "usage limit", "quota", - "credits", "fetch failed", "websocket error", "connection error", "econnreset", "unavailable") +TRANSIENT = ("rate limit", "overloaded", "429", "502", "503", "529", "bad gateway", "timed out", "timeout", "temporarily", + "usage limit", "quota", "credits", "fetch failed", "websocket error", "connection error", "econnreset", "unavailable") def cli_version(binary: str, env: dict | None = None) -> str | None: diff --git a/benchmarks/agent/adapters/devin.py b/benchmarks/agent/adapters/devin.py index bdbd55f2..b66ee21e 100755 --- a/benchmarks/agent/adapters/devin.py +++ b/benchmarks/agent/adapters/devin.py @@ -78,10 +78,10 @@ def parse_export(export: dict) -> tuple[list[dict], list[str], list[str]]: return rows, loaded, plugins -def transient(text: str) -> bool: - """A provider or infrastructure failure. An empty model catalog means the catalog could not be fetched, not that the model is wrong.""" - lowered = text.lower() - return any(s in lowered for s in TRANSIENT) or re.search(r"unknown model.*\navailable:\s*$", lowered.strip(), re.S) is not None +def transient(stdout: str, stderr: str) -> bool: + """A provider or infrastructure failure. The agent's own text on stdout can mimic provider wording, so only stderr and the + CLI's anchored model-catalog error count. An empty catalog means the catalog could not be fetched, not that the model is wrong.""" + return any(s in stderr.lower() for s in TRANSIENT) or re.search(r"unknown model.*\navailable:\s*$", (stdout + stderr).lower().strip(), re.S) is not None def main() -> int: @@ -113,7 +113,7 @@ def main() -> int: sys.stdout.write(proc.stdout) sys.stderr.write(proc.stderr) if proc.returncode != 0: - return INFRA_EXIT if transient(proc.stdout + proc.stderr) else proc.returncode + return INFRA_EXIT if transient(proc.stdout, proc.stderr) else proc.returncode return 0 diff --git a/benchmarks/agent/adapters/direct.py b/benchmarks/agent/adapters/direct.py index dca4e840..2932525f 100644 --- a/benchmarks/agent/adapters/direct.py +++ b/benchmarks/agent/adapters/direct.py @@ -138,7 +138,7 @@ def skill_listing(skills: list[Path]) -> tuple[str, list[str]]: target = Path(".agents/skills") / skill.name shutil.copytree(skill, target) text = (target / "SKILL.md").read_text() - name = re.search(r"^name:\s*(.+)$", text, re.M) + name = re.search(r"^name:\s*(\S+)", text, re.M) # same first-word rule as the harness's skill_name check description = re.search(r"^description:\s*(.+)$", text, re.M) names.append(name.group(1).strip() if name else skill.name) lines.append(f"- {names[-1]}: {description.group(1).strip() if description else ''} (read {target}/SKILL.md when relevant)") diff --git a/benchmarks/agent/adapters/grok.py b/benchmarks/agent/adapters/grok.py index a216f7bb..1ffc0eec 100644 --- a/benchmarks/agent/adapters/grok.py +++ b/benchmarks/agent/adapters/grok.py @@ -58,18 +58,21 @@ def main() -> int: event = json.loads(line) except json.JSONDecodeError: continue + if not isinstance(event, dict): + continue now = time.time() transcript.write(json.dumps({"t": now, **event}) + "\n") - if event["type"] == "system" and event.get("subtype") == "init": + message = event.get("message") or {} + if event.get("type") == "system" and event.get("subtype") == "init": init = event - elif event["type"] == "assistant": - pending.append(event["message"].get("usage") or {}) - text = "".join(b.get("text", "") for b in event["message"].get("content", []) if b.get("type") == "text") + elif event.get("type") == "assistant": + pending.append(message.get("usage") or {}) + text = "".join(b.get("text", "") for b in message.get("content", []) if b.get("type") == "text") final = text or final - elif event["type"] == "result": + elif event.get("type") == "result": final = event.get("result") or final if event.get("is_error"): - error = json.dumps(event.get("errors") or event.get("result") or "error") + error = json.dumps(event.get("errors") or "error") # only the structured error classifies; the result text is the agent's own limit = next((m.get("contextWindow") for m in (event.get("modelUsage") or {}).values()), None) for u in pending: # the context window is only known from the final result line usage.write(json.dumps(usage_row(u, limit, time.time())) + "\n") diff --git a/benchmarks/agent/adapters/opencode.py b/benchmarks/agent/adapters/opencode.py index 62c71683..8877b10d 100644 --- a/benchmarks/agent/adapters/opencode.py +++ b/benchmarks/agent/adapters/opencode.py @@ -70,15 +70,17 @@ def main() -> int: event = json.loads(line) except json.JSONDecodeError: continue - now, part = time.time(), event.get("part") or {} + if not isinstance(event, dict): + continue + now, part, etype = time.time(), event.get("part") or {}, event.get("type") transcript.write(json.dumps({"t": now, **event}) + "\n") - if event["type"] == "step_finish" and part.get("tokens"): + if etype == "step_finish" and part.get("tokens"): usage.write(json.dumps(usage_row(part["tokens"], now)) + "\n") - elif event["type"] == "text": + elif etype == "text": final = part.get("text") or final - elif event["type"] == "tool_use" and part.get("tool"): + elif etype == "tool_use" and part.get("tool"): tools.add(part["tool"]) - elif event["type"] == "error": + elif etype == "error": error = json.dumps(event.get("error")) stderr = proc.stderr.read() code = proc.wait() diff --git a/benchmarks/agent/adapters/pi.py b/benchmarks/agent/adapters/pi.py index 26545dce..7f6f483e 100755 --- a/benchmarks/agent/adapters/pi.py +++ b/benchmarks/agent/adapters/pi.py @@ -94,17 +94,19 @@ def main() -> int: event = json.loads(line) except json.JSONDecodeError: continue + if not isinstance(event, dict): + continue now = time.time() - message = event.get("message") or {} - if event["type"] != "message_update": + message, etype = event.get("message") or {}, event.get("type") + if etype != "message_update": transcript.write(json.dumps({"t": now, **event}) + "\n") - if event["type"] == "message_start" and message.get("role") == "system": + if etype == "message_start" and message.get("role") == "system": loaded = skill_names(message) - if event["type"] == "message_end" and message.get("role") == "assistant": + if etype == "message_end" and message.get("role") == "assistant": usage.write(json.dumps(usage_row(message, limit, now)) + "\n") final = message_text(message) or final if message.get("stopReason") == "error": - error = str(message.get("errorMessage") or message_text(message)) + error = str(message.get("errorMessage") or "error") # only the structured error classifies; message text is the agent's own else: error = "" stderr = proc.stderr.read() diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index f5d35686..f91dfc7b 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -216,14 +216,19 @@ def canaries(cell: dict, env: dict, workspace: Path, args: argparse.Namespace) - } -CREDENTIALS = { # (real file relative to the home directory, its isolated copy relative to the run directory) per adapter +CREDENTIALS = { # (real file relative to the home directory, its isolated copy relative to the run directory) per adapter; preflight requires every one "codex": [(".codex/auth.json", "codex-home/.codex/auth.json")], - "pi": [(".pi/agent/auth.json", "pi-home/.pi/agent/auth.json"), (".pi/agent/antigravity-accounts.json", "pi-home/.pi/agent/antigravity-accounts.json")], - "devin": [(".local/share/devin/credentials.toml", "devin-home/.local/share/devin/credentials.toml")], - "agy": [(".gemini/antigravity-cli/antigravity-oauth-token", "agy-home/.gemini/antigravity-cli/antigravity-oauth-token")], + "pi": [(".pi/agent/auth.json", "pi-home/.pi/agent/auth.json")], + "devin": [(".local/share/devin/credentials.toml", "devin-home/.local/share/devin/credentials.toml"), + (".config/devin/config.json", "devin-home/.config/devin/config.json")], + "agy": [(".gemini/antigravity-cli/antigravity-oauth-token", "agy-home/.gemini/antigravity-cli/antigravity-oauth-token"), + (".gemini/antigravity-cli/installation_id", "agy-home/.gemini/antigravity-cli/installation_id")], "opencode": [(".local/share/opencode/auth.json", "opencode-home/.local/share/opencode/auth.json")], "grok": [(".grok/auth.json", "grok-home/auth.json")], } +OPTIONAL_CREDENTIALS = { # synced back when present, but not every install needs them (pi's antigravity login) + "pi": [(".pi/agent/antigravity-accounts.json", "pi-home/.pi/agent/antigravity-accounts.json")], +} def sync_credentials_back(pairs: list[tuple[str, str]], run_dir: Path, home: Path | None = None) -> list[str]: @@ -429,10 +434,6 @@ def file_sha256(path: str) -> str: return hashlib.sha256(Path(path).read_bytes()).hexdigest() -def suite_identity(scenarios: str) -> tuple: - """Recomputed per call: the suite may change between trials of a long-running suite.""" - return tuple(bench_grade.suite_identity(Path(scenarios)).items()) - def base_result(scenario: dict, cell: dict, args: argparse.Namespace, out: Path, seconds: float, agent_exit: int | None) -> dict: return { @@ -449,7 +450,7 @@ def base_result(scenario: dict, cell: dict, args: argparse.Namespace, out: Path, "wright": subprocess.run([args.wright, "--version"], capture_output=True, text=True).stdout.strip(), "wrightSha256": file_sha256(args.wright), "harness": harness_commit(), - "suite": dict(suite_identity(str(SCENARIOS))), + "suite": bench_grade.suite_identity(Path(SCENARIOS)), # recomputed per trial: the suite may change during a days-long run "timestamp": datetime.now(timezone.utc).isoformat(timespec="seconds"), "skills": {name: skill_identity(name, Path(args.skill_dirs[name])) for name in cell["skills"]}, **({"wiki": bench_wiki.identity(Path(args.wiki_dir))} if cell["knowledge"] == "wiki" else {}), @@ -501,6 +502,10 @@ def cmd_matrix(args: argparse.Namespace) -> int: print(f"not applicable: {scenario_id} {label}", flush=True) state = {"streak": 0, "interrupted": 0, "unattempted": 0, "failed": 0} lock = threading.Lock() + options = {k: ({n: Path(v) for n, v in val.items()} if k == "skill_dirs" else Path(val) if k == "wiki_dir" and val else val) for k, val in config.get("options", {}).items()} + merged = {**vars(args), **options} + if merged.get("deny_read") and not merged.get("file_sandbox"): + raise SystemExit("config options: deny_read does nothing without file_sandbox") def work(job: tuple) -> None: scenario_id, agent, cell, trial = job @@ -512,8 +517,7 @@ def work(job: tuple) -> None: if state["streak"] >= STOP_AFTER_INTERRUPTIONS: state["unattempted"] += 1 return - options = {k: ({n: Path(v) for n, v in val.items()} if k == "skill_dirs" else Path(val) if k == "wiki_dir" and val else val) for k, val in config.get("options", {}).items()} - trial_args = argparse.Namespace(**{**vars(args), "agent_id": agent["id"], "agent_cmd": agent["cmd"], **options}) + trial_args = argparse.Namespace(**{**merged, "agent_id": agent["id"], "agent_cmd": agent["cmd"]}) result = run_trial(load_scenario(scenario_id), cell, trial_args, out) with lock: state["streak"] = state["streak"] + 1 if result["status"] == "provider-interrupted" else 0 @@ -553,7 +557,7 @@ def user_defaults() -> dict: if unknown: raise SystemExit(f"{CONFIG_PATH}: unknown key(s) {sorted(unknown)}") for i, entry in enumerate(config.get("models", [])): - if entry.get("adapter") not in ADAPTERS or not entry.get("model"): + if not isinstance(entry, dict) or entry.get("adapter") not in ADAPTERS or not entry.get("model"): raise SystemExit(f"{CONFIG_PATH}: models[{i}] needs an adapter in {sorted(ADAPTERS)} and a model") defaults = {k: v for k, v in config.items() if k not in ("skill_dirs", "out", "wiki_dir")} defaults.update({k: Path(v).expanduser() for k, v in config.items() if k in ("out", "wiki_dir")}) @@ -579,8 +583,9 @@ def preflight(args: argparse.Namespace, cells: list[dict]) -> list[str]: problems.append("tool 'overpy' needs the pinned oracle: run `agent_bench.py setup-oracle`") if getattr(args, "file_sandbox", False) and (sys.platform != "darwin" or not shutil.which("sandbox-exec")): problems.append("the file sandbox needs macOS sandbox-exec; pass --no-file-sandbox for an unprotected run") - if (credentials := getattr(args, "credentials", [])) and not (Path.home() / credentials[0][0]).is_file(): - problems.append(f"adapter '{args.adapter}' needs its login at ~/{credentials[0][0]} (sign in with that program first)") + for real_rel, _ in CREDENTIALS.get(getattr(args, "adapter", ""), []): + if not (Path.home() / real_rel).is_file(): + problems.append(f"adapter '{args.adapter}' needs ~/{real_rel} (sign in with that program first)") return problems @@ -607,7 +612,7 @@ def cmd_evaluate(args: argparse.Namespace) -> int: args.file_sandbox = not args.no_file_sandbox # evaluation hides the answer keys and the rest of the host from the agent by default if args.deny_read and not args.file_sandbox: raise SystemExit("--deny-read does nothing without the file sandbox") - args.credentials = CREDENTIALS.get(args.adapter, []) + args.credentials = CREDENTIALS.get(args.adapter, []) + OPTIONAL_CREDENTIALS.get(args.adapter, []) args.allow_read = [*(str(Path.home() / rel) for rel in ADAPTER_READS[args.adapter]), *args.allow_read] script = Path(__file__).parent / "adapters" / ADAPTERS[args.adapter] effort = f"BENCH_THINKING={shlex.quote(args.effort)} " if args.effort else "" diff --git a/benchmarks/agent/bench_leaderboard.py b/benchmarks/agent/bench_leaderboard.py index 9eaacca7..88681b72 100644 --- a/benchmarks/agent/bench_leaderboard.py +++ b/benchmarks/agent/bench_leaderboard.py @@ -155,6 +155,8 @@ def page(data: dict) -> str: body = (f"{head}{''.join(rows)}
    #AgentModelEffortAgainst the top
    " if rows else "

    No results yet.

    ") li = lambda items: "".join(f"
  • {esc(s.replace('**', ''))}
  • " for s in items) + excluded = ("

    Not comparable

      " + "".join(f"
    • {esc(o['run'])}: {esc(o['reason'])}
    • " for o in data["excluded"]) + "
    " if data["excluded"] else "") + prose = (f"

    How to read this

      {li(reading(data))}

    Limits

      {li(limits(data))}
    " if data["entries"] else "") sha12 = lambda v: esc(str(v or "not recorded")[:12]) return f""" Wright Agent Score @@ -170,8 +172,7 @@ def page(data: dict) -> str:

    Wright Agent Score

    How well coding agents work on real Overwatch Workshop projects with Wright. Results of {esc(data.get("date", ""))}.

    {body} -

    How to read this

      {li(reading(data))}
    -

    Limits

      {li(limits(data))}
    +{prose}{excluded}

    What was run

    Wright {esc(str(env.get("wright") or "not recorded"))} · sha256 {sha12(env.get("wrightSha256"))} · skills {esc(skills)} · task suite {sha12(env.get("suite"))}

    """ diff --git a/benchmarks/agent/bench_report.py b/benchmarks/agent/bench_report.py index 0aa2afee..f37f2105 100644 --- a/benchmarks/agent/bench_report.py +++ b/benchmarks/agent/bench_report.py @@ -32,7 +32,7 @@ def load(dirs: list[Path]) -> list[dict]: for base in dirs: for path in sorted(base.rglob("result.json")): result = json.loads(path.read_text()) - if str(result.get("contract", "")).startswith("wright-agent-bench/"): + if str(result.get("contract", "")).startswith("wright-agent-bench/") and all(k in result for k in ("status", "language", "condition", "scenario", "agent", "environment")): result["_dir"] = path.parent match = re.search(r"-(\d+)$", path.parent.name) result["_trial"] = int(match.group(1)) if match else 0 diff --git a/benchmarks/agent/bench_score.py b/benchmarks/agent/bench_score.py index b6054bf4..bb5aab63 100644 --- a/benchmarks/agent/bench_score.py +++ b/benchmarks/agent/bench_score.py @@ -52,7 +52,7 @@ def identity_of(run: dict) -> dict: def card(results: list[dict], language: str, expected: list[str]) -> dict: """The score card of one language track, or a refusal when the runs are not one comparable environment.""" - track = [r for r in results if r["language"] == language and r["condition"]["label"] == CANONICAL and r.get("split") == "test"] + track = [r for r in results if r.get("language") == language and (r.get("condition") or {}).get("label") == CANONICAL and r.get("split") == "test"] excluded = defaultdict(list) valid = [] for run in track: @@ -99,8 +99,8 @@ def card(results: list[dict], language: str, expected: list[str]) -> dict: "identity": identity, "suite": (valid[0]["environment"].get("suite") or {}), "harness": sorted({r["environment"].get("harness") for r in valid if r["environment"].get("harness")}), - "networkEnforcement": sorted({r.get("networkEnforcement") for r in valid}), - "fileWriteEnforcement": sorted({r.get("fileWriteEnforcement") for r in valid}), + "networkEnforcement": sorted({r.get("networkEnforcement") or "not recorded" for r in valid}), + "fileWriteEnforcement": sorted({r.get("fileWriteEnforcement") or "not recorded" for r in valid}), "exclusions": {status: len(runs) for status, runs in sorted(excluded.items())}, "perScenario": [{"scenario": s, "usable": sum(v), "valid": len(v), "rate": round(rates[s], 3)} for s, v in sorted(per.items())], "byFamily": {f: round(100 * sum(v) / len(v), 1) for f, v in sorted(families.items())}, diff --git a/benchmarks/agent/test_adapters.py b/benchmarks/agent/test_adapters.py index d842ad7f..8337b19e 100644 --- a/benchmarks/agent/test_adapters.py +++ b/benchmarks/agent/test_adapters.py @@ -107,9 +107,12 @@ def test_config_reads_no_other_tools_and_denies_web_unless_asked(self): class DevinTransientTest(unittest.TestCase): def test_empty_model_catalog_is_a_provider_failure_but_a_wrong_model_is_not(self): - self.assertTrue(devin.transient("Error: Unknown model: 'swe-2-max'\nAvailable:\n")) - self.assertFalse(devin.transient("Error: Unknown model: 'nope'\nAvailable:\n swe-2-max\n swe-2\n")) - self.assertTrue(devin.transient("429 rate limit")) + self.assertTrue(devin.transient("Error: Unknown model: 'swe-2-max'\nAvailable:\n", "")) + self.assertFalse(devin.transient("Error: Unknown model: 'nope'\nAvailable:\n swe-2-max\n swe-2\n", "")) + self.assertTrue(devin.transient("", "429 rate limit")) + + def test_agent_text_on_stdout_does_not_classify_as_a_provider_failure(self): + self.assertFalse(devin.transient("I could not finish: my test run timed out and the quota for retries is used up", "")) class DevinEffortTest(unittest.TestCase): diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 17fc252e..982e8bbd 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -231,13 +231,13 @@ def test_preflight_names_what_is_missing_before_a_run_starts(self): self.assertTrue(any("wright binary not found" in p for p in problems) and any("`devin` is not on PATH" in p for p in problems) and any("--skill-dir wright-skill" in p for p in problems)) def test_preflight_names_a_missing_adapter_login_and_an_unusable_sandbox(self): - args = argparse.Namespace(wright=str(Path(WRIGHT).resolve()), adapter="devin", skill_dirs={}, file_sandbox=True, - credentials=[(".missing-bench-cred/auth.json", "x"), (".also-missing/secondary.json", "y")]) - with patch.object(agent_bench.shutil, "which", side_effect=lambda b: f"/bin/{b}" if b != "sandbox-exec" else None): + args = argparse.Namespace(wright=str(Path(WRIGHT).resolve()), adapter="grok", skill_dirs={}, file_sandbox=True, credentials=[]) + creds = {"grok": [(".missing-bench-cred/auth.json", "x"), (".missing-bench-cred/secondary", "y")]} + with patch.object(agent_bench.shutil, "which", side_effect=lambda b: f"/bin/{b}" if b != "sandbox-exec" else None), patch.dict(agent_bench.CREDENTIALS, creds, clear=True): problems = agent_bench.preflight(args, [agent_bench.normalize_cell({"tool": "wright", "skills": [], "knowledge": "none", "network": "off"})]) self.assertTrue(any("sandbox-exec" in p for p in problems)) self.assertTrue(any("~/.missing-bench-cred/auth.json" in p for p in problems)) - self.assertFalse(any("secondary.json" in p for p in problems)) # only the primary login is required + self.assertTrue(any("~/.missing-bench-cred/secondary" in p for p in problems)) # every file the adapter needs is named def test_read_policy_hides_the_host_and_allows_only_what_the_run_needs(self): root = self.out From 0112e6ddb0b4c930506c0e4b61b844eef5bd7e87 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sat, 3 Oct 2026 21:06:26 +0800 Subject: [PATCH 16/32] fix(bench): finish hardening adapter stream parsing and retry cleanup Follow-up to the previous fix, from the same review pass: nested event fields could still be truthy non-dicts (a str is truthy but has no .get), one direct index survived (agy step_index), and a retried devin attempt could read the previous attempt's export file. - add as_dict() in adapters/common and use it for every nested event container the stream loops dereference - clear devin-export.json between infrastructure retries - tighten the devin catalog regex to one line and check each stream's tail independently --- benchmarks/agent/adapters/agy.py | 10 +++++----- benchmarks/agent/adapters/claude_code.py | 7 ++++--- benchmarks/agent/adapters/codex.py | 10 +++++----- benchmarks/agent/adapters/common.py | 5 +++++ benchmarks/agent/adapters/devin.py | 4 +++- benchmarks/agent/adapters/grok.py | 8 ++++---- benchmarks/agent/adapters/opencode.py | 4 ++-- benchmarks/agent/adapters/pi.py | 4 ++-- benchmarks/agent/agent_bench.py | 1 + 9 files changed, 31 insertions(+), 22 deletions(-) diff --git a/benchmarks/agent/adapters/agy.py b/benchmarks/agent/adapters/agy.py index 644a9c79..60e1164a 100644 --- a/benchmarks/agent/adapters/agy.py +++ b/benchmarks/agent/adapters/agy.py @@ -11,7 +11,7 @@ import time from pathlib import Path -from common import cli_version, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, INFRA_EXIT, TRANSIENT def usage_row(usage: dict, timestamp: float) -> dict: @@ -58,17 +58,17 @@ def main() -> int: transcript.write(json.dumps({"t": now, **event}) + "\n") if event.get("event") == "init": init = event.get("init") or {} - step = event.get("step_update") or {} + step = as_dict(event.get("step_update")) tool = step.get("tool_name", "") if env["BENCH_NETWORK"] == "off" and tool in {"search_web", "read_url_content", "browser_subagent", "open_browser_url"}: unexpected.add("network-tool:" + tool) process.terminate() - if step.get("state") == "DONE" and step.get("usage") and step["step_index"] not in seen: - seen.add(step["step_index"]) + if step.get("state") == "DONE" and step.get("usage") and (index := step.get("step_index")) is not None and index not in seen: + seen.add(index) usage.write(json.dumps(usage_row(step["usage"], now)) + "\n") if event.get("event") == "result": - result = event.get("result") or {} + result = as_dict(event.get("result")) code = process.wait() Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"installed": installed, "audit": "isolated-home; CLI does not export loaded skill context", "unexpected": sorted(unexpected)})) Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "agy", "version": cli_version(binary), "model": env["BENCH_MODEL"], "effort": effort, "tools": None, "note": "the CLI does not expose its tool list"}, indent=2)) diff --git a/benchmarks/agent/adapters/claude_code.py b/benchmarks/agent/adapters/claude_code.py index ce3189e6..cb415213 100755 --- a/benchmarks/agent/adapters/claude_code.py +++ b/benchmarks/agent/adapters/claude_code.py @@ -20,7 +20,7 @@ import time from pathlib import Path -from common import cli_version, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, INFRA_EXIT, TRANSIENT TOOLS = ["Bash", "Read", "Edit", "Write", "Glob", "Grep"] WEB_TOOLS = ["WebFetch", "WebSearch"] @@ -71,7 +71,7 @@ def main() -> int: if event.get("type") == "system" and event.get("subtype") == "init": init = event if event.get("type") == "assistant": - u = (event.get("message") or {}).get("usage") or {} + u = as_dict(event.get("message")).get("usage") or {} cached = u.get("cache_read_input_tokens") or 0 written = u.get("cache_creation_input_tokens") or 0 usage.write(json.dumps({ @@ -80,7 +80,8 @@ def main() -> int: "context": (u.get("input_tokens") or 0) + cached + written, "context_limit": None, }) + "\n") if event.get("type") == "result": - final, errored = event.get("result", ""), bool(event.get("is_error")) + result_value = event.get("result") + final, errored = (result_value if isinstance(result_value, str) else final), bool(event.get("is_error")) stderr = proc.stderr.read() code = proc.wait() Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded})) diff --git a/benchmarks/agent/adapters/codex.py b/benchmarks/agent/adapters/codex.py index 6c0557fb..427f2d12 100644 --- a/benchmarks/agent/adapters/codex.py +++ b/benchmarks/agent/adapters/codex.py @@ -13,7 +13,7 @@ from datetime import datetime from pathlib import Path -from common import cli_version, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, INFRA_EXIT, TRANSIENT def usage_row(usage: dict, timestamp: float, limit: int | None) -> dict: @@ -44,8 +44,8 @@ def session_usage(state: Path): continue if not isinstance(event, dict): continue - payload = event.get("payload") or {} - info = payload.get("info") or {} + payload = as_dict(event.get("payload")) + info = as_dict(payload.get("info")) if payload.get("type") == "token_count" and info.get("total_token_usage") and info.get("last_token_usage"): try: stamp = datetime.fromisoformat(event.get("timestamp") or "").timestamp() @@ -88,7 +88,7 @@ def main() -> int: if not isinstance(event, dict): continue transcript.write(json.dumps({"t": time.time(), **event}) + "\n") - item, etype = event.get("item") or {}, event.get("type") + item, etype = as_dict(event.get("item")), event.get("type") item_types.add(item.get("type")) if item.get("type") == "mcp_tool_call": servers.add(item.get("server", "unknown")) @@ -127,7 +127,7 @@ def main() -> int: observed = {k: payload.get(k) for k in ("model", "effort")} if event.get("type") == "response_item" and payload.get("role") == "developer": for part in payload.get("content", []): - names, builtins = loaded_skills(part.get("text", "")) + names, builtins = loaded_skills(as_dict(part).get("text", "")) loaded += names builtin += builtins Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": sorted(set(loaded) | {f"unexpected-mcp:{server}" for server in servers}), "builtinSkills": sorted(set(builtin))})) diff --git a/benchmarks/agent/adapters/common.py b/benchmarks/agent/adapters/common.py index fc100810..81fdf758 100644 --- a/benchmarks/agent/adapters/common.py +++ b/benchmarks/agent/adapters/common.py @@ -10,6 +10,11 @@ "usage limit", "quota", "credits", "fetch failed", "websocket error", "connection error", "econnreset", "unavailable") +def as_dict(value) -> dict: + """value when it is a dict, else {} — `or {}` alone does not guard a truthy non-dict from a malformed stream line.""" + return value if isinstance(value, dict) else {} + + def cli_version(binary: str, env: dict | None = None) -> str | None: """First line of ` --version`, or None when the CLI does not answer.""" try: diff --git a/benchmarks/agent/adapters/devin.py b/benchmarks/agent/adapters/devin.py index b66ee21e..0257436d 100755 --- a/benchmarks/agent/adapters/devin.py +++ b/benchmarks/agent/adapters/devin.py @@ -81,7 +81,9 @@ def parse_export(export: dict) -> tuple[list[dict], list[str], list[str]]: def transient(stdout: str, stderr: str) -> bool: """A provider or infrastructure failure. The agent's own text on stdout can mimic provider wording, so only stderr and the CLI's anchored model-catalog error count. An empty catalog means the catalog could not be fetched, not that the model is wrong.""" - return any(s in stderr.lower() for s in TRANSIENT) or re.search(r"unknown model.*\navailable:\s*$", (stdout + stderr).lower().strip(), re.S) is not None + catalog = r"unknown model[^\n]*\n\s*available:\s*$" + return (any(s in stderr.lower() for s in TRANSIENT) + or re.search(catalog, stdout.lower().strip()) is not None or re.search(catalog, stderr.lower().strip()) is not None) def main() -> int: diff --git a/benchmarks/agent/adapters/grok.py b/benchmarks/agent/adapters/grok.py index 1ffc0eec..d4fedb7f 100644 --- a/benchmarks/agent/adapters/grok.py +++ b/benchmarks/agent/adapters/grok.py @@ -18,7 +18,7 @@ import time from pathlib import Path -from common import cli_version, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, INFRA_EXIT, TRANSIENT def usage_row(usage: dict, limit: int | None, now: float) -> dict: @@ -62,18 +62,18 @@ def main() -> int: continue now = time.time() transcript.write(json.dumps({"t": now, **event}) + "\n") - message = event.get("message") or {} + message = as_dict(event.get("message")) if event.get("type") == "system" and event.get("subtype") == "init": init = event elif event.get("type") == "assistant": pending.append(message.get("usage") or {}) - text = "".join(b.get("text", "") for b in message.get("content", []) if b.get("type") == "text") + text = "".join(as_dict(b).get("text", "") for b in message.get("content", []) if isinstance(b, dict) and b.get("type") == "text") final = text or final elif event.get("type") == "result": final = event.get("result") or final if event.get("is_error"): error = json.dumps(event.get("errors") or "error") # only the structured error classifies; the result text is the agent's own - limit = next((m.get("contextWindow") for m in (event.get("modelUsage") or {}).values()), None) + limit = next((m.get("contextWindow") for m in as_dict(event.get("modelUsage")).values()), None) for u in pending: # the context window is only known from the final result line usage.write(json.dumps(usage_row(u, limit, time.time())) + "\n") stderr = proc.stderr.read() diff --git a/benchmarks/agent/adapters/opencode.py b/benchmarks/agent/adapters/opencode.py index 8877b10d..d49b2252 100644 --- a/benchmarks/agent/adapters/opencode.py +++ b/benchmarks/agent/adapters/opencode.py @@ -18,7 +18,7 @@ import time from pathlib import Path -from common import cli_version, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, INFRA_EXIT, TRANSIENT WEB_PERMISSIONS = {"webfetch": "deny", "websearch": "deny"} @@ -72,7 +72,7 @@ def main() -> int: continue if not isinstance(event, dict): continue - now, part, etype = time.time(), event.get("part") or {}, event.get("type") + now, part, etype = time.time(), as_dict(event.get("part")), event.get("type") transcript.write(json.dumps({"t": now, **event}) + "\n") if etype == "step_finish" and part.get("tokens"): usage.write(json.dumps(usage_row(part["tokens"], now)) + "\n") diff --git a/benchmarks/agent/adapters/pi.py b/benchmarks/agent/adapters/pi.py index 7f6f483e..2d93deab 100755 --- a/benchmarks/agent/adapters/pi.py +++ b/benchmarks/agent/adapters/pi.py @@ -21,7 +21,7 @@ import time from pathlib import Path -from common import cli_version, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, INFRA_EXIT, TRANSIENT SCALE = {"K": 1_000, "M": 1_000_000} @@ -97,7 +97,7 @@ def main() -> int: if not isinstance(event, dict): continue now = time.time() - message, etype = event.get("message") or {}, event.get("type") + message, etype = as_dict(event.get("message")), event.get("type") if etype != "message_update": transcript.write(json.dumps({"t": now, **event}) + "\n") if etype == "message_start" and message.get("role") == "system": diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index f91dfc7b..dea0bdba 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -343,6 +343,7 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - while True: shutil.rmtree(workspace, ignore_errors=True) for stale in ("home", "bin", "real", "tool-trace.jsonl", "tool-calls", "usage.jsonl", "transcript.jsonl", "context.json", "agent-info.json", "snapshots", "tmp", + "devin-export.json", # devin only writes this on success; a retry that produces none would parse the previous attempt's export *(child.name for child in out.iterdir() if child.name.endswith(("-home", "-user")))): # adapters keep their isolated homes here; a retry starts from none target = out / stale shutil.rmtree(target, ignore_errors=True) if target.is_dir() else target.unlink(missing_ok=True) From 6b45345aae561e26804e77cbe791b7690d3c3db5 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sat, 3 Oct 2026 21:16:53 +0800 Subject: [PATCH 17/32] test(bench): make the preflight count assertion host-independent --- benchmarks/agent/test_agent_bench.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 982e8bbd..383f5598 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -225,10 +225,10 @@ def test_user_defaults_are_read_from_the_config_file(self): def test_preflight_names_what_is_missing_before_a_run_starts(self): args = argparse.Namespace(wright=str(self.out / "nope"), adapter="devin", skill_dirs={}) - with patch.object(agent_bench.shutil, "which", return_value=None): + with patch.object(agent_bench.shutil, "which", return_value=None), patch.dict(agent_bench.CREDENTIALS, {"devin": [(".missing-bench-cred/auth.json", "x")]}): problems = agent_bench.preflight(args, [agent_bench.normalize_cell({"tool": "wright", "skills": ["wright-skill"], "knowledge": "none", "network": "off"})]) - self.assertEqual(len(problems), 3) - self.assertTrue(any("wright binary not found" in p for p in problems) and any("`devin` is not on PATH" in p for p in problems) and any("--skill-dir wright-skill" in p for p in problems)) + self.assertEqual(len(problems), 4) + self.assertTrue(any("wright binary not found" in p for p in problems) and any("`devin` is not on PATH" in p for p in problems) and any("--skill-dir wright-skill" in p for p in problems) and any("~/.missing-bench-cred/auth.json" in p for p in problems)) def test_preflight_names_a_missing_adapter_login_and_an_unusable_sandbox(self): args = argparse.Namespace(wright=str(Path(WRIGHT).resolve()), adapter="grok", skill_dirs={}, file_sandbox=True, credentials=[]) From 16e434feccdff55e9959b6f0fe582f0f36b0609f Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 01:20:13 +0800 Subject: [PATCH 18/32] fix(bench): treat an Antigravity dropped connection as a provider failure and read the effort from the model id --- benchmarks/agent/adapters/agy.py | 12 +++++++++--- benchmarks/agent/adapters/common.py | 11 +++++++++++ benchmarks/agent/adapters/devin.py | 13 +------------ benchmarks/agent/agent_bench.py | 3 ++- benchmarks/agent/test_adapters.py | 7 +++++++ benchmarks/agent/test_agent_bench.py | 4 ++++ 6 files changed, 34 insertions(+), 16 deletions(-) diff --git a/benchmarks/agent/adapters/agy.py b/benchmarks/agent/adapters/agy.py index 60e1164a..1f84a75f 100644 --- a/benchmarks/agent/adapters/agy.py +++ b/benchmarks/agent/adapters/agy.py @@ -11,7 +11,13 @@ import time from pathlib import Path -from common import as_dict, cli_version, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, split_effort, INFRA_EXIT, TRANSIENT + + +def transient(error: str) -> bool: + """A provider failure. The CLI reports a dropped connection as `API error (attempt N): request failed: ... EOF`, even after the answer was written.""" + lowered = error.lower() + return any(s in lowered for s in TRANSIENT) or "api error (attempt" in lowered or "request failed" in lowered def usage_row(usage: dict, timestamp: float) -> dict: @@ -71,7 +77,7 @@ def main() -> int: result = as_dict(event.get("result")) code = process.wait() Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"installed": installed, "audit": "isolated-home; CLI does not export loaded skill context", "unexpected": sorted(unexpected)})) - Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "agy", "version": cli_version(binary), "model": env["BENCH_MODEL"], "effort": effort, "tools": None, "note": "the CLI does not expose its tool list"}, indent=2)) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "agy", "version": cli_version(binary), "model": split_effort(env["BENCH_MODEL"])[0], "modelId": env["BENCH_MODEL"], "effort": effort or split_effort(env["BENCH_MODEL"])[1], "tools": None, "note": "the CLI does not expose its tool list"}, indent=2)) (run / "adapter.json").write_text(json.dumps({"agent": "agy", "requestedModel": env["BENCH_MODEL"], "requestedEffort": effort, "observedModel": init.get("model"), "status": result.get("status"), "usageSource": "stream-step-usage", "providerUsage": result.get("usage")}, indent=2)) sys.stdout.write(result.get("response", "")) stderr_text = (run / "agy-stderr.log").read_text() @@ -80,7 +86,7 @@ def main() -> int: if "no output produced" in stderr_text and "headless" in stderr_text: return INFRA_EXIT if code or result.get("status") != "SUCCESS": - return INFRA_EXIT if any(s in error.lower() for s in TRANSIENT) else (code or 1) + return INFRA_EXIT if transient(error) else (code or 1) return 0 diff --git a/benchmarks/agent/adapters/common.py b/benchmarks/agent/adapters/common.py index 81fdf758..85923b7b 100644 --- a/benchmarks/agent/adapters/common.py +++ b/benchmarks/agent/adapters/common.py @@ -22,3 +22,14 @@ def cli_version(binary: str, env: dict | None = None) -> str | None: except (OSError, subprocess.TimeoutExpired): return None return (done.stdout or done.stderr).strip().splitlines()[0] if (done.stdout or done.stderr).strip() else None + + +EFFORTS = ("low", "medium", "high", "xhigh", "max") + + +def split_effort(model_id: str) -> tuple[str, str | None]: + """Some CLIs name the effort in the model id (`swe-2-max`, `gemini-3.8-flash-high`) and never report it separately.""" + for effort in EFFORTS: + if model_id.endswith(f"-{effort}"): + return model_id[: -len(effort) - 1], effort + return model_id, None diff --git a/benchmarks/agent/adapters/devin.py b/benchmarks/agent/adapters/devin.py index 0257436d..964de476 100755 --- a/benchmarks/agent/adapters/devin.py +++ b/benchmarks/agent/adapters/devin.py @@ -22,23 +22,12 @@ from datetime import datetime from pathlib import Path -from common import cli_version, INFRA_EXIT, TRANSIENT +from common import cli_version, split_effort, INFRA_EXIT, TRANSIENT WEB_TOOLS = ["WebFetch", "WebSearch", "webfetch", "web_search"] MCP_TOOLS = ["mcp_call_tool", "mcp_list_tools", "mcp_list_servers", "mcp_read_resource"] # org-managed plugins install MCP servers NO_TOOL_CONFIG = {"claude": False, "cursor": False, "windsurf": False, "codex": False} -EFFORTS = ("low", "medium", "high", "xhigh", "max") - - -def split_effort(model_id: str) -> tuple[str, str | None]: - """Devin names the effort in the model id (`swe-2-max` is model `swe-2` at effort `max`), so the CLI never reports it separately.""" - for effort in EFFORTS: - if model_id.endswith(f"-{effort}"): - return model_id[: -len(effort) - 1], effort - return model_id, None - - def isolated_config(user_config: dict, model: str, web: bool) -> dict: config = copy.deepcopy(user_config) config["read_config_from"] = NO_TOOL_CONFIG diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index dea0bdba..851635b3 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -658,7 +658,8 @@ def cmd_evaluate(args: argparse.Namespace) -> int: def model_slug(entry: dict) -> str: - return "-".join(filter(None, (entry["adapter"], entry["model"].replace("/", "_"), entry.get("effort")))) + effort = entry.get("effort") + return "-".join(filter(None, (entry["adapter"], entry["model"].replace("/", "_"), None if effort and entry["model"].endswith(f"-{effort}") else effort))) # an effort the model id names is not repeated def cmd_suite(args: argparse.Namespace) -> int: diff --git a/benchmarks/agent/test_adapters.py b/benchmarks/agent/test_adapters.py index 8337b19e..e97b69a7 100644 --- a/benchmarks/agent/test_adapters.py +++ b/benchmarks/agent/test_adapters.py @@ -115,6 +115,13 @@ def test_agent_text_on_stdout_does_not_classify_as_a_provider_failure(self): self.assertFalse(devin.transient("I could not finish: my test run timed out and the quota for retries is used up", "")) +class AgyTransientTest(unittest.TestCase): + def test_a_dropped_connection_is_a_provider_failure_even_with_an_answer_written(self): + self.assertTrue(agy.transient('API error (attempt 1): request failed: Post "https://x/v1internal:streamGenerateContent": EOF')) + self.assertTrue(agy.transient("quota exceeded")) + self.assertFalse(agy.transient("the agent wrote an invalid file")) + + class DevinEffortTest(unittest.TestCase): def test_effort_is_read_from_the_model_id(self): self.assertEqual(devin.split_effort("swe-2-max"), ("swe-2", "max")) diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 383f5598..c7850dce 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -336,6 +336,10 @@ def test_a_repeated_run_refuses_a_different_wright_binary(self): (trial / "result.json").write_text(json.dumps({"environment": {"wright": "x", "wrightSha256": agent_bench.file_sha256(other)}})) self.assertIsNone(agent_bench.wright_mismatch(run, str(other))) + def test_an_effort_the_model_id_already_names_is_not_repeated_in_the_run_name(self): + self.assertEqual(agent_bench.model_slug({"adapter": "agy", "model": "gemini-3.8-flash-high", "effort": "high"}), "agy-gemini-3.8-flash-high") + self.assertEqual(agent_bench.model_slug({"adapter": "codex", "model": "gpt-6-luna", "effort": "xhigh"}), "codex-gpt-6-luna-xhigh") + def test_tools_differ_only_in_availability(self): agent = f"cp {reference()}/* . && (wright check mode.ws >/dev/null 2>&1 || echo no-wright > missing-wright.txt)" none = self.trial(agent, tool="none") From 7d00b00df84021e448c122b2f9a042c4d2e2d391 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 11:13:03 +0800 Subject: [PATCH 19/32] fix(bench): serialize evaluate's effective options so its matrix.json reproduces the run The manifest only carried skill_dirs and wiki_dir, so 'agent_bench.py matrix /matrix.json' dropped every trial-time setting the run was made with: it ran unsandboxed, wrote into --out instead of the run directory, and lost env_pass (HOME or the direct API keys), credentials, allow/deny lists, timeout, canary, ancestor check, retry policy, adapter, and wright. The options dict now records each option at its effective value; out and out_root serialize as "." and "..", which the loader resolves against the matrix file's directory, so the file stays a self-contained, relocatable manifest of its run. trial_dir honors options.out, and skill_dirs/wiki_dir/out/out_root resolve relative to the file. --- benchmarks/agent/agent_bench.py | 21 ++++++++++++--- benchmarks/agent/test_agent_bench.py | 38 ++++++++++++++++++++++++++++ docs/agent-benchmark.md | 9 ++++--- 3 files changed, 61 insertions(+), 7 deletions(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 851635b3..1683a380 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -503,14 +503,21 @@ def cmd_matrix(args: argparse.Namespace) -> int: print(f"not applicable: {scenario_id} {label}", flush=True) state = {"streak": 0, "interrupted": 0, "unattempted": 0, "failed": 0} lock = threading.Lock() - options = {k: ({n: Path(v) for n, v in val.items()} if k == "skill_dirs" else Path(val) if k == "wiki_dir" and val else val) for k, val in config.get("options", {}).items()} + base = args.config.resolve().parent # relative option paths resolve against the matrix file, so a run's manifest is self-contained + + def option_path(value: str) -> Path: + path = Path(value) + return path if path.is_absolute() else (base / path).resolve() + + options = {k: ({n: option_path(v) for n, v in val.items()} if k == "skill_dirs" else option_path(val) if k in ("wiki_dir", "out", "out_root") and val else val) + for k, val in config.get("options", {}).items()} merged = {**vars(args), **options} if merged.get("deny_read") and not merged.get("file_sandbox"): raise SystemExit("config options: deny_read does nothing without file_sandbox") def work(job: tuple) -> None: scenario_id, agent, cell, trial = job - out = trial_dir(args.out, scenario_id, agent["id"], cell, trial) + out = trial_dir(merged["out"], scenario_id, agent["id"], cell, trial) finished = out / "result.json" if finished.is_file() and json.loads(finished.read_text()).get("status") != "provider-interrupted": # an interrupted trial is retried on the next run return @@ -640,8 +647,14 @@ def cmd_evaluate(args: argparse.Namespace) -> int: args.out_root = args.out # sibling evaluation runs must stay unreadable too args.out = args.out / args.name args.out.mkdir(parents=True, exist_ok=True) + # every option a trial reads is serialized at its effective value, so `matrix` on this file reproduces the run config = {"agents": [{"id": agent_id, "cmd": cmd}], "cells": cells, "scenarios": scenarios, "trials": args.trials, "parallel": args.parallel, "seed": args.seed, - "options": {"skill_dirs": {k: str(v) for k, v in args.skill_dirs.items()}, "wiki_dir": str(args.wiki_dir) if args.wiki_dir else None}} + "options": {"out": ".", "out_root": "..", "wright": args.wright, "adapter": args.adapter, + "file_sandbox": args.file_sandbox, "env_pass": args.env_pass, "credentials": args.credentials, + "allow_read": args.allow_read, "deny_read": args.deny_read, + "timeout": args.timeout, "canary_cmd": args.canary_cmd, "check_ancestors": args.check_ancestors, + "infra_retries": args.infra_retries, "infra_backoff": args.infra_backoff, + "skill_dirs": {k: str(v) for k, v in args.skill_dirs.items()}, "wiki_dir": str(args.wiki_dir) if args.wiki_dir else None}} args.config = args.out / "matrix.json" args.config.write_text(json.dumps(config, indent=2) + "\n") status = cmd_matrix(args) @@ -651,7 +664,7 @@ def cmd_evaluate(args: argparse.Namespace) -> int: languages = ["workshop", "opy"] expected = {lang: [s for s in all_scenario_ids() if load_scenario(s)["language"] == lang and load_scenario(s).get("split") == "test"] for lang in languages} bench_score.main([args.out], languages, expected, None) - (args.out / "RESULTS.md").write_text(f"# Agent benchmark results: {args.name}\n\nAgent `{agent_id}`. Generated by `agent_bench.py evaluate`; `matrix.json` reproduces it.\n\n" + (args.out / "RESULTS.md").write_text(f"# Agent benchmark results: {args.name}\n\nAgent `{agent_id}`. Generated by `agent_bench.py evaluate`; `agent_bench.py matrix /matrix.json` resumes it in place.\n\n" f"## Score\n\n```\n{(args.out / 'score.txt').read_text()}```\n\n{(args.out / 'report.md').read_text()}") print(f"wrote {args.out / 'RESULTS.md'}") return status diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index c7850dce..401c4f5e 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -167,6 +167,44 @@ def test_matrix_skips_inapplicable_cells_and_stops_after_repeated_interruptions( self.assertTrue(all(json.loads(p.read_text())["status"] != "provider-interrupted" for p in (self.out / "m").rglob("result.json"))) self.assertEqual(len(list((self.out / "m").rglob("result.json"))), 4) # the interrupted two were rerun and the rest ran + def test_evaluate_serializes_the_run_options_so_its_matrix_reproduces_the_run(self): + import contextlib + import io + script = agent_bench.HERE / "adapters" / "_test_fake_adapter.py" + script.write_text("import os, sys\nsys.stdin.read()\nopen('probe.txt', 'w').write(os.environ.get('BENCH_PROBE', '') + '|' + os.environ.get('HOME', ''))\n") + self.addCleanup(script.unlink, True) + args = argparse.Namespace( + adapter="fake", model="m", effort=None, name="eval", out=self.out, wright=str(Path(WRIGHT).resolve()), + skill_dirs={"wright-skill": self.skill_dir("wright-skill")}, wiki_dir=None, env_pass=["BENCH_PROBE"], + allow_read=[str(self.out)], deny_read=[], canary_cmd=None, check_ancestors=False, + timeout=30, infra_retries=0, infra_backoff=0, no_file_sandbox=True, dry_run=False, + cells="score", split="test", scenarios=[SCENARIO], trials=1, parallel=1, seed=1) + run_dir = self.out / "eval" + with patch.dict(agent_bench.ADAPTERS, {"fake": "_test_fake_adapter.py"}), patch.dict(agent_bench.ADAPTER_READS, {"fake": []}), \ + patch.dict(os.environ, {"BENCH_PROBE": "present"}), contextlib.redirect_stdout(io.StringIO()): + code = agent_bench.cmd_evaluate(args) + self.assertEqual(code, 0) + options = json.loads((run_dir / "matrix.json").read_text())["options"] + self.assertEqual((options["out"], options["out_root"], options["adapter"]), (".", "..", "fake")) + self.assertEqual((options["wright"], options["timeout"], options["file_sandbox"]), (str(Path(WRIGHT).resolve()), 30, False)) + self.assertIn("BENCH_PROBE", options["env_pass"]) + self.assertIn("HOME", options["env_pass"]) + self.assertIn(str(self.out), options["allow_read"]) + trial = run_dir / SCENARIO / "fake-m" / "wright+wright-skill_none_off-1" + self.assertEqual((trial / "workspace" / "probe.txt").read_text(), f"present|{Path.home()}") + elsewhere = self.out / "elsewhere" # flag values that would all be wrong; every serialized option must win + margs = argparse.Namespace(config=run_dir / "matrix.json", out=elsewhere, wright="/missing", skill_dirs={}, wiki_dir=None, + env_pass=[], check_ancestors=True, canary_cmd=None, timeout=99, infra_retries=5, infra_backoff=5, + file_sandbox=True, deny_read=[], allow_read=[]) + (trial / "result.json").unlink() + with patch.dict(os.environ, {"BENCH_PROBE": "present"}), contextlib.redirect_stdout(io.StringIO()): + code = agent_bench.cmd_matrix(margs) + self.assertEqual(code, 0) + self.assertFalse(elsewhere.exists()) # the run's own directory, not --out, receives the rerun trial + result = json.loads((trial / "result.json").read_text()) + self.assertEqual((result["protocol"]["timeoutSeconds"], result["fileWriteEnforcement"]), (30, "unrestricted")) + self.assertEqual((trial / "workspace" / "probe.txt").read_text(), f"present|{Path.home()}") # env_pass reached the agent again + def test_scenarios_are_solvable_and_not_vacuous(self): self.assertTrue(agent_bench.validate(WRIGHT, self.out / "validate")) diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index bf3cacc4..20b6b7a7 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -234,9 +234,12 @@ python3 benchmarks/agent/agent_bench.py score target/agent-bench # Wright Agen ``` [`matrix.example.json`](../benchmarks/agent/matrix.example.json) is the Tier 1 matrix. `matrix.json` lists `agents` (`{id, cmd}`), `cells`, optional `scenarios`, -`trials`, `parallel`, `seed` (run order is shuffled by it), and `options` -(`skill_dirs` as `{name: dir}`, `wiki_dir`, `env_pass`, ...). Cells not applicable to a scenario's language are skipped; the matrix stops after two consecutive provider interruptions and exits 3 when any occurred. Finished runs are skipped, so an -interrupted matrix resumes. +`trials`, `parallel`, `seed` (run order is shuffled by it), and `options`. Options +are the trial-time settings a run needs, overriding their command-line counterparts: `out` and `out_root` (relative paths resolve against the matrix +file's directory, so `evaluate`'s `out: "."` makes the file's own directory the run directory), `wright`, `adapter`, `file_sandbox`, `env_pass`, `credentials`, +`allow_read`/`deny_read`, `timeout`, `canary_cmd`, `check_ancestors`, `infra_retries`/`infra_backoff`, `skill_dirs` as `{name: dir}`, `wiki_dir`. Cells not +applicable to a scenario's language are skipped; the matrix stops after two consecutive provider interruptions and exits 3 when any occurred. Finished runs +are skipped, so an interrupted matrix resumes; `evaluate` writes its effective options into `matrix.json`, so `matrix /matrix.json` resumes that run in place. For a local offline-declared pilot, use [`matrix.pilot.example.json`](../benchmarks/agent/matrix.pilot.example.json): one From c5117cc9deec9c1d3f7f1aea40615af05b0e2fd6 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 11:13:14 +0800 Subject: [PATCH 20/32] fix(bench): refuse a score card when a run directory mixes read/write/network enforcement MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit identity_of covered the binary, skills, suite, agent, and protocol but not how the agent's filesystem and network were policed. A directory holding both sandboxed and unrestricted trials — which an unsandboxed 'matrix' rerun used to produce — still emitted a card, silently mixing trials that could read the answer keys with ones that could not. The card now records and discloses the read policy (allow-list mode, not the per-trial path lists) and compare() warns when runs differ. --- benchmarks/agent/bench_score.py | 12 +++++++++++- benchmarks/agent/test_score.py | 23 +++++++++++++++++++++-- 2 files changed, 32 insertions(+), 3 deletions(-) diff --git a/benchmarks/agent/bench_score.py b/benchmarks/agent/bench_score.py index bb5aab63..b8637fa5 100644 --- a/benchmarks/agent/bench_score.py +++ b/benchmarks/agent/bench_score.py @@ -38,6 +38,12 @@ def pass_power_k(usable: int, valid: int, k: int) -> float: return comb(usable, k) / comb(valid, k) if valid >= k else 0.0 +def read_enforcement(run: dict) -> str | None: + """The file-read policy in one word: 'allow-list' is recorded with its hidden/allowed paths, which differ per trial.""" + enforcement = run.get("fileReadEnforcement") + return enforcement.get("mode") if isinstance(enforcement, dict) else enforcement + + def identity_of(run: dict) -> dict: env = run["environment"] info = run.get("agentInfo") or {} @@ -47,6 +53,8 @@ def identity_of(run: dict) -> dict: "suite": (env.get("suite") or {}).get("hash"), "agent": run["agent"]["id"], "model": info.get("model"), "effort": info.get("effort"), "protocol": run.get("protocol"), + "fileReadEnforcement": read_enforcement(run), "fileWriteEnforcement": run.get("fileWriteEnforcement"), + "networkEnforcement": run.get("networkEnforcement"), # what the agent could reach is part of the environment being scored } @@ -101,6 +109,7 @@ def card(results: list[dict], language: str, expected: list[str]) -> dict: "harness": sorted({r["environment"].get("harness") for r in valid if r["environment"].get("harness")}), "networkEnforcement": sorted({r.get("networkEnforcement") or "not recorded" for r in valid}), "fileWriteEnforcement": sorted({r.get("fileWriteEnforcement") or "not recorded" for r in valid}), + "fileReadEnforcement": sorted({read_enforcement(r) or "not recorded" for r in valid}), "exclusions": {status: len(runs) for status, runs in sorted(excluded.items())}, "perScenario": [{"scenario": s, "usable": sum(v), "valid": len(v), "rate": round(rates[s], 3)} for s, v in sorted(per.items())], "byFamily": {f: round(100 * sum(v) / len(v), 1) for f, v in sorted(families.items())}, @@ -126,6 +135,7 @@ def render(c: dict) -> str: f"Suite: {c['suite'].get('version')} {c['suite'].get('hash')}", f"Wright: {ident['wright']} sha256 {ident['wrightSha256']}", f"Skills: {', '.join(f'{n} {h}' for n, h in ident['skills'].items()) or 'none'}", f"Harness: {', '.join(c['harness']) or 'not recorded'}", f"Network: {', '.join(x or 'not recorded' for x in c['networkEnforcement'])}", f"File writes: {', '.join(x or 'not recorded' for x in c['fileWriteEnforcement'])}", + f"File reads: {', '.join(x or 'not recorded' for x in c['fileReadEnforcement'])}", f"Excluded: {c['exclusions'] or 'none'}", ] if c["provisional"]: @@ -137,7 +147,7 @@ def render(c: dict) -> str: return "\n".join(lines) + "\n" -COMPARABLE = ("wrightSha256", "skills", "suite") # what must match for two cards to be read side by side; agent, model, and effort are what is being compared +COMPARABLE = ("wrightSha256", "skills", "suite", "fileReadEnforcement", "fileWriteEnforcement", "networkEnforcement") # what must match for two cards to be read side by side; agent, model, and effort are what is being compared def compare(dirs: list[Path]) -> str: diff --git a/benchmarks/agent/test_score.py b/benchmarks/agent/test_score.py index 1883710b..742e0881 100644 --- a/benchmarks/agent/test_score.py +++ b/benchmarks/agent/test_score.py @@ -7,13 +7,14 @@ import bench_score -def run(scenario, trial, usable, label=bench_score.CANONICAL, status="completed", split="test", language="opy", sha="a" * 64, model="m"): +def run(scenario, trial, usable, label=bench_score.CANONICAL, status="completed", split="test", language="opy", sha="a" * 64, model="m", **fields): return { "scenario": scenario, "family": "diagnosis" if scenario.endswith("1") else "modification", "language": language, "split": split, "condition": {"label": label}, "status": status, "usable": usable, "failedLayers": [] if usable else ["agent"], "agent": {"id": "codex", "seconds": 10.0}, "agentInfo": {"model": model, "effort": "high"}, "protocol": {"timeoutSeconds": 60, "infraRetries": 0}, "environment": {"wright": "wright 0.5.0", "wrightSha256": sha, "skills": {"wright-skill": {"sha256": "b" * 64}}, "suite": {"version": "v1", "hash": "c" * 64}, "harness": "abc"}, - "networkEnforcement": "declared-only", "fileWriteEnforcement": "trial-directory-only", "usage": {"totalTokens": 1000}, "_trial": trial, + "networkEnforcement": "declared-only", "fileWriteEnforcement": "trial-directory-only", + "fileReadEnforcement": {"mode": "allow-list", "hidden": ["/home"], "allowed": ["/run"]}, "usage": {"totalTokens": 1000}, "_trial": trial, **fields, } @@ -70,6 +71,14 @@ def test_runs_from_different_environments_are_refused(self): self.assertIn("no score", bench_score.render(c)) self.assertIn("model", bench_score.card(runs({"s0"}) + [run("s1", 9, True, model="other")], "opy", SCENARIOS)["refused"]) + def test_mixed_file_read_enforcement_is_refused(self): + mixed = runs({"s0"}) + [run("s0", 9, True, fileReadEnforcement="unrestricted")] + c = bench_score.card(mixed, "opy", SCENARIOS) + self.assertIn("fileReadEnforcement", c["refused"]) + clean = bench_score.card(runs(set(SCENARIOS)), "opy", SCENARIOS) + self.assertEqual(clean["fileReadEnforcement"], ["allow-list"]) # the allow-list mode, not the per-trial path lists + self.assertIn("File reads: allow-list", bench_score.render(clean)) + def test_small_suites_and_missing_scenarios_are_provisional(self): c = bench_score.card(runs(set(SCENARIOS))[:21], "opy", SCENARIOS) self.assertTrue(any("missing held-out scenarios" in p for p in c["provisional"])) @@ -117,6 +126,16 @@ def test_compare_tables_runs_and_warns_when_the_environment_differs(self): same = bench_score.compare([root / "codex-run"]) self.assertNotIn("WARNING", same) + def test_compare_warns_when_the_file_read_policy_differs(self): + root = Path(tempfile.mkdtemp(dir=Path(__file__).resolve().parents[2] / "target")) + self.addCleanup(shutil.rmtree, root, True) + for name, enforcement in (("sandboxed", {"mode": "allow-list"}), ("open", "unrestricted")): + directory = root / name + directory.mkdir() + card = bench_score.card(runs(SCENARIOS[:6], fileReadEnforcement=enforcement), "opy", SCENARIOS) + (directory / "score.json").write_text(json.dumps({"contract": bench_score.CONTRACT, "cards": [card]})) + self.assertIn("runs differ in fileReadEnforcement", bench_score.compare([root / "sandboxed", root / "open"])) + if __name__ == "__main__": unittest.main() From 404a0759d7d7b5e0f9d0353c72b83b0f3d04fa38 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 11:13:17 +0800 Subject: [PATCH 21/32] fix(bench): create the credential sync temp file at 0o600 and flag the claude-code sandbox login note sync_credentials_back wrote the staging file at the default umask before chmod; it is now created at 0o600 (replacing any leftover first). evaluate prints a one-line note that claude-code can read but not refresh its login under the file sandbox, and the vestigial single-element argument loop over the suite parser is flattened. --- benchmarks/agent/agent_bench.py | 21 ++++++++++++--------- 1 file changed, 12 insertions(+), 9 deletions(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 1683a380..5cc88574 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -243,6 +243,8 @@ def sync_credentials_back(pairs: list[tuple[str, str]], run_dir: Path, home: Pat if not isolated.is_file() or not real.is_file() or isolated.read_bytes() == real.read_bytes() or isolated.stat().st_mtime <= real.stat().st_mtime: continue temporary = real.with_name(f".{real.name}.bench-sync") + temporary.unlink(missing_ok=True) # a leftover temp keeps its mode; start fresh at 0o600 so a credential never sits at the default umask + temporary.touch(mode=0o600) temporary.write_bytes(isolated.read_bytes()) temporary.chmod(real.stat().st_mode & 0o777) os.replace(temporary, real) @@ -622,6 +624,8 @@ def cmd_evaluate(args: argparse.Namespace) -> int: raise SystemExit("--deny-read does nothing without the file sandbox") args.credentials = CREDENTIALS.get(args.adapter, []) + OPTIONAL_CREDENTIALS.get(args.adapter, []) args.allow_read = [*(str(Path.home() / rel) for rel in ADAPTER_READS[args.adapter]), *args.allow_read] + if args.adapter == "claude-code" and args.file_sandbox: + print("note: the file sandbox lets claude-code read but not refresh its login; pass --no-file-sandbox when its token may rotate mid-run", flush=True) script = Path(__file__).parent / "adapters" / ADAPTERS[args.adapter] effort = f"BENCH_THINKING={shlex.quote(args.effort)} " if args.effort else "" agent_id = model_slug({"adapter": args.adapter, "model": args.model, "effort": args.effort}) @@ -756,15 +760,14 @@ def main() -> int: su = sub.choices["suite"] su.add_argument("--suite-name", default="results", help="directory under --out holding every model's run and the results page") su.add_argument("--only", nargs="*", metavar="ADAPTER[:MODEL]", help="evaluate only these entries of the models list") - for shared in (su,): - shared.add_argument("--cells", choices=("score", "controls"), default="score") - shared.add_argument("--split", choices=("test", "train", "all"), default="test") - shared.add_argument("--scenarios", nargs="*", choices=all_scenario_ids()) - shared.add_argument("--trials", type=int, default=3) - shared.add_argument("--parallel", type=int, default=1) - shared.add_argument("--seed", type=int, default=1) - shared.add_argument("--dry-run", action="store_true") - shared.add_argument("--no-file-sandbox", action="store_true") + su.add_argument("--cells", choices=("score", "controls"), default="score") + su.add_argument("--split", choices=("test", "train", "all"), default="test") + su.add_argument("--scenarios", nargs="*", choices=all_scenario_ids()) + su.add_argument("--trials", type=int, default=3) + su.add_argument("--parallel", type=int, default=1) + su.add_argument("--seed", type=int, default=1) + su.add_argument("--dry-run", action="store_true") + su.add_argument("--no-file-sandbox", action="store_true") lb = sub.add_parser("leaderboard", help="write the publishable results page (Markdown, HTML, JSON) from evaluation run directories") lb.add_argument("dirs", nargs="+", type=Path) lb.add_argument("--page-out", type=Path, help="directory for the page; `leaderboard` inside the first directory's parent by default") From 16ebe2a9447dbca4ca3a7811644452f26b742b85 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 11:52:28 +0800 Subject: [PATCH 22/32] fix(bench): resolve user path options when evaluate serializes its matrix Relative --skill-dir/--wiki-dir/--allow-read/--deny-read values were written to matrix.json raw, then resolved against the run directory on replay instead of the evaluate working directory, so a reproduced run looked for skills, wiki, and read policies under the wrong root. Serialize them resolved to absolute paths; the generated out/out_root stay manifest-relative ('.'/'..') so the run directory can still move or be archived. On the read side the same manifest-relative rule now covers allow_read/deny_read entries and ~ expansion, and a null path option means 'unset' rather than overriding with a value that crashes path handling. docs/agent-benchmark.md now names every path-valued option, documents fileReadEnforcement/fileWriteEnforcement, and states that mixed-enforcement runs refuse a score card. --- benchmarks/agent/agent_bench.py | 22 +++++++++++++++++----- benchmarks/agent/test_agent_bench.py | 13 ++++++++----- docs/agent-benchmark.md | 12 ++++++++---- 3 files changed, 33 insertions(+), 14 deletions(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 5cc88574..b17725bf 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -508,11 +508,20 @@ def cmd_matrix(args: argparse.Namespace) -> int: base = args.config.resolve().parent # relative option paths resolve against the matrix file, so a run's manifest is self-contained def option_path(value: str) -> Path: - path = Path(value) + path = Path(value).expanduser() return path if path.is_absolute() else (base / path).resolve() - options = {k: ({n: option_path(v) for n, v in val.items()} if k == "skill_dirs" else option_path(val) if k in ("wiki_dir", "out", "out_root") and val else val) - for k, val in config.get("options", {}).items()} + options = {} + for key, val in config.get("options", {}).items(): + if key in ("wiki_dir", "out", "out_root") and not val: + continue # a null path option means 'unset', not an override + if key == "skill_dirs": + val = {n: option_path(v) for n, v in val.items()} + elif key in ("wiki_dir", "out", "out_root"): + val = option_path(val) + elif key in ("allow_read", "deny_read") and isinstance(val, list): + val = [option_path(v) for v in val] + options[key] = val merged = {**vars(args), **options} if merged.get("deny_read") and not merged.get("file_sandbox"): raise SystemExit("config options: deny_read does nothing without file_sandbox") @@ -655,10 +664,13 @@ def cmd_evaluate(args: argparse.Namespace) -> int: config = {"agents": [{"id": agent_id, "cmd": cmd}], "cells": cells, "scenarios": scenarios, "trials": args.trials, "parallel": args.parallel, "seed": args.seed, "options": {"out": ".", "out_root": "..", "wright": args.wright, "adapter": args.adapter, "file_sandbox": args.file_sandbox, "env_pass": args.env_pass, "credentials": args.credentials, - "allow_read": args.allow_read, "deny_read": args.deny_read, + # paths the caller gave resolve now, against this cwd — relative ones in the file resolve against the file's directory + "allow_read": [str(Path(p).expanduser().resolve()) for p in args.allow_read], + "deny_read": [str(Path(p).expanduser().resolve()) for p in args.deny_read], "timeout": args.timeout, "canary_cmd": args.canary_cmd, "check_ancestors": args.check_ancestors, "infra_retries": args.infra_retries, "infra_backoff": args.infra_backoff, - "skill_dirs": {k: str(v) for k, v in args.skill_dirs.items()}, "wiki_dir": str(args.wiki_dir) if args.wiki_dir else None}} + "skill_dirs": {k: str(Path(v).resolve()) for k, v in args.skill_dirs.items()}, + "wiki_dir": str(args.wiki_dir.resolve()) if args.wiki_dir else None}} args.config = args.out / "matrix.json" args.config.write_text(json.dumps(config, indent=2) + "\n") status = cmd_matrix(args) diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 401c4f5e..15b9c3e8 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -173,15 +173,16 @@ def test_evaluate_serializes_the_run_options_so_its_matrix_reproduces_the_run(se script = agent_bench.HERE / "adapters" / "_test_fake_adapter.py" script.write_text("import os, sys\nsys.stdin.read()\nopen('probe.txt', 'w').write(os.environ.get('BENCH_PROBE', '') + '|' + os.environ.get('HOME', ''))\n") self.addCleanup(script.unlink, True) - args = argparse.Namespace( + self.skill_dir("wright-skill") # reached as a relative path below, resolved against the evaluate cwd + args = argparse.Namespace( # relative paths must serialize resolved: the file resolves its own against its directory adapter="fake", model="m", effort=None, name="eval", out=self.out, wright=str(Path(WRIGHT).resolve()), - skill_dirs={"wright-skill": self.skill_dir("wright-skill")}, wiki_dir=None, env_pass=["BENCH_PROBE"], - allow_read=[str(self.out)], deny_read=[], canary_cmd=None, check_ancestors=False, + skill_dirs={"wright-skill": Path("skills/wright-skill")}, wiki_dir=None, env_pass=["BENCH_PROBE"], + allow_read=["allow-this"], deny_read=[], canary_cmd=None, check_ancestors=False, timeout=30, infra_retries=0, infra_backoff=0, no_file_sandbox=True, dry_run=False, cells="score", split="test", scenarios=[SCENARIO], trials=1, parallel=1, seed=1) run_dir = self.out / "eval" with patch.dict(agent_bench.ADAPTERS, {"fake": "_test_fake_adapter.py"}), patch.dict(agent_bench.ADAPTER_READS, {"fake": []}), \ - patch.dict(os.environ, {"BENCH_PROBE": "present"}), contextlib.redirect_stdout(io.StringIO()): + patch.dict(os.environ, {"BENCH_PROBE": "present"}), contextlib.chdir(self.out), contextlib.redirect_stdout(io.StringIO()): code = agent_bench.cmd_evaluate(args) self.assertEqual(code, 0) options = json.loads((run_dir / "matrix.json").read_text())["options"] @@ -189,7 +190,9 @@ def test_evaluate_serializes_the_run_options_so_its_matrix_reproduces_the_run(se self.assertEqual((options["wright"], options["timeout"], options["file_sandbox"]), (str(Path(WRIGHT).resolve()), 30, False)) self.assertIn("BENCH_PROBE", options["env_pass"]) self.assertIn("HOME", options["env_pass"]) - self.assertIn(str(self.out), options["allow_read"]) + self.assertEqual(options["skill_dirs"]["wright-skill"], str((self.out / "skills/wright-skill").resolve())) + self.assertEqual(options["deny_read"], []) + self.assertIn(str((self.out / "allow-this").resolve()), options["allow_read"]) trial = run_dir / SCENARIO / "fake-m" / "wright+wright-skill_none_off-1" self.assertEqual((trial / "workspace" / "probe.txt").read_text(), f"present|{Path.home()}") elsewhere = self.out / "elsewhere" # flag values that would all be wrong; every serialized option must win diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index 20b6b7a7..43a2285d 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -237,7 +237,9 @@ python3 benchmarks/agent/agent_bench.py score target/agent-bench # Wright Agen `trials`, `parallel`, `seed` (run order is shuffled by it), and `options`. Options are the trial-time settings a run needs, overriding their command-line counterparts: `out` and `out_root` (relative paths resolve against the matrix file's directory, so `evaluate`'s `out: "."` makes the file's own directory the run directory), `wright`, `adapter`, `file_sandbox`, `env_pass`, `credentials`, -`allow_read`/`deny_read`, `timeout`, `canary_cmd`, `check_ancestors`, `infra_retries`/`infra_backoff`, `skill_dirs` as `{name: dir}`, `wiki_dir`. Cells not +`allow_read`/`deny_read`, `timeout`, `canary_cmd`, `check_ancestors`, `infra_retries`/`infra_backoff`, `skill_dirs` as `{name: dir}`, `wiki_dir`. Every +path-valued option (`out`, `out_root`, `skill_dirs`, `wiki_dir`, `allow_read`, `deny_read`) follows the same rule: relative resolves against the matrix file's +directory, and `evaluate` writes its own path options already resolved so the file reproduces the run from any cwd. Cells not applicable to a scenario's language are skipped; the matrix stops after two consecutive provider interruptions and exits 3 when any occurred. Finished runs are skipped, so an interrupted matrix resumes; `evaluate` writes its effective options into `matrix.json`, so `matrix /matrix.json` resumes that run in place. @@ -274,7 +276,7 @@ with the workspace, `agent.log`, snapshots, and the Wright trace beside it. | `friction`, `expectations` | Usage errors, unknown subcommands, help lookups, retries, malformed `serve` requests, unparsed `serve` responses, identical repeats; expectation E01-E12 verdicts | | `snapshots` | Strict validity of each snapshot of the entry, first valid index, and valid-to-invalid regressions | | `usage`, `context` | Turns, tokens by kind, peak context (and its share of the limit), tokens to first valid; loaded context | -| `invalid`, `infraRetries`, `networkEnforcement` | Present when the run was excluded or retried; whether network `off` was checked by a canary or only declared | +| `invalid`, `infraRetries`, `fileReadEnforcement`, `fileWriteEnforcement`, `networkEnforcement` | Present when the run was excluded or retried; how file reads (`allow-list` with the hidden and allowed paths, or `unrestricted`), file writes (`trial-directory-only` or `unrestricted`), and network `off` (`canary-checked` or `declared-only`) were enforced | `toolUse` is recorded per CLI invocation through the shim; `wright serve` sessions are teed line by line into the trace. Comparing cells for the same scenario shows what @@ -306,9 +308,11 @@ over scenarios, and gives a two-stage bootstrap 95% interval (10,000 draws, seed 467) and Pass^k as a secondary figure. Provider-interrupted, invalid, and agent-error runs are published as exclusions; timeouts count. The score is refused if the runs differ in Wright binary, skill hashes, suite hash, agent, -model, effort, or protocol, and it is marked provisional with fewer than eight +model, effort, protocol, or file-read/file-write/network enforcement — a run +where the agent could read answer keys does not score beside a sandboxed one — +and it is marked provisional with fewer than eight held-out scenarios, missing scenarios, or unequal trials. The card discloses -`networkEnforcement` (`declared-only` or `canary-checked`). +`fileReadEnforcement`, `fileWriteEnforcement`, and `networkEnforcement` modes. ## Evaluating without an agent harness From fee362eda73b52d531e6c750bdf44f26e5d157a4 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 11:52:32 +0800 Subject: [PATCH 23/32] fix(bench): compare score identities with missing fields as 'not recorded' compare() indexed every COMPARABLE key on each card's identity, so a score.json written before file-read/write/network enforcement joined the identity crashed with KeyError instead of warning that the cards differ. ident.get(k) reports the absent field as a difference, which is the same treatment card() gives unrecorded enforcement. --- benchmarks/agent/bench_score.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/benchmarks/agent/bench_score.py b/benchmarks/agent/bench_score.py index b8637fa5..3192a18b 100644 --- a/benchmarks/agent/bench_score.py +++ b/benchmarks/agent/bench_score.py @@ -163,7 +163,7 @@ def compare(dirs: list[Path]) -> str: rows.append((card_["track"], directory.name, "no score", card_["refused"])) continue ident = card_["identity"] - identities.setdefault(card_["track"], []).append((directory.name, {k: ident[k] for k in COMPARABLE})) + identities.setdefault(card_["track"], []).append((directory.name, {k: ident.get(k) for k in COMPARABLE})) # cards written before a field existed compare as 'not recorded' label = " ".join(filter(None, (ident["agent"], ident["effort"] and f"effort {ident['effort']}"))) note = "provisional: " + "; ".join(card_["provisional"]) if card_["provisional"] else "" rows.append((card_["track"], directory.name, f"{card_['score']} [{card_['ci95'][0]}-{card_['ci95'][1]}]", f"{label}; {card_['trialsPerScenario']} trial(s) x {card_['scenarios']} scenarios; excluded {card_['exclusions'] or 'none'}. {note}".strip())) From e884ee5164c98f22a0515f5a315889e001c8ccb8 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 12:06:48 +0800 Subject: [PATCH 24/32] fix(bench): close the remaining path and name gaps in manifest options Review nits on the reproduction fix: --skill-dir/--wiki-dir now expand ~ like the read policies do; wright follows the same manifest-relative rule as the other path options so a hand-authored relative binary resolves against the file instead of the trial workspace; a null skill_dirs means unset like the other path keys instead of crashing on .items(); compare() reads agent/effort with .get for uniformly missing-field-tolerant old cards; and evaluate rejects --name values that are not a single directory component ('..', 'a/b', empty), which previously let the run directory escape --out. The reproduction test also covers relative --wiki-dir and infra_retries overrides. --- benchmarks/agent/agent_bench.py | 11 +++++++---- benchmarks/agent/bench_score.py | 2 +- benchmarks/agent/test_agent_bench.py | 7 +++++-- docs/agent-benchmark.md | 2 +- 4 files changed, 14 insertions(+), 8 deletions(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index b17725bf..ae406f11 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -511,13 +511,14 @@ def option_path(value: str) -> Path: path = Path(value).expanduser() return path if path.is_absolute() else (base / path).resolve() + path_keys = ("wiki_dir", "out", "out_root", "wright") options = {} for key, val in config.get("options", {}).items(): - if key in ("wiki_dir", "out", "out_root") and not val: + if key in path_keys + ("skill_dirs",) and not val: continue # a null path option means 'unset', not an override if key == "skill_dirs": val = {n: option_path(v) for n, v in val.items()} - elif key in ("wiki_dir", "out", "out_root"): + elif key in path_keys: val = option_path(val) elif key in ("allow_read", "deny_read") and isinstance(val, list): val = [option_path(v) for v in val] @@ -636,6 +637,8 @@ def cmd_evaluate(args: argparse.Namespace) -> int: if args.adapter == "claude-code" and args.file_sandbox: print("note: the file sandbox lets claude-code read but not refresh its login; pass --no-file-sandbox when its token may rotate mid-run", flush=True) script = Path(__file__).parent / "adapters" / ADAPTERS[args.adapter] + if not args.name or args.name in (".", "..") or Path(args.name).name != args.name: + raise SystemExit("--name must be a single directory name") effort = f"BENCH_THINKING={shlex.quote(args.effort)} " if args.effort else "" agent_id = model_slug({"adapter": args.adapter, "model": args.model, "effort": args.effort}) cmd = f"BENCH_MODEL={shlex.quote(args.model)} {effort}{shlex.quote(sys.executable)} {shlex.quote(str(script))}" @@ -669,8 +672,8 @@ def cmd_evaluate(args: argparse.Namespace) -> int: "deny_read": [str(Path(p).expanduser().resolve()) for p in args.deny_read], "timeout": args.timeout, "canary_cmd": args.canary_cmd, "check_ancestors": args.check_ancestors, "infra_retries": args.infra_retries, "infra_backoff": args.infra_backoff, - "skill_dirs": {k: str(Path(v).resolve()) for k, v in args.skill_dirs.items()}, - "wiki_dir": str(args.wiki_dir.resolve()) if args.wiki_dir else None}} + "skill_dirs": {k: str(Path(v).expanduser().resolve()) for k, v in args.skill_dirs.items()}, + "wiki_dir": str(args.wiki_dir.expanduser().resolve()) if args.wiki_dir else None}} args.config = args.out / "matrix.json" args.config.write_text(json.dumps(config, indent=2) + "\n") status = cmd_matrix(args) diff --git a/benchmarks/agent/bench_score.py b/benchmarks/agent/bench_score.py index 3192a18b..aa10a2f7 100644 --- a/benchmarks/agent/bench_score.py +++ b/benchmarks/agent/bench_score.py @@ -164,7 +164,7 @@ def compare(dirs: list[Path]) -> str: continue ident = card_["identity"] identities.setdefault(card_["track"], []).append((directory.name, {k: ident.get(k) for k in COMPARABLE})) # cards written before a field existed compare as 'not recorded' - label = " ".join(filter(None, (ident["agent"], ident["effort"] and f"effort {ident['effort']}"))) + label = " ".join(filter(None, (ident.get("agent"), ident.get("effort") and f"effort {ident['effort']}"))) note = "provisional: " + "; ".join(card_["provisional"]) if card_["provisional"] else "" rows.append((card_["track"], directory.name, f"{card_['score']} [{card_['ci95'][0]}-{card_['ci95'][1]}]", f"{label}; {card_['trialsPerScenario']} trial(s) x {card_['scenarios']} scenarios; excluded {card_['exclusions'] or 'none'}. {note}".strip())) lines = ["| track | run | score [95% CI] | agent and notes |", "| --- | --- | --- | --- |", *(f"| {t} | {d} | {s} | {n} |" for t, d, s, n in sorted(rows))] diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 15b9c3e8..9e8b1363 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -174,11 +174,12 @@ def test_evaluate_serializes_the_run_options_so_its_matrix_reproduces_the_run(se script.write_text("import os, sys\nsys.stdin.read()\nopen('probe.txt', 'w').write(os.environ.get('BENCH_PROBE', '') + '|' + os.environ.get('HOME', ''))\n") self.addCleanup(script.unlink, True) self.skill_dir("wright-skill") # reached as a relative path below, resolved against the evaluate cwd + (self.out / "wiki-snap").mkdir() args = argparse.Namespace( # relative paths must serialize resolved: the file resolves its own against its directory adapter="fake", model="m", effort=None, name="eval", out=self.out, wright=str(Path(WRIGHT).resolve()), - skill_dirs={"wright-skill": Path("skills/wright-skill")}, wiki_dir=None, env_pass=["BENCH_PROBE"], + skill_dirs={"wright-skill": Path("skills/wright-skill")}, wiki_dir=Path("wiki-snap"), env_pass=["BENCH_PROBE"], allow_read=["allow-this"], deny_read=[], canary_cmd=None, check_ancestors=False, - timeout=30, infra_retries=0, infra_backoff=0, no_file_sandbox=True, dry_run=False, + timeout=30, infra_retries=3, infra_backoff=0, no_file_sandbox=True, dry_run=False, cells="score", split="test", scenarios=[SCENARIO], trials=1, parallel=1, seed=1) run_dir = self.out / "eval" with patch.dict(agent_bench.ADAPTERS, {"fake": "_test_fake_adapter.py"}), patch.dict(agent_bench.ADAPTER_READS, {"fake": []}), \ @@ -191,7 +192,9 @@ def test_evaluate_serializes_the_run_options_so_its_matrix_reproduces_the_run(se self.assertIn("BENCH_PROBE", options["env_pass"]) self.assertIn("HOME", options["env_pass"]) self.assertEqual(options["skill_dirs"]["wright-skill"], str((self.out / "skills/wright-skill").resolve())) + self.assertEqual(options["wiki_dir"], str((self.out / "wiki-snap").resolve())) self.assertEqual(options["deny_read"], []) + self.assertEqual(options["infra_retries"], 3) self.assertIn(str((self.out / "allow-this").resolve()), options["allow_read"]) trial = run_dir / SCENARIO / "fake-m" / "wright+wright-skill_none_off-1" self.assertEqual((trial / "workspace" / "probe.txt").read_text(), f"present|{Path.home()}") diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index 43a2285d..ac4c410b 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -238,7 +238,7 @@ python3 benchmarks/agent/agent_bench.py score target/agent-bench # Wright Agen are the trial-time settings a run needs, overriding their command-line counterparts: `out` and `out_root` (relative paths resolve against the matrix file's directory, so `evaluate`'s `out: "."` makes the file's own directory the run directory), `wright`, `adapter`, `file_sandbox`, `env_pass`, `credentials`, `allow_read`/`deny_read`, `timeout`, `canary_cmd`, `check_ancestors`, `infra_retries`/`infra_backoff`, `skill_dirs` as `{name: dir}`, `wiki_dir`. Every -path-valued option (`out`, `out_root`, `skill_dirs`, `wiki_dir`, `allow_read`, `deny_read`) follows the same rule: relative resolves against the matrix file's +path-valued option (`out`, `out_root`, `wright`, `skill_dirs`, `wiki_dir`, `allow_read`, `deny_read`) follows the same rule: relative resolves against the matrix file's directory, and `evaluate` writes its own path options already resolved so the file reproduces the run from any cwd. Cells not applicable to a scenario's language are skipped; the matrix stops after two consecutive provider interruptions and exits 3 when any occurred. Finished runs are skipped, so an interrupted matrix resumes; `evaluate` writes its effective options into `matrix.json`, so `matrix /matrix.json` resumes that run in place. From a78c18d82386dfb545cd0c243a10269b6c44d3b6 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 12:41:49 +0800 Subject: [PATCH 25/32] fix(bench): classify malformed provider payloads and unblock adapter stdin MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit direct.py parsed API replies with raw indexing, so a malformed payload raised KeyError/IndexError/JSONDecodeError as a traceback instead of a classified failure; both providers now wrap response-shape reads as a non-transient ProviderError. The opencode, pi, codex, and claude-code adapters also wrote the whole prompt to stdin before draining the pipes, which deadlocks on a prompt larger than the pipe buffer against a chatty CLI — they now share common.feed_stdin, which feeds stdin on a thread. --- benchmarks/agent/adapters/claude_code.py | 5 ++--- benchmarks/agent/adapters/codex.py | 5 ++--- benchmarks/agent/adapters/common.py | 12 ++++++++++++ benchmarks/agent/adapters/direct.py | 17 ++++++++++++----- benchmarks/agent/adapters/opencode.py | 5 ++--- benchmarks/agent/adapters/pi.py | 5 ++--- benchmarks/agent/test_adapters.py | 7 +++++++ 7 files changed, 39 insertions(+), 17 deletions(-) diff --git a/benchmarks/agent/adapters/claude_code.py b/benchmarks/agent/adapters/claude_code.py index cb415213..16a614b7 100755 --- a/benchmarks/agent/adapters/claude_code.py +++ b/benchmarks/agent/adapters/claude_code.py @@ -20,7 +20,7 @@ import time from pathlib import Path -from common import as_dict, cli_version, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, feed_stdin, INFRA_EXIT, TRANSIENT TOOLS = ["Bash", "Read", "Edit", "Write", "Glob", "Grep"] WEB_TOOLS = ["WebFetch", "WebSearch"] @@ -56,8 +56,7 @@ def main() -> int: cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env={**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "CLAUDE_CODE_DISABLE_CLAUDE_MDS": "1"}, ) - proc.stdin.write(prompt) - proc.stdin.close() + feed_stdin(proc, prompt) final, errored = "", False with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: # line-buffered: a killed run keeps its usage for line in proc.stdout: diff --git a/benchmarks/agent/adapters/codex.py b/benchmarks/agent/adapters/codex.py index 427f2d12..f1a5b4e3 100644 --- a/benchmarks/agent/adapters/codex.py +++ b/benchmarks/agent/adapters/codex.py @@ -13,7 +13,7 @@ from datetime import datetime from pathlib import Path -from common import as_dict, cli_version, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, feed_stdin, INFRA_EXIT, TRANSIENT def usage_row(usage: dict, timestamp: float, limit: int | None) -> dict: @@ -78,8 +78,7 @@ def main() -> int: final, errors, summary, seen, servers, item_types, scanned = "", [], None, set(), set(), set(), 0.0 with (run / "codex-stderr.log").open("w") as stderr, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript, open(env["BENCH_USAGE"], "w", buffering=1) as usage: process = subprocess.Popen(command, env=child_env, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=stderr, text=True) - process.stdin.write(prompt) - process.stdin.close() + feed_stdin(process, prompt) for line in process.stdout: try: event = json.loads(line) diff --git a/benchmarks/agent/adapters/common.py b/benchmarks/agent/adapters/common.py index 85923b7b..bae7bac8 100644 --- a/benchmarks/agent/adapters/common.py +++ b/benchmarks/agent/adapters/common.py @@ -3,6 +3,7 @@ from __future__ import annotations import subprocess +import threading INFRA_EXIT = 75 # EX_TEMPFAIL: a provider or infrastructure failure, not an agent failure; the harness retries the trial later # Error text that means the provider, not the agent, failed. Shared so the classification cannot drift between adapters. @@ -15,6 +16,17 @@ def as_dict(value) -> dict: return value if isinstance(value, dict) else {} +def feed_stdin(proc: subprocess.Popen, text: str) -> None: + """Write the prompt to the child's stdin on a thread — a large prompt otherwise blocks the loop that must drain the pipes.""" + def feed() -> None: + try: + proc.stdin.write(text) + proc.stdin.close() + except (BrokenPipeError, ValueError): # the child exited before taking the whole prompt + pass + threading.Thread(target=feed, daemon=True).start() + + def cli_version(binary: str, env: dict | None = None) -> str | None: """First line of ` --version`, or None when the CLI does not answer.""" try: diff --git a/benchmarks/agent/adapters/direct.py b/benchmarks/agent/adapters/direct.py index 2932525f..05e5c244 100644 --- a/benchmarks/agent/adapters/direct.py +++ b/benchmarks/agent/adapters/direct.py @@ -82,10 +82,14 @@ def results(self, results: list[tuple[str, str]]) -> None: def step(self) -> tuple[str, list[tuple[str, str, dict]], dict]: data = post(self.url, self.headers, {"model": self.model, "max_tokens": MAX_OUTPUT_TOKENS, "system": self.system, "tools": self.tools, "messages": self.messages}) - self.messages.append({"role": "assistant", "content": data["content"]}) + try: + content = data["content"] + text = "".join(b.get("text", "") for b in content if b["type"] == "text") + calls = [(b["id"], b["name"], b["input"]) for b in content if b["type"] == "tool_use"] + except (KeyError, IndexError, TypeError, AttributeError) as error: + raise ProviderError(f"malformed response: {error}", False) from error + self.messages.append({"role": "assistant", "content": content}) u = data.get("usage") or {} - text = "".join(b.get("text", "") for b in data["content"] if b["type"] == "text") - calls = [(b["id"], b["name"], b["input"]) for b in data["content"] if b["type"] == "tool_use"] return text, calls, usage_row(u.get("input_tokens") or 0, u.get("output_tokens") or 0, u.get("cache_read_input_tokens") or 0, u.get("cache_creation_input_tokens") or 0, None) @@ -106,12 +110,15 @@ def results(self, results: list[tuple[str, str]]) -> None: def step(self) -> tuple[str, list[tuple[str, str, dict]], dict]: body = {"model": self.model, "tools": self.tools, "messages": self.messages, **({"reasoning_effort": self.effort} if self.effort else {})} data = post(self.url, self.headers, body) - message = data["choices"][0]["message"] + try: + message = data["choices"][0]["message"] + calls = [(c["id"], c["function"]["name"], json.loads(c["function"]["arguments"] or "{}")) for c in message.get("tool_calls") or []] + except (KeyError, IndexError, TypeError, AttributeError, json.JSONDecodeError) as error: + raise ProviderError(f"malformed response: {error}", False) from error self.messages.append(message) u = data.get("usage") or {} cached = (u.get("prompt_tokens_details") or {}).get("cached_tokens") or 0 reasoning = (u.get("completion_tokens_details") or {}).get("reasoning_tokens") or 0 - calls = [(c["id"], c["function"]["name"], json.loads(c["function"]["arguments"] or "{}")) for c in message.get("tool_calls") or []] return message.get("content") or "", calls, usage_row((u.get("prompt_tokens") or 0) - cached, (u.get("completion_tokens") or 0) - reasoning, cached, 0, reasoning) diff --git a/benchmarks/agent/adapters/opencode.py b/benchmarks/agent/adapters/opencode.py index d49b2252..6301df39 100644 --- a/benchmarks/agent/adapters/opencode.py +++ b/benchmarks/agent/adapters/opencode.py @@ -18,7 +18,7 @@ import time from pathlib import Path -from common import as_dict, cli_version, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, feed_stdin, INFRA_EXIT, TRANSIENT WEB_PERMISSIONS = {"webfetch": "deny", "websearch": "deny"} @@ -61,8 +61,7 @@ def main() -> int: if env.get("BENCH_THINKING"): cmd += ["--variant", env["BENCH_THINKING"]] proc = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=child_env) - proc.stdin.write(sys.stdin.read()) - proc.stdin.close() + feed_stdin(proc, sys.stdin.read()) final, error, tools = "", "", set() with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: # line-buffered: a killed run keeps its usage for line in proc.stdout: diff --git a/benchmarks/agent/adapters/pi.py b/benchmarks/agent/adapters/pi.py index 2d93deab..067d4b48 100755 --- a/benchmarks/agent/adapters/pi.py +++ b/benchmarks/agent/adapters/pi.py @@ -21,7 +21,7 @@ import time from pathlib import Path -from common import as_dict, cli_version, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, feed_stdin, INFRA_EXIT, TRANSIENT SCALE = {"K": 1_000, "M": 1_000_000} @@ -84,8 +84,7 @@ def main() -> int: limit = context_limit(pi, model, child_env, extensions) Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "pi", "version": cli_version(pi), "model": model, "effort": env.get("BENCH_THINKING"), "tools": ["read", "bash", "edit", "write"], "extensions": [e for e in extensions if e]}, indent=2)) proc = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=child_env) - proc.stdin.write(prompt) - proc.stdin.close() + feed_stdin(proc, prompt) loaded: list[str] = [] final, error = "", "" with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: # line-buffered: a killed run keeps its usage diff --git a/benchmarks/agent/test_adapters.py b/benchmarks/agent/test_adapters.py index e97b69a7..d08283e7 100644 --- a/benchmarks/agent/test_adapters.py +++ b/benchmarks/agent/test_adapters.py @@ -225,6 +225,13 @@ def test_client_error_is_not_an_infrastructure_failure(self): code, *_ = self.run_direct("anthropic/m", "ANTHROPIC_BASE_URL", [(400, {"error": "bad"})]) self.assertEqual(code, 1) + def test_a_malformed_payload_is_a_clean_provider_error_not_a_traceback(self): + for model, base_var, replies in (("anthropic/m", "ANTHROPIC_BASE_URL", [(200, {"unexpected": "shape"})]), + ("openai/m", "OPENAI_BASE_URL", [(200, {"choices": []})])): + with self.subTest(model=model): + code, *_ = self.run_direct(model, base_var, replies) + self.assertEqual(code, 1) + if __name__ == "__main__": unittest.main() From fd3e45fc02a1237cca06639195f1947a9af8f4b3 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 12:42:00 +0800 Subject: [PATCH 26/32] fix(bench): hide the grading oracle outside overpy cells and the sandbox profile everywhere MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit read_policy allowed benchmarks/agent/oracle unconditionally, so a wright or none cell could read and run the pinned upstream compiler — a grading authority — blurring the tool-condition contrast the sandbox exists to enforce. It is readable only when BENCH_TOOL is overpy, whose launcher execs it. The generated agent.sb profile sat inside the readable run directory and listed the hidden paths; the profile now ends with a deny rule for itself. --- benchmarks/agent/agent_bench.py | 7 +++++-- benchmarks/agent/test_agent_bench.py | 13 +++++++++---- 2 files changed, 14 insertions(+), 6 deletions(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index ae406f11..b92223db 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -269,8 +269,10 @@ def read_policy(args: argparse.Namespace, env: dict) -> tuple[list[Path], list[P for binary in (shutil.which("node"), shutil.which(ADAPTER_BINARY.get(getattr(args, "adapter", ""), ""))): if binary: runtime.append(Path(binary)) - allowed = [run_dir, HERE / "adapters", HERE / "bench_trace.py", HERE / "oracle", Path(sys.prefix), Path(sys.base_prefix), + allowed = [run_dir, HERE / "adapters", HERE / "bench_trace.py", Path(sys.prefix), Path(sys.base_prefix), *(Path(args.skill_dirs[name]) for name in selected), *(Path(p).expanduser() for p in getattr(args, "allow_read", []) or [])] + if env.get("BENCH_TOOL") == "overpy": # only the overpy launcher execs the pinned oracle; other cells must not read the grading authority + allowed.append(HERE / "oracle") for binary in runtime: # the directory of the binary and of the file its symlink resolves to allowed += [binary.parent, binary.resolve().parent] unique = lambda paths: list(dict.fromkeys(p.resolve() for p in paths)) @@ -298,7 +300,8 @@ def run_agent(args: argparse.Namespace, env: dict, workspace: Path, prompt: str) profile.write_text('(version 1)\n(allow default)\n(deny file-write*)\n' f'(allow file-write* (subpath {json.dumps(str(run_dir.resolve()))}) (subpath "/dev"))\n' '(deny file-read-data (require-all (regex "/(AGENTS|CLAUDE|GEMINI)[.]md$") ' - f'(require-not (subpath {json.dumps(str(workspace.resolve()))}))))\n' + rules) + f'(require-not (subpath {json.dumps(str(workspace.resolve()))}))))\n' + rules + + f'(deny file-read-data (literal {json.dumps(str(profile.resolve()))}))\n') # the profile lists what is hidden; the agent must not read it command = ["sandbox-exec", "-f", str(profile), "/bin/sh", "-c", args.agent_cmd] proc = subprocess.Popen( command, shell=isinstance(command, str), cwd=workspace, env=env, text=True, diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 9e8b1363..f2563001 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -287,14 +287,18 @@ def test_read_policy_hides_the_host_and_allows_only_what_the_run_needs(self): root = self.out skills = {"wright-skill": root / "s1", "opy-skill": root / "s2"} args = argparse.Namespace(out=root / "run-a", out_root=root, wright=WRIGHT, skill_dirs=skills, wiki_dir=None, deny_read=[], allow_read=[str(root / "creds")], adapter="devin") - hidden, allowed = agent_bench.read_policy(args, {"BENCH_RUN_DIR": str(root / "run-a" / "t"), "BENCH_SKILLS": "wright-skill"}) + env = {"BENCH_RUN_DIR": str(root / "run-a" / "t"), "BENCH_SKILLS": "wright-skill", "BENCH_TOOL": "wright"} + hidden, allowed = agent_bench.read_policy(args, env) self.assertTrue(all(p in hidden for p in (Path("/Users"), root.resolve(), agent_bench.HERE))) self.assertTrue(all(d.resolve() in hidden for d in agent_bench.DATA_ROOTS)) self.assertIn(skills["opy-skill"].resolve(), hidden) self.assertNotIn(skills["wright-skill"].resolve(), hidden) - for needed in (root / "run-a" / "t", skills["wright-skill"], root / "creds", agent_bench.HERE / "adapters", agent_bench.HERE / "bench_trace.py", agent_bench.HERE / "oracle", Path(WRIGHT).resolve().parent): + for needed in (root / "run-a" / "t", skills["wright-skill"], root / "creds", agent_bench.HERE / "adapters", agent_bench.HERE / "bench_trace.py", Path(WRIGHT).resolve().parent): self.assertIn(needed.resolve(), allowed) self.assertNotIn(agent_bench.HERE / "bench_grade.py", allowed) + self.assertNotIn((agent_bench.HERE / "oracle").resolve(), allowed) # the grading authority is readable only where the tool needs it + _, allowed = agent_bench.read_policy(args, {**env, "BENCH_TOOL": "overpy"}) + self.assertIn((agent_bench.HERE / "oracle").resolve(), allowed) @unittest.skipUnless(sys.platform == "darwin" and shutil.which("sandbox-exec"), "macOS file sandbox") def test_the_agent_sees_only_its_run_the_allowed_paths_and_the_shim(self): @@ -306,13 +310,14 @@ def test_the_agent_sees_only_its_run_the_allowed_paths_and_the_shim(self): granted.mkdir() (granted / "note.txt").write_text("a file the adapter needs") probes = {"unlisted": other / "note.txt", "granted": granted / "note.txt", "grader": agent_bench.HERE / "bench_grade.py", "shim": agent_bench.HERE / "bench_trace.py", - "answer": agent_bench.SCENARIOS / SCENARIO / "reference" / "mode.ws", "oracle": agent_bench.HERE / "oracle" / "compile.js"} + "answer": agent_bench.SCENARIOS / SCENARIO / "reference" / "mode.ws", "oracle": agent_bench.HERE / "oracle" / "compile.js", + "profile": self.out / f"{SCENARIO}-wright" / "agent.sb"} # the sandbox profile itself, which lists the hidden paths code = ("import json\nfrom pathlib import Path\nout = {}\n" + "".join(f"try:\n Path({str(p)!r}).read_text(); out[{k!r}] = 'read'\nexcept PermissionError:\n out[{k!r}] = 'blocked'\n" for k, p in probes.items()) + "Path('probe.json').write_text(json.dumps(out))\n") result = self.trial(f'{shlex.quote(sys.executable)} -c {shlex.quote(code)}', file_sandbox=True, allow_read=[str(granted)]) seen = json.loads((self.out / f"{SCENARIO}-wright/workspace/probe.json").read_text()) - self.assertEqual(seen, {"unlisted": "blocked", "granted": "read", "grader": "blocked", "shim": "read", "answer": "blocked", "oracle": "read"}) + self.assertEqual(seen, {"unlisted": "blocked", "granted": "read", "grader": "blocked", "shim": "read", "answer": "blocked", "oracle": "blocked", "profile": "blocked"}) # wright cells must not read the grading authority, and no cell reads its own sandbox profile self.assertEqual(result["fileReadEnforcement"]["mode"], "allow-list") def test_the_tool_shim_runs_without_the_grader(self): From 90e63774ee5072b30fce6b633b48198a9be7b334 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 12:42:01 +0800 Subject: [PATCH 27/32] fix(bench): match --only on the adapter alone or adapter:model exactly cmd_suite filtered with startswith, so --only codex:gpt-6 also selected codex:gpt-6-luna. Selection is now exact against the adapter name or the adapter:model pair. --- benchmarks/agent/agent_bench.py | 2 +- benchmarks/agent/test_agent_bench.py | 7 +++++++ 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index b92223db..533e272e 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -701,7 +701,7 @@ def cmd_suite(args: argparse.Namespace) -> int: """Evaluate every model of the user's list one after another, then write the publishable results page. Safe to repeat: finished runs are skipped, so quota or time limits only postpone the rest. Exit 3 when something is waiting for a rerun.""" - models = [m for m in (args.models or []) if not args.only or any(f"{m['adapter']}:{m['model']}".startswith(o) for o in args.only)] + models = [m for m in (args.models or []) if not args.only or any(m["adapter"] == o or f"{m['adapter']}:{m['model']}" == o for o in args.only)] # exact adapter or adapter:model — 'codex:gpt-6' must not swallow 'codex:gpt-6-luna' if not models: raise SystemExit(f'no models: add "models": [{{"adapter": "devin", "model": "swe-2-max"}}, ...] to {CONFIG_PATH}') root = args.out / args.suite_name diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index f2563001..ec7e6594 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -367,6 +367,13 @@ def evaluate(sub): with patch.object(agent_bench, "cmd_evaluate", evaluate): only = agent_bench.cmd_suite(argparse.Namespace(models=models, only=["devin"], out=self.out, suite_name="s2", dry_run=True)) self.assertEqual(only, 0) + calls.clear() # --only matches the adapter alone or adapter:model exactly — a prefix must not pick another model + with patch.object(agent_bench, "cmd_evaluate", evaluate): + only = agent_bench.cmd_suite(argparse.Namespace(models=models, only=["codex:gpt-6-luna"], out=self.out, suite_name="s4", dry_run=True)) + self.assertEqual(only, 3) # only codex ran, and it is waiting for a rerun + self.assertEqual([c[1] for c in calls], ["codex-gpt-6-luna-xhigh"]) + with self.assertRaisesRegex(SystemExit, "no models"): + agent_bench.cmd_suite(argparse.Namespace(models=models, only=["codex:gpt-6"], out=self.out, suite_name="s5", dry_run=True)) with patch.object(agent_bench, "cmd_evaluate", return_value=1): failed = agent_bench.cmd_suite(argparse.Namespace(models=[{"adapter": "devin", "model": "m"}], only=None, out=self.out, suite_name="s3", dry_run=True)) self.assertEqual(failed, 1) # a model that finished with errors fails the suite, it does not pass silently From 3ca0c2765b9419e0cb84bd437166d34515c41588 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 13:05:38 +0800 Subject: [PATCH 28/32] fix(bench): drain adapter stderr concurrently with stdout the adapters that piped stderr read it only after stdout closed, so a child that wrote more than a pipe buffer of stderr (or any stderr before consuming a large stdin prompt) blocked forever. drain() reads the pipe on a thread and returns its text after wait(), the same shape as feed_stdin; codex/agy already redirect stderr to a log file and devin/direct use run()/communicate(). --- benchmarks/agent/adapters/claude_code.py | 5 +++-- benchmarks/agent/adapters/common.py | 15 +++++++++++++++ benchmarks/agent/adapters/grok.py | 5 +++-- benchmarks/agent/adapters/opencode.py | 5 +++-- benchmarks/agent/adapters/pi.py | 5 +++-- benchmarks/agent/test_adapters.py | 14 ++++++++++++++ 6 files changed, 41 insertions(+), 8 deletions(-) diff --git a/benchmarks/agent/adapters/claude_code.py b/benchmarks/agent/adapters/claude_code.py index 16a614b7..db1bf61b 100755 --- a/benchmarks/agent/adapters/claude_code.py +++ b/benchmarks/agent/adapters/claude_code.py @@ -20,7 +20,7 @@ import time from pathlib import Path -from common import as_dict, cli_version, feed_stdin, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, drain, feed_stdin, INFRA_EXIT, TRANSIENT TOOLS = ["Bash", "Read", "Edit", "Write", "Glob", "Grep"] WEB_TOOLS = ["WebFetch", "WebSearch"] @@ -57,6 +57,7 @@ def main() -> int: env={**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "CLAUDE_CODE_DISABLE_CLAUDE_MDS": "1"}, ) feed_stdin(proc, prompt) + stderr_text = drain(proc.stderr) final, errored = "", False with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: # line-buffered: a killed run keeps its usage for line in proc.stdout: @@ -81,7 +82,7 @@ def main() -> int: if event.get("type") == "result": result_value = event.get("result") final, errored = (result_value if isinstance(result_value, str) else final), bool(event.get("is_error")) - stderr = proc.stderr.read() + stderr = stderr_text() code = proc.wait() Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded})) Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "claude-code", "version": cli_version(claude), "model": init.get("model") or env.get("BENCH_MODEL", "sonnet"), "tools": init.get("tools") or TOOLS + (WEB_TOOLS if web else []), "toolsSource": "init" if init.get("tools") else "requested", "mcpServers": init.get("mcp_servers")}, indent=2)) diff --git a/benchmarks/agent/adapters/common.py b/benchmarks/agent/adapters/common.py index bae7bac8..8b9a72e8 100644 --- a/benchmarks/agent/adapters/common.py +++ b/benchmarks/agent/adapters/common.py @@ -4,6 +4,7 @@ import subprocess import threading +from collections.abc import Callable INFRA_EXIT = 75 # EX_TEMPFAIL: a provider or infrastructure failure, not an agent failure; the harness retries the trial later # Error text that means the provider, not the agent, failed. Shared so the classification cannot drift between adapters. @@ -27,6 +28,20 @@ def feed() -> None: threading.Thread(target=feed, daemon=True).start() +def drain(stream) -> Callable[[], str]: + """Read a child pipe to EOF on a thread and return a callable giving its text — a pipe read only after stdout closes + leaves the child blocked on a full buffer. Call the result after proc.wait().""" + lines: list[str] = [] + thread = threading.Thread(target=lambda: lines.extend(stream), daemon=True) + thread.start() + + def text() -> str: + thread.join() + return "".join(lines) + + return text + + def cli_version(binary: str, env: dict | None = None) -> str | None: """First line of ` --version`, or None when the CLI does not answer.""" try: diff --git a/benchmarks/agent/adapters/grok.py b/benchmarks/agent/adapters/grok.py index d4fedb7f..d6dfef22 100644 --- a/benchmarks/agent/adapters/grok.py +++ b/benchmarks/agent/adapters/grok.py @@ -18,7 +18,7 @@ import time from pathlib import Path -from common import as_dict, cli_version, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, drain, INFRA_EXIT, TRANSIENT def usage_row(usage: dict, limit: int | None, now: float) -> dict: @@ -49,6 +49,7 @@ def main() -> int: cmd += ["--reasoning-effort", env["BENCH_THINKING"]] version = cli_version(grok, child_env) proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=child_env) + stderr_text = drain(proc.stderr) init: dict = {} pending: list[dict] = [] final, error, limit = "", "", None @@ -76,7 +77,7 @@ def main() -> int: limit = next((m.get("contextWindow") for m in as_dict(event.get("modelUsage")).values()), None) for u in pending: # the context window is only known from the final result line usage.write(json.dumps(usage_row(u, limit, time.time())) + "\n") - stderr = proc.stderr.read() + stderr = stderr_text() code = proc.wait() Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": init.get("skills") or []})) Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "grok", "version": version, "model": init.get("model") or model, "effort": env.get("BENCH_THINKING"), "tools": init.get("tools"), diff --git a/benchmarks/agent/adapters/opencode.py b/benchmarks/agent/adapters/opencode.py index 6301df39..33cf2ad0 100644 --- a/benchmarks/agent/adapters/opencode.py +++ b/benchmarks/agent/adapters/opencode.py @@ -18,7 +18,7 @@ import time from pathlib import Path -from common import as_dict, cli_version, feed_stdin, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, drain, feed_stdin, INFRA_EXIT, TRANSIENT WEB_PERMISSIONS = {"webfetch": "deny", "websearch": "deny"} @@ -61,6 +61,7 @@ def main() -> int: if env.get("BENCH_THINKING"): cmd += ["--variant", env["BENCH_THINKING"]] proc = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=child_env) + stderr_text = drain(proc.stderr) feed_stdin(proc, sys.stdin.read()) final, error, tools = "", "", set() with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: # line-buffered: a killed run keeps its usage @@ -81,7 +82,7 @@ def main() -> int: tools.add(part["tool"]) elif etype == "error": error = json.dumps(event.get("error")) - stderr = proc.stderr.read() + stderr = stderr_text() code = proc.wait() Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded})) Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "opencode", "version": version, "model": model, "effort": env.get("BENCH_THINKING"), "tools": sorted(tools), "toolsNote": "tools the agent used; the CLI does not list its tools", "denied": [] if web else sorted(WEB_PERMISSIONS)}, indent=2)) diff --git a/benchmarks/agent/adapters/pi.py b/benchmarks/agent/adapters/pi.py index 067d4b48..0e0d2643 100755 --- a/benchmarks/agent/adapters/pi.py +++ b/benchmarks/agent/adapters/pi.py @@ -21,7 +21,7 @@ import time from pathlib import Path -from common import as_dict, cli_version, feed_stdin, INFRA_EXIT, TRANSIENT +from common import as_dict, cli_version, drain, feed_stdin, INFRA_EXIT, TRANSIENT SCALE = {"K": 1_000, "M": 1_000_000} @@ -84,6 +84,7 @@ def main() -> int: limit = context_limit(pi, model, child_env, extensions) Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "pi", "version": cli_version(pi), "model": model, "effort": env.get("BENCH_THINKING"), "tools": ["read", "bash", "edit", "write"], "extensions": [e for e in extensions if e]}, indent=2)) proc = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=child_env) + stderr_text = drain(proc.stderr) feed_stdin(proc, prompt) loaded: list[str] = [] final, error = "", "" @@ -108,7 +109,7 @@ def main() -> int: error = str(message.get("errorMessage") or "error") # only the structured error classifies; message text is the agent's own else: error = "" - stderr = proc.stderr.read() + stderr = stderr_text() code = proc.wait() Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded})) sys.stdout.write(final) diff --git a/benchmarks/agent/test_adapters.py b/benchmarks/agent/test_adapters.py index d08283e7..b5a5dc5f 100644 --- a/benchmarks/agent/test_adapters.py +++ b/benchmarks/agent/test_adapters.py @@ -233,5 +233,19 @@ def test_a_malformed_payload_is_a_clean_provider_error_not_a_traceback(self): self.assertEqual(code, 1) +class PipeTest(unittest.TestCase): + def test_drain_keeps_a_chatty_stderr_from_deadlocking_the_stdout_read(self): + # a child that floods stderr past the pipe buffer blocks unless someone drains it concurrently with stdout + import subprocess + import common + child = subprocess.Popen( + [sys.executable, "-c", "import sys; sys.stderr.write('x' * 262144); sys.stderr.flush(); print('ok')"], + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + stderr_text = common.drain(child.stderr) + self.assertEqual(child.stdout.read(), "ok\n") + child.wait() + self.assertEqual(stderr_text(), "x" * 262144) + + if __name__ == "__main__": unittest.main() From 778c8270e7e6188b9917032320883e771217331e Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 13:05:42 +0800 Subject: [PATCH 29/32] fix(bench): lock the sandbox profile so the agent cannot read it under another name the read deny on agent.sb was literal-path only, and the profile sits inside the writable run dir, so the agent could rename or hardlink it to a name the deny does not cover and read the hidden-path list. the profile now also denies file-write on its own literal (last matching rule wins over the run-dir allow, blocking rename, unlink, overwrite, and the chflags that would clear the lock), and the harness sets UF_IMMUTABLE on it for the trial, which makes hardlinking fail outright where path rules cannot reach. the sandbox probes now try the rename and link escapes themselves. --- benchmarks/agent/agent_bench.py | 47 +++++++++++++++++++--------- benchmarks/agent/test_agent_bench.py | 7 ++++- 2 files changed, 39 insertions(+), 15 deletions(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 533e272e..11e75463 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -4,6 +4,7 @@ from __future__ import annotations import argparse +import contextlib import hashlib import json import os @@ -21,6 +22,8 @@ from datetime import datetime, timezone from pathlib import Path +import stat + import bench_grade import bench_leaderboard import bench_report @@ -297,31 +300,47 @@ def run_agent(args: argparse.Namespace, env: dict, workspace: Path, prompt: str) env = {**env, "TMPDIR": str(temporary), "PYTHONDONTWRITEBYTECODE": "1"} rules = sandbox_read_rules(*read_policy(args, env)) profile = run_dir / "agent.sb" + if profile.exists(): + os.chflags(profile, 0) # an attempt killed mid-run leaves it locked + # the profile lists what is hidden, so the agent must not read it — and it sits inside the writable run dir, so a + # bare read deny is renamed around. the write deny keeps the name bound; UF_IMMUTABLE backs it where path rules + # cannot reach (hardlinking an immutable file fails outright), and the write deny in turn blocks the chflags + # that would clear the flag. + locked = str(profile.resolve()) profile.write_text('(version 1)\n(allow default)\n(deny file-write*)\n' f'(allow file-write* (subpath {json.dumps(str(run_dir.resolve()))}) (subpath "/dev"))\n' '(deny file-read-data (require-all (regex "/(AGENTS|CLAUDE|GEMINI)[.]md$") ' f'(require-not (subpath {json.dumps(str(workspace.resolve()))}))))\n' + rules - + f'(deny file-read-data (literal {json.dumps(str(profile.resolve()))}))\n') # the profile lists what is hidden; the agent must not read it + + f'(deny file-read-data (literal {json.dumps(locked)}))\n' + + f'(deny file-write* (literal {json.dumps(locked)}))\n') + os.chflags(profile, stat.UF_IMMUTABLE) command = ["sandbox-exec", "-f", str(profile), "/bin/sh", "-c", args.agent_cmd] + else: + profile = None proc = subprocess.Popen( command, shell=isinstance(command, str), cwd=workspace, env=env, text=True, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, start_new_session=True, ) try: - stdout, stderr = proc.communicate(input=prompt, timeout=args.timeout) - return proc.returncode, stdout, stderr - except subprocess.TimeoutExpired: - try: - os.killpg(proc.pid, signal.SIGKILL) - except ProcessLookupError: - pass try: - stdout, stderr = proc.communicate(timeout=10) - except subprocess.TimeoutExpired: # a detached grandchild still holds the pipes; the adapter's own transcript has the record - proc.stdout.close() - proc.stderr.close() - stdout, stderr = "", "" - return None, stdout, f"{stderr}\ntimeout" if stderr else "timeout" + stdout, stderr = proc.communicate(input=prompt, timeout=args.timeout) + return proc.returncode, stdout, stderr + except subprocess.TimeoutExpired: + try: + os.killpg(proc.pid, signal.SIGKILL) + except ProcessLookupError: + pass + try: + stdout, stderr = proc.communicate(timeout=10) + except subprocess.TimeoutExpired: # a detached grandchild still holds the pipes; the adapter's own transcript has the record + proc.stdout.close() + proc.stderr.close() + stdout, stderr = "", "" + return None, stdout, f"{stderr}\ntimeout" if stderr else "timeout" + finally: + if profile is not None: + with contextlib.suppress(OSError): + os.chflags(profile, 0) def context_report(out: Path, skill_names: list[str]) -> dict: diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index ec7e6594..f8b51ea8 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -314,10 +314,15 @@ def test_the_agent_sees_only_its_run_the_allowed_paths_and_the_shim(self): "profile": self.out / f"{SCENARIO}-wright" / "agent.sb"} # the sandbox profile itself, which lists the hidden paths code = ("import json\nfrom pathlib import Path\nout = {}\n" + "".join(f"try:\n Path({str(p)!r}).read_text(); out[{k!r}] = 'read'\nexcept PermissionError:\n out[{k!r}] = 'blocked'\n" for k, p in probes.items()) + # a read deny on the profile's own path is not enough: inside the writable run dir the agent can move or + # link the file to a name the deny does not cover, so the probes must try the rename and link themselves + + f"try:\n moved = Path('agent-moved.sb')\n Path({str(probes['profile'])!r}).rename(moved)\n moved.read_text(); out['profile-renamed'] = 'read'\nexcept PermissionError:\n out['profile-renamed'] = 'blocked'\n" + + f"try:\n linked = Path('agent-linked.sb')\n linked.hardlink_to({str(probes['profile'])!r})\n linked.read_text(); out['profile-linked'] = 'read'\nexcept PermissionError:\n out['profile-linked'] = 'blocked'\n" + "Path('probe.json').write_text(json.dumps(out))\n") result = self.trial(f'{shlex.quote(sys.executable)} -c {shlex.quote(code)}', file_sandbox=True, allow_read=[str(granted)]) seen = json.loads((self.out / f"{SCENARIO}-wright/workspace/probe.json").read_text()) - self.assertEqual(seen, {"unlisted": "blocked", "granted": "read", "grader": "blocked", "shim": "read", "answer": "blocked", "oracle": "blocked", "profile": "blocked"}) # wright cells must not read the grading authority, and no cell reads its own sandbox profile + self.assertEqual(seen, {"unlisted": "blocked", "granted": "read", "grader": "blocked", "shim": "read", "answer": "blocked", "oracle": "blocked", + "profile": "blocked", "profile-renamed": "blocked", "profile-linked": "blocked"}) # wright cells must not read the grading authority, and no cell reads its own sandbox profile self.assertEqual(result["fileReadEnforcement"]["mode"], "allow-list") def test_the_tool_shim_runs_without_the_grader(self): From e87deacabeaf541aa8438942eb7a32c1360312d4 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 13:16:53 +0800 Subject: [PATCH 30/32] fix(bench): let trial cleanup remove files the agent locked or hid a run killed between setting agent.sb's immutable flag and clearing it left rmtree unable to remove the run dir, and the next trial failed at mkdir. the agent itself can also chflags or chmod anything inside its writable tree. drop() now restores writability at the failing node and retries; rmtree's onexc does the same inside each tree. --- benchmarks/agent/agent_bench.py | 27 +++++++++++++++++++++++---- benchmarks/agent/test_agent_bench.py | 18 ++++++++++++++++++ 2 files changed, 41 insertions(+), 4 deletions(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 11e75463..6301f424 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -343,6 +343,26 @@ def run_agent(args: argparse.Namespace, env: dict, workspace: Path, prompt: str) os.chflags(profile, 0) +def drop(path: Path) -> None: + """Best-effort removal of a path inside the agent-writable run tree. The agent can chflags or chmod its own files — + and a run killed mid-trial can leave agent.sb locked — so a plain rmtree/unlink can hit EPERM; restore writability + at the failing node and retry rather than leave the run directory permanently unreusable.""" + def unlock_and_retry(func, node, _exc): + with contextlib.suppress(OSError, AttributeError): # no chflags outside the BSDs + os.chflags(node, 0) + with contextlib.suppress(OSError): + os.chmod(node, 0o700) + with contextlib.suppress(OSError): + func(node) + if path.is_dir() and not path.is_symlink(): + shutil.rmtree(path, onexc=unlock_and_retry) + else: + with contextlib.suppress(OSError, AttributeError): + os.chflags(path, 0) + with contextlib.suppress(OSError): + path.unlink() + + def context_report(out: Path, skill_names: list[str]) -> dict: path = out / "context.json" if not path.is_file(): @@ -359,18 +379,17 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - check_cell(cell, args) if not applicable(scenario, cell): raise SystemExit(f"condition {cell_label(cell)} does not apply to the {scenario['language']} scenario {scenario['id']}") - shutil.rmtree(out, ignore_errors=True) + drop(out) out.mkdir(parents=True) workspace = out / "workspace" prompt = (scenario["dir"] / "prompt.md").read_text() # the exact pinned prompt: the harness adds no text infra_retries = 0 while True: - shutil.rmtree(workspace, ignore_errors=True) + drop(workspace) for stale in ("home", "bin", "real", "tool-trace.jsonl", "tool-calls", "usage.jsonl", "transcript.jsonl", "context.json", "agent-info.json", "snapshots", "tmp", "devin-export.json", # devin only writes this on success; a retry that produces none would parse the previous attempt's export *(child.name for child in out.iterdir() if child.name.endswith(("-home", "-user")))): # adapters keep their isolated homes here; a retry starts from none - target = out / stale - shutil.rmtree(target, ignore_errors=True) if target.is_dir() else target.unlink(missing_ok=True) + drop(out / stale) materialize(scenario, workspace) if cell["knowledge"] == "wiki": shutil.copytree(Path(args.wiki_dir), workspace / "wiki") # a real copy: tools that skip symlinks (rg) must see it diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index f8b51ea8..5f5d4153 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -1,8 +1,10 @@ import argparse +import contextlib import hashlib import json import os import shutil +import stat import sys import tempfile import unittest @@ -325,6 +327,22 @@ def test_the_agent_sees_only_its_run_the_allowed_paths_and_the_shim(self): "profile": "blocked", "profile-renamed": "blocked", "profile-linked": "blocked"}) # wright cells must not read the grading authority, and no cell reads its own sandbox profile self.assertEqual(result["fileReadEnforcement"]["mode"], "allow-list") + @unittest.skipUnless(sys.platform == "darwin", "file flags are the macOS enforcement") + def test_a_locked_profile_left_by_a_killed_run_does_not_block_the_next_trial(self): + out = self.out / f"{SCENARIO}-wright" + locked = {"profile": out / "agent.sb", "agent-file": out / "workspace" / "agent-locked.txt"} + for name, path in locked.items(): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(f"left by a run killed before cleanup ({name})") + os.chflags(path, stat.UF_IMMUTABLE) # the harness locks agent.sb; the agent can lock anything inside its run dir + try: + result = self.trial("true") + self.assertNotIn("invalid", result) + finally: + for path in locked.values(): # if drop left them, free the test's own cleanup + with contextlib.suppress(OSError): + os.chflags(path, 0) + def test_the_tool_shim_runs_without_the_grader(self): shim = (agent_bench.HERE / "bench_trace.py").read_text() self.assertNotIn("import bench_grade", shim) From 9e29351f1c3fc9c6e3ea187198824d23792b4b2f Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 13:23:09 +0800 Subject: [PATCH 31/32] fix(bench): keep cleanup repairs on the run-dir node and retry by pass retrying the failed op in place broke on traversal callbacks like os.open, which take more than the path; and chflags/chmod follow symlinks by default, so a planted stale symlink could steer the repairs onto a host file. leaf removals still retry immediately, a blocked traversal is repaired and picked up by the next pass, and the repairs no longer follow links. --- benchmarks/agent/agent_bench.py | 34 ++++++++++++----------- benchmarks/agent/test_agent_bench.py | 41 ++++++++++++++++++++++++++++ 2 files changed, 59 insertions(+), 16 deletions(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 6301f424..15e4c87a 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -345,22 +345,24 @@ def run_agent(args: argparse.Namespace, env: dict, workspace: Path, prompt: str) def drop(path: Path) -> None: """Best-effort removal of a path inside the agent-writable run tree. The agent can chflags or chmod its own files — - and a run killed mid-trial can leave agent.sb locked — so a plain rmtree/unlink can hit EPERM; restore writability - at the failing node and retry rather than leave the run directory permanently unreusable.""" - def unlock_and_retry(func, node, _exc): - with contextlib.suppress(OSError, AttributeError): # no chflags outside the BSDs - os.chflags(node, 0) - with contextlib.suppress(OSError): - os.chmod(node, 0o700) - with contextlib.suppress(OSError): - func(node) - if path.is_dir() and not path.is_symlink(): - shutil.rmtree(path, onexc=unlock_and_retry) - else: - with contextlib.suppress(OSError, AttributeError): - os.chflags(path, 0) - with contextlib.suppress(OSError): - path.unlink() + and a run killed mid-trial can leave agent.sb locked — so a plain rmtree/unlink can hit EPERM. Repairs stay on the + run-dir node itself so a planted symlink cannot redirect them onto a host file; a leaf removal retries right away, + and the next pass reaches whatever a blocked traversal skipped.""" + def unlock(func, node, _exc): + with contextlib.suppress(OSError, AttributeError, NotImplementedError): # chflags is a BSD mechanism + os.chflags(node, 0, follow_symlinks=False) + with contextlib.suppress(OSError, NotImplementedError): + os.chmod(node, 0o700, follow_symlinks=False) + if func in (os.unlink, os.rmdir): + with contextlib.suppress(OSError): + func(node) + for _ in range(8): + if not os.path.lexists(path): + return + if path.is_dir() and not path.is_symlink(): + shutil.rmtree(path, onexc=unlock) + else: + unlock(os.unlink, str(path), None) def context_report(out: Path, skill_names: list[str]) -> dict: diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 5f5d4153..fdfbb30c 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -564,6 +564,47 @@ def test_oracle_disagreement_is_reported(self): self.assertEqual(auth["disagreement"]["kind"], "wright-accepts-oracle-rejects") +@unittest.skipUnless(sys.platform == "darwin", "file flags are the macOS enforcement") +class DropTest(unittest.TestCase): + def setUp(self): + (agent_bench.ROOT / "target").mkdir(exist_ok=True) + self.out = Path(tempfile.mkdtemp(dir=agent_bench.ROOT / "target")).resolve() + self.addCleanup(shutil.rmtree, self.out, True) + + def test_repairs_never_reach_through_a_symlink_to_a_host_file(self): + host_file = self.out / "host-file.txt" + host_file.write_text("host data the run must not mutate") + host_file.chmod(0o400) + os.chflags(host_file, stat.UF_IMMUTABLE) + link = self.out / "run" / "stale-link" + link.parent.mkdir(parents=True) + link.symlink_to(host_file) + try: + agent_bench.drop(link) + self.assertFalse(os.path.lexists(link)) + self.assertTrue(os.lstat(host_file).st_flags & stat.UF_IMMUTABLE, "the link redirected the flag repair onto the host file") + self.assertEqual(host_file.stat().st_mode & 0o777, 0o400) + finally: + with contextlib.suppress(OSError): + os.chflags(host_file, 0) + host_file.chmod(0o600) + + def test_nested_chmodded_directories_come_down_a_level_per_pass(self): + root = self.out / "run" / "denied" + inner = root / "inner" + inner.mkdir(parents=True) + (inner / "left.txt").write_text("agent-owned") + for directory in (inner, root): + directory.chmod(0) # the agent can make its own directories untraversable — deepest first, the parent must stay resolvable + try: + agent_bench.drop(self.out / "run") + self.assertFalse(os.path.lexists(self.out / "run")) + finally: + for directory in (root, inner): + with contextlib.suppress(OSError): + directory.chmod(0o700) + + class DetectorTest(unittest.TestCase): def call(self, argv, exit_code=0, t=0.0, **extra): return {"tool": "wright", "type": "call", "t": t, "argv": argv, "exit": exit_code, "seconds": 0.1, "stdoutBytes": 40, "stderrBytes": 0, "stderrHead": "", "envelope": None, **extra} From 6d3b205ce83ef997dcb42e64f2e5275bca7f1829 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 14:00:55 +0800 Subject: [PATCH 32/32] fix(bench): harden run startup and resume against reviewer findings MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - wright_mismatch returns early when the binary is missing, so a nonexistent --wright reports through preflight's "cannot start" list instead of crashing evaluate/suite with FileNotFoundError - result.json is written atomically (temp + os.replace); a partial file left by a mid-write kill is treated as unfinished and retried by matrix instead of crashing pool.map with JSONDecodeError — the same tolerance now covers wright_mismatch's environment scan - --suite-name takes the same single-directory-name check as --name, closing the ".." escape under --out - file_sha256's signature admits the Path callers actually pass --- benchmarks/agent/agent_bench.py | 27 ++++++++++++++---- benchmarks/agent/test_agent_bench.py | 42 ++++++++++++++++++++++++++++ 2 files changed, 64 insertions(+), 5 deletions(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 15e4c87a..dac30b6d 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -400,7 +400,7 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - if reason: result = base_result(scenario, cell, args, out, 0.0, None) result.update(invalid=reason, status="invalid") - (out / "result.json").write_text(json.dumps(result, indent=2) + "\n") + write_json(out / "result.json", result) return result snapshots = bench_trace.Snapshots(workspace, scenario.get("watch", [scenario["entry"]]), out / "snapshots") snapshots.start() @@ -443,7 +443,7 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - first_valid = next((s["t"] for s in result["snapshots"]["series"] if s["valid"]), None) result["usage"] = bench_trace.usage_summary(out / "usage.jsonl", first_valid) result["status"] = run_status(result) - (out / "result.json").write_text(json.dumps(result, indent=2) + "\n") + write_json(out / "result.json", result) return result @@ -475,11 +475,17 @@ def harness_commit() -> str: return proc.stdout.strip() + ("+dirty" if dirty else "") -def file_sha256(path: str) -> str: +def file_sha256(path: str | Path) -> str: """Re-hashed per call: a suite resumes for days in one process and the binary may be rebuilt between trials.""" return hashlib.sha256(Path(path).read_bytes()).hexdigest() +def write_json(path: Path, data: dict) -> None: + """A kill mid-write leaves no partial JSON to crash the next resume: temp file in the same directory, then replace.""" + temporary = path.with_name(f".{path.name}.tmp") + temporary.write_text(json.dumps(data, indent=2) + "\n") + os.replace(temporary, path) + def base_result(scenario: dict, cell: dict, args: argparse.Namespace, out: Path, seconds: float, agent_exit: int | None) -> dict: return { @@ -574,7 +580,11 @@ def work(job: tuple) -> None: scenario_id, agent, cell, trial = job out = trial_dir(merged["out"], scenario_id, agent["id"], cell, trial) finished = out / "result.json" - if finished.is_file() and json.loads(finished.read_text()).get("status") != "provider-interrupted": # an interrupted trial is retried on the next run + try: + status = json.loads(finished.read_text()).get("status") if finished.is_file() else None + except json.JSONDecodeError: + status = None # a mid-write kill left a partial file; the trial is unfinished and retried + if status is not None and status != "provider-interrupted": # an interrupted trial is retried on the next run return with lock: if state["streak"] >= STOP_AFTER_INTERRUPTIONS: @@ -657,9 +667,14 @@ def wright_mismatch(run_dir: Path, wright: str) -> str | None: A run is repeated to finish it later, possibly after `wright` on PATH was upgraded. Trials made with two binaries cannot be scored together, so the mismatch is refused up front instead of being found when the score card is refused.""" + if not Path(wright).is_file(): + return None # preflight already names the missing binary current = file_sha256(Path(wright)) for path in sorted(run_dir.glob("*/*/*/result.json")): - env = json.loads(path.read_text()).get("environment", {}) + try: + env = json.loads(path.read_text()).get("environment", {}) + except json.JSONDecodeError: + continue # a partial file means the trial never finished; it will be retried recorded = env.get("wrightSha256") if recorded and recorded != current: return (f"{run_dir.name} already has trials made with {env.get('wright')} (sha256 {recorded[:12]}), but {wright} is a different binary. " @@ -744,6 +759,8 @@ def cmd_suite(args: argparse.Namespace) -> int: models = [m for m in (args.models or []) if not args.only or any(m["adapter"] == o or f"{m['adapter']}:{m['model']}" == o for o in args.only)] # exact adapter or adapter:model — 'codex:gpt-6' must not swallow 'codex:gpt-6-luna' if not models: raise SystemExit(f'no models: add "models": [{{"adapter": "devin", "model": "swe-2-max"}}, ...] to {CONFIG_PATH}') + if not args.suite_name or args.suite_name in (".", "..") or Path(args.suite_name).name != args.suite_name: + raise SystemExit("--suite-name must be a single directory name") root = args.out / args.suite_name outcome = {} for entry in models: diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index fdfbb30c..62443ede 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -169,6 +169,25 @@ def test_matrix_skips_inapplicable_cells_and_stops_after_repeated_interruptions( self.assertTrue(all(json.loads(p.read_text())["status"] != "provider-interrupted" for p in (self.out / "m").rglob("result.json"))) self.assertEqual(len(list((self.out / "m").rglob("result.json"))), 4) # the interrupted two were rerun and the rest ran + def test_a_partial_result_from_a_killed_run_is_retried_instead_of_crashing(self): + import io + cell = agent_bench.normalize_cell({"tool": "none", "skills": [], "knowledge": "none", "network": "off"}) + config = self.out / "matrix.json" + config.write_text(json.dumps({ + "agents": [{"id": "fake", "cmd": "exit 0"}], + "cells": [cell], "scenarios": [SCENARIO], "trials": 1, "parallel": 1, "seed": 1, + "options": {"timeout": 30, "infra_retries": 0}, + })) + out = agent_bench.trial_dir(self.out / "m", SCENARIO, "fake", cell, 1) + out.mkdir(parents=True) + (out / "result.json").write_text('{"status": "com') # a kill mid-write left this + args = argparse.Namespace(config=config, out=self.out / "m", wright=str(Path(WRIGHT).resolve()), skill_dirs={}, wiki_dir=None, env_pass=[], + check_ancestors=False, canary_cmd=None, timeout=30, infra_retries=0, file_sandbox=False) + with contextlib.redirect_stdout(io.StringIO()): + code = agent_bench.cmd_matrix(args) + self.assertEqual(code, 0) + self.assertEqual(json.loads((out / "result.json").read_text())["status"], "completed") + def test_evaluate_serializes_the_run_options_so_its_matrix_reproduces_the_run(self): import contextlib import io @@ -213,6 +232,21 @@ def test_evaluate_serializes_the_run_options_so_its_matrix_reproduces_the_run(se self.assertEqual((result["protocol"]["timeoutSeconds"], result["fileWriteEnforcement"]), (30, "unrestricted")) self.assertEqual((trial / "workspace" / "probe.txt").read_text(), f"present|{Path.home()}") # env_pass reached the agent again + def test_evaluate_reports_a_missing_wright_binary_instead_of_crashing(self): + import io + self.skill_dir("wright-skill") + args = argparse.Namespace( + adapter="fake", model="m", effort=None, name="eval", out=self.out, wright=str(self.out / "no-such-wright"), + skill_dirs={"wright-skill": self.out / "skills" / "wright-skill"}, wiki_dir=None, env_pass=[], + allow_read=[], deny_read=[], canary_cmd=None, check_ancestors=False, + timeout=30, infra_retries=0, infra_backoff=0, no_file_sandbox=True, dry_run=False, + cells="score", split="test", scenarios=[SCENARIO], trials=1, parallel=1, seed=1) + with patch.dict(agent_bench.ADAPTERS, {"fake": "x.py"}), patch.dict(agent_bench.ADAPTER_READS, {"fake": []}), \ + contextlib.redirect_stdout(io.StringIO()), self.assertRaises(SystemExit) as stop: + agent_bench.cmd_evaluate(args) + self.assertIn("cannot start", str(stop.exception.code)) + self.assertIn("wright binary not found", str(stop.exception.code)) + def test_scenarios_are_solvable_and_not_vacuous(self): self.assertTrue(agent_bench.validate(WRIGHT, self.out / "validate")) @@ -401,6 +435,11 @@ def evaluate(sub): failed = agent_bench.cmd_suite(argparse.Namespace(models=[{"adapter": "devin", "model": "m"}], only=None, out=self.out, suite_name="s3", dry_run=True)) self.assertEqual(failed, 1) # a model that finished with errors fails the suite, it does not pass silently + models = [{"adapter": "devin", "model": "m"}] + for bad in ("..", "a/b", ""): + with self.assertRaises(SystemExit): + agent_bench.cmd_suite(argparse.Namespace(models=models, only=None, out=self.out, suite_name=bad, dry_run=True)) + def test_a_repeated_run_refuses_a_different_wright_binary(self): run = self.out / "r" trial = run / "s" / "agent" / "cell-1" @@ -414,6 +453,9 @@ def test_a_repeated_run_refuses_a_different_wright_binary(self): self.assertIsNone(agent_bench.wright_mismatch(self.out / "fresh", str(other))) # a new run has nothing to disagree with (trial / "result.json").write_text(json.dumps({"environment": {"wright": "x", "wrightSha256": agent_bench.file_sha256(other)}})) self.assertIsNone(agent_bench.wright_mismatch(run, str(other))) + self.assertIsNone(agent_bench.wright_mismatch(run, str(self.out / "no-such-binary"))) # preflight names the missing binary; hashing it must not crash first + (trial / "result.json").write_text('{"environment": {"wrightSha') # a mid-write kill left a partial file: the trial is unfinished, not evidence + self.assertIsNone(agent_bench.wright_mismatch(run, str(other))) def test_an_effort_the_model_id_already_names_is_not_repeated_in_the_run_name(self): self.assertEqual(agent_bench.model_slug({"adapter": "agy", "model": "gemini-3.8-flash-high", "effort": "high"}), "agy-gemini-3.8-flash-high")