diff --git a/benchmarks/agent/adapters/agy.py b/benchmarks/agent/adapters/agy.py index eab1d515..1f84a75f 100644 --- a/benchmarks/agent/adapters/agy.py +++ b/benchmarks/agent/adapters/agy.py @@ -11,6 +11,14 @@ import time from pathlib import Path +from common import as_dict, cli_version, split_effort, INFRA_EXIT, TRANSIENT + + +def transient(error: str) -> bool: + """A provider failure. The CLI reports a dropped connection as `API error (attempt N): request failed: ... EOF`, even after the answer was written.""" + lowered = error.lower() + return any(s in lowered for s in TRANSIENT) or "api error (attempt" in lowered or "request failed" in lowered + def usage_row(usage: dict, timestamp: float) -> dict: cached = usage.get("cache_read_tokens") or 0 @@ -38,41 +46,47 @@ def main() -> int: shutil.copytree(skill, Path.cwd() / ".agents/skills" / skill.name) installed.append(skill.name) binary = shutil.which("agy", path=env.get("BENCH_HOST_PATH")) or "agy" - command = [binary, "--model", env["BENCH_MODEL"], "--effort", env["BENCH_THINKING"], + effort = env.get("BENCH_THINKING") + command = [binary, "--model", env["BENCH_MODEL"], *(["--effort", effort] if effort else []), "--output-format", "stream-json", "--print-timeout", "0", "--print", sys.stdin.read()] child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home)} seen, result, init, unexpected = set(), {}, {}, set() with (run / "agy-stderr.log").open("w") as stderr, open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: process = subprocess.Popen(command, env=child_env, stdout=subprocess.PIPE, stderr=stderr, text=True) for line in process.stdout: - event = json.loads(line) + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + if not isinstance(event, dict): + continue now = time.time() transcript.write(json.dumps({"t": now, **event}) + "\n") - if event["event"] == "init": - init = event["init"] - step = event.get("step_update") or {} + if event.get("event") == "init": + init = event.get("init") or {} + step = as_dict(event.get("step_update")) tool = step.get("tool_name", "") if env["BENCH_NETWORK"] == "off" and tool in {"search_web", "read_url_content", "browser_subagent", "open_browser_url"}: unexpected.add("network-tool:" + tool) process.terminate() - if step.get("state") == "DONE" and step.get("usage") and step["step_index"] not in seen: - seen.add(step["step_index"]) + if step.get("state") == "DONE" and step.get("usage") and (index := step.get("step_index")) is not None and index not in seen: + seen.add(index) usage.write(json.dumps(usage_row(step["usage"], now)) + "\n") - if event["event"] == "result": - result = event["result"] + if event.get("event") == "result": + result = as_dict(event.get("result")) code = process.wait() Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"installed": installed, "audit": "isolated-home; CLI does not export loaded skill context", "unexpected": sorted(unexpected)})) - Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "agy", "model": env["BENCH_MODEL"], "effort": env["BENCH_THINKING"], "tools": None, "note": "the CLI does not expose its tool list"}, indent=2)) - (run / "adapter.json").write_text(json.dumps({"agent": "agy", "requestedModel": env["BENCH_MODEL"], "requestedEffort": env["BENCH_THINKING"], "observedModel": init.get("model"), "status": result.get("status"), "usageSource": "stream-step-usage", "providerUsage": result.get("usage")}, indent=2)) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "agy", "version": cli_version(binary), "model": split_effort(env["BENCH_MODEL"])[0], "modelId": env["BENCH_MODEL"], "effort": effort or split_effort(env["BENCH_MODEL"])[1], "tools": None, "note": "the CLI does not expose its tool list"}, indent=2)) + (run / "adapter.json").write_text(json.dumps({"agent": "agy", "requestedModel": env["BENCH_MODEL"], "requestedEffort": effort, "observedModel": init.get("model"), "status": result.get("status"), "usageSource": "stream-step-usage", "providerUsage": result.get("usage")}, indent=2)) sys.stdout.write(result.get("response", "")) stderr_text = (run / "agy-stderr.log").read_text() sys.stderr.write(stderr_text) error = str(result.get("error", "")) + stderr_text if "no output produced" in stderr_text and "headless" in stderr_text: - return 75 + return INFRA_EXIT if code or result.get("status") != "SUCCESS": - return 75 if any(s in error.lower() for s in ("quota", "rate limit", "429", "temporarily", "503", "credits")) else (code or 1) + return INFRA_EXIT if transient(error) else (code or 1) return 0 diff --git a/benchmarks/agent/adapters/claude_code.py b/benchmarks/agent/adapters/claude_code.py index add3a71a..db1bf61b 100755 --- a/benchmarks/agent/adapters/claude_code.py +++ b/benchmarks/agent/adapters/claude_code.py @@ -13,14 +13,15 @@ import json import os +import re import shutil import subprocess import sys -import tempfile import time from pathlib import Path -INFRA_EXIT = 75 +from common import as_dict, cli_version, drain, feed_stdin, INFRA_EXIT, TRANSIENT + TOOLS = ["Bash", "Read", "Edit", "Write", "Glob", "Grep"] WEB_TOOLS = ["WebFetch", "WebSearch"] @@ -40,20 +41,23 @@ def main() -> int: loaded: list[str] = [] skills = [Path(p) for p in env.get("BENCH_SKILL_DIRS", "").split(os.pathsep) if p] if skills: - plugin = Path(tempfile.mkdtemp(dir=env["BENCH_RUN_DIR"])) - (plugin / ".claude-plugin").mkdir() + plugin = Path(env["BENCH_RUN_DIR"]) / "claude-plugin" # a fixed name: an infra retry would else orphan a mkdtemp dir + shutil.rmtree(plugin, ignore_errors=True) + (plugin / ".claude-plugin").mkdir(parents=True) (plugin / ".claude-plugin/plugin.json").write_text(json.dumps({"name": "bench", "version": "0.0.0", "description": "benchmark skills"})) for skill in skills: shutil.copytree(skill, plugin / "skills" / skill.name) - loaded.append(skill.name) + match = re.search(r"^name:\s*(\S+)", (skill / "SKILL.md").read_text(), re.M) # the name the harness checks, not the directory name + loaded.append(match.group(1) if match else skill.name) cmd += ["--plugin-dir", str(plugin)] - Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "claude-code", "model": env.get("BENCH_MODEL", "sonnet"), "tools": TOOLS + (WEB_TOOLS if web else [])}, indent=2)) + claude = cmd[0] + init: dict = {} proc = subprocess.Popen( cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env={**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "CLAUDE_CODE_DISABLE_CLAUDE_MDS": "1"}, ) - proc.stdin.write(prompt) - proc.stdin.close() + feed_stdin(proc, prompt) + stderr_text = drain(proc.stderr) final, errored = "", False with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: # line-buffered: a killed run keeps its usage for line in proc.stdout: @@ -61,9 +65,13 @@ def main() -> int: event = json.loads(line) except json.JSONDecodeError: continue + if not isinstance(event, dict): + continue transcript.write(json.dumps({"t": time.time(), **event}) + "\n") + if event.get("type") == "system" and event.get("subtype") == "init": + init = event if event.get("type") == "assistant": - u = event["message"].get("usage") or {} + u = as_dict(event.get("message")).get("usage") or {} cached = u.get("cache_read_input_tokens") or 0 written = u.get("cache_creation_input_tokens") or 0 usage.write(json.dumps({ @@ -72,15 +80,16 @@ def main() -> int: "context": (u.get("input_tokens") or 0) + cached + written, "context_limit": None, }) + "\n") if event.get("type") == "result": - final, errored = event.get("result", ""), bool(event.get("is_error")) - stderr = proc.stderr.read() + result_value = event.get("result") + final, errored = (result_value if isinstance(result_value, str) else final), bool(event.get("is_error")) + stderr = stderr_text() code = proc.wait() Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded})) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "claude-code", "version": cli_version(claude), "model": init.get("model") or env.get("BENCH_MODEL", "sonnet"), "tools": init.get("tools") or TOOLS + (WEB_TOOLS if web else []), "toolsSource": "init" if init.get("tools") else "requested", "mcpServers": init.get("mcp_servers")}, indent=2)) sys.stdout.write(final) sys.stderr.write(stderr) if errored or code != 0: - transient = any(s in stderr.lower() + final.lower() for s in ("overloaded", "rate limit", "529", "timed out")) - return INFRA_EXIT if transient else (code or 1) + return INFRA_EXIT if any(s in stderr.lower() for s in TRANSIENT) else (code or 1) # `final` is the agent's own text; only provider channels classify return 0 diff --git a/benchmarks/agent/adapters/codex.py b/benchmarks/agent/adapters/codex.py index cecc6848..f1a5b4e3 100644 --- a/benchmarks/agent/adapters/codex.py +++ b/benchmarks/agent/adapters/codex.py @@ -13,6 +13,8 @@ from datetime import datetime from pathlib import Path +from common import as_dict, cli_version, feed_stdin, INFRA_EXIT, TRANSIENT + def usage_row(usage: dict, timestamp: float, limit: int | None) -> dict: cached = usage.get("cached_input_tokens") or 0 @@ -40,16 +42,22 @@ def session_usage(state: Path): event = json.loads(line) except json.JSONDecodeError: continue - payload = event.get("payload") or {} - info = payload.get("info") - if payload.get("type") == "token_count" and info: - yield json.dumps(info["total_token_usage"], sort_keys=True), usage_row(info["last_token_usage"], datetime.fromisoformat(event["timestamp"]).timestamp(), info.get("model_context_window")) + if not isinstance(event, dict): + continue + payload = as_dict(event.get("payload")) + info = as_dict(payload.get("info")) + if payload.get("type") == "token_count" and info.get("total_token_usage") and info.get("last_token_usage"): + try: + stamp = datetime.fromisoformat(event.get("timestamp") or "").timestamp() + except ValueError: + stamp = time.time() + yield json.dumps(info["total_token_usage"], sort_keys=True), usage_row(info["last_token_usage"], stamp, info.get("model_context_window")) def main() -> int: env = os.environ model = env["BENCH_MODEL"] - effort = env["BENCH_THINKING"] + effort = env.get("BENCH_THINKING") run = Path(env["BENCH_RUN_DIR"]) home = run / "codex-home" state = home / ".codex" @@ -64,31 +72,37 @@ def main() -> int: "--disable", "apps", "--disable", "plugins", "--disable", "remote_plugin", "--disable", "skill_mcp_dependency_install", "--sandbox", "danger-full-access", "-c", 'approval_policy="never"', - "-c", 'web_search="disabled"', "-c", f'model_reasoning_effort="{effort}"', "-m", model, "-"] + "-c", 'web_search="disabled"', *(["-c", f'model_reasoning_effort="{effort}"'] if effort else []), "-m", model, "-"] child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home), "CODEX_HOME": str(state)} prompt = sys.stdin.read() - final, errors, summary, seen, servers, item_types = "", [], None, set(), set(), set() + final, errors, summary, seen, servers, item_types, scanned = "", [], None, set(), set(), set(), 0.0 with (run / "codex-stderr.log").open("w") as stderr, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript, open(env["BENCH_USAGE"], "w", buffering=1) as usage: process = subprocess.Popen(command, env=child_env, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=stderr, text=True) - process.stdin.write(prompt) - process.stdin.close() + feed_stdin(process, prompt) for line in process.stdout: - event = json.loads(line) + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + if not isinstance(event, dict): + continue transcript.write(json.dumps({"t": time.time(), **event}) + "\n") - item = event.get("item") or {} + item, etype = as_dict(event.get("item")), event.get("type") item_types.add(item.get("type")) if item.get("type") == "mcp_tool_call": servers.add(item.get("server", "unknown")) if item.get("type") == "agent_message": final = item.get("text", final) - if event["type"] in ("error", "turn.failed"): + if etype in ("error", "turn.failed"): errors.append(str(event.get("message") or event.get("error"))) - if event["type"] == "turn.completed": + if etype == "turn.completed": summary = event.get("usage") - for key, row in session_usage(state): - if key not in seen: - seen.add(key) - usage.write(json.dumps(row) + "\n") + if time.time() - scanned > 1: # re-globbing every session file per line is quadratic on long sessions + scanned = time.time() + for key, row in session_usage(state): + if key not in seen: + seen.add(key) + usage.write(json.dumps(row) + "\n") code = process.wait() for key, row in session_usage(state): if key not in seen: @@ -101,24 +115,29 @@ def main() -> int: loaded, builtin, observed = [], [], {} for path in state.glob("sessions/**/*.jsonl"): for line in path.read_text().splitlines(): - event = json.loads(line) + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + if not isinstance(event, dict): + continue payload = event.get("payload") or {} - if event["type"] == "turn_context": + if event.get("type") == "turn_context": observed = {k: payload.get(k) for k in ("model", "effort")} - if event["type"] == "response_item" and payload.get("role") == "developer": + if event.get("type") == "response_item" and payload.get("role") == "developer": for part in payload.get("content", []): - names, builtins = loaded_skills(part.get("text", "")) + names, builtins = loaded_skills(as_dict(part).get("text", "")) loaded += names builtin += builtins Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": sorted(set(loaded) | {f"unexpected-mcp:{server}" for server in servers}), "builtinSkills": sorted(set(builtin))})) - Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "codex", "model": observed.get("model") or model, "effort": observed.get("effort") or effort, "sandbox": "danger-full-access inside the harness file sandbox", "toolsObserved": sorted(t for t in item_types if t)}, indent=2)) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "codex", "version": cli_version(binary), "model": observed.get("model") or model, "effort": observed.get("effort") or effort, "sandbox": "danger-full-access inside the harness file sandbox", "toolsObserved": sorted(t for t in item_types if t)}, indent=2)) (run / "adapter.json").write_text(json.dumps({"agent": "codex", "requestedModel": model, "requestedEffort": effort, "observed": observed, "usageSource": "session-token-count" if observed else "turn-summary"}, indent=2)) sys.stdout.write(final) stderr_text = (run / "codex-stderr.log").read_text() sys.stderr.write(stderr_text) failure = " ".join(errors) + stderr_text if code or errors: - return 75 if any(s in failure.lower() for s in ("rate limit", "usage limit", "429", "quota", "temporarily", "503")) else (code or 1) + return INFRA_EXIT if any(s in failure.lower() for s in TRANSIENT) else (code or 1) return 0 diff --git a/benchmarks/agent/adapters/common.py b/benchmarks/agent/adapters/common.py new file mode 100644 index 00000000..8b9a72e8 --- /dev/null +++ b/benchmarks/agent/adapters/common.py @@ -0,0 +1,62 @@ +"""Helpers shared by the adapters.""" + +from __future__ import annotations + +import subprocess +import threading +from collections.abc import Callable + +INFRA_EXIT = 75 # EX_TEMPFAIL: a provider or infrastructure failure, not an agent failure; the harness retries the trial later +# Error text that means the provider, not the agent, failed. Shared so the classification cannot drift between adapters. +TRANSIENT = ("rate limit", "overloaded", "429", "502", "503", "529", "bad gateway", "timed out", "timeout", "temporarily", + "usage limit", "quota", "credits", "fetch failed", "websocket error", "connection error", "econnreset", "unavailable") + + +def as_dict(value) -> dict: + """value when it is a dict, else {} — `or {}` alone does not guard a truthy non-dict from a malformed stream line.""" + return value if isinstance(value, dict) else {} + + +def feed_stdin(proc: subprocess.Popen, text: str) -> None: + """Write the prompt to the child's stdin on a thread — a large prompt otherwise blocks the loop that must drain the pipes.""" + def feed() -> None: + try: + proc.stdin.write(text) + proc.stdin.close() + except (BrokenPipeError, ValueError): # the child exited before taking the whole prompt + pass + threading.Thread(target=feed, daemon=True).start() + + +def drain(stream) -> Callable[[], str]: + """Read a child pipe to EOF on a thread and return a callable giving its text — a pipe read only after stdout closes + leaves the child blocked on a full buffer. Call the result after proc.wait().""" + lines: list[str] = [] + thread = threading.Thread(target=lambda: lines.extend(stream), daemon=True) + thread.start() + + def text() -> str: + thread.join() + return "".join(lines) + + return text + + +def cli_version(binary: str, env: dict | None = None) -> str | None: + """First line of ` --version`, or None when the CLI does not answer.""" + try: + done = subprocess.run([binary, "--version"], capture_output=True, text=True, timeout=30, env=env) + except (OSError, subprocess.TimeoutExpired): + return None + return (done.stdout or done.stderr).strip().splitlines()[0] if (done.stdout or done.stderr).strip() else None + + +EFFORTS = ("low", "medium", "high", "xhigh", "max") + + +def split_effort(model_id: str) -> tuple[str, str | None]: + """Some CLIs name the effort in the model id (`swe-2-max`, `gemini-3.8-flash-high`) and never report it separately.""" + for effort in EFFORTS: + if model_id.endswith(f"-{effort}"): + return model_id[: -len(effort) - 1], effort + return model_id, None diff --git a/benchmarks/agent/adapters/devin.py b/benchmarks/agent/adapters/devin.py index 383a7ba1..964de476 100755 --- a/benchmarks/agent/adapters/devin.py +++ b/benchmarks/agent/adapters/devin.py @@ -22,8 +22,7 @@ from datetime import datetime from pathlib import Path -INFRA_EXIT = 75 -TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily", "usage limit", "quota", "insufficient credits", "fetch failed", "websocket error", "connection error") +from common import cli_version, split_effort, INFRA_EXIT, TRANSIENT WEB_TOOLS = ["WebFetch", "WebSearch", "webfetch", "web_search"] MCP_TOOLS = ["mcp_call_tool", "mcp_list_tools", "mcp_list_servers", "mcp_read_resource"] # org-managed plugins install MCP servers NO_TOOL_CONFIG = {"claude": False, "cursor": False, "windsurf": False, "codex": False} @@ -33,6 +32,8 @@ def isolated_config(user_config: dict, model: str, web: bool) -> dict: config = copy.deepcopy(user_config) config["read_config_from"] = NO_TOOL_CONFIG config["permissions"] = {"deny": MCP_TOOLS + ([] if web else WEB_TOOLS)} + # A denied call ends a headless session, so the tools are also hidden from the agent: it never reaches for what the condition withholds. + config["disabled_tools"] = MCP_TOOLS + ([] if web else [t for t in WEB_TOOLS if t.islower()]) config.setdefault("agent", {})["model"] = model return config @@ -66,6 +67,14 @@ def parse_export(export: dict) -> tuple[list[dict], list[str], list[str]]: return rows, loaded, plugins +def transient(stdout: str, stderr: str) -> bool: + """A provider or infrastructure failure. The agent's own text on stdout can mimic provider wording, so only stderr and the + CLI's anchored model-catalog error count. An empty catalog means the catalog could not be fetched, not that the model is wrong.""" + catalog = r"unknown model[^\n]*\n\s*available:\s*$" + return (any(s in stderr.lower() for s in TRANSIENT) + or re.search(catalog, stdout.lower().strip()) is not None or re.search(catalog, stderr.lower().strip()) is not None) + + def main() -> int: env = os.environ model = env.get("BENCH_MODEL") or sys.exit("BENCH_MODEL is required: set it in --agent-cmd, for example BENCH_MODEL=swe-2-max python3 adapters/devin.py") @@ -90,12 +99,12 @@ def main() -> int: Path(env["BENCH_USAGE"]).write_text("".join(json.dumps(r) + "\n" for r in rows)) Path(env["BENCH_TRANSCRIPT"]).write_text("".join(json.dumps(s) + "\n" for s in (json.loads(export.read_text())["steps"] if export.is_file() else []))) exported = json.loads(export.read_text()) if export.is_file() else {} - Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "devin", "model": model, "tools": [t["function"]["name"] for t in exported.get("agent", {}).get("tool_definitions", [])], "denied": MCP_TOOLS + ([] if env["BENCH_KNOWLEDGE"] == "web" else WEB_TOOLS)}, indent=2)) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "devin", "version": cli_version(devin), "model": split_effort(model)[0], "effort": split_effort(model)[1], "modelId": model, "tools": [t["function"]["name"] for t in exported.get("agent", {}).get("tool_definitions", [])], "denied": MCP_TOOLS + ([] if env["BENCH_KNOWLEDGE"] == "web" else WEB_TOOLS)}, indent=2)) Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded, "ignoredPluginSkills": plugins})) sys.stdout.write(proc.stdout) sys.stderr.write(proc.stderr) if proc.returncode != 0: - return INFRA_EXIT if any(s in (proc.stdout + proc.stderr).lower() for s in TRANSIENT) else proc.returncode + return INFRA_EXIT if transient(proc.stdout, proc.stderr) else proc.returncode return 0 diff --git a/benchmarks/agent/adapters/direct.py b/benchmarks/agent/adapters/direct.py new file mode 100644 index 00000000..05e5c244 --- /dev/null +++ b/benchmarks/agent/adapters/direct.py @@ -0,0 +1,197 @@ +#!/usr/bin/env python3 +"""Built-in agent loop: run one benchmark trial against a model API with no agent harness around it. + +Honors the BENCH_* contract (docs/agent-benchmark.md). BENCH_MODEL is `anthropic/` (ANTHROPIC_API_KEY, optional +ANTHROPIC_BASE_URL) or `openai/` (OPENAI_API_KEY, optional OPENAI_BASE_URL, so any OpenAI-compatible endpoint works). +The model gets one `bash` tool that runs in the workspace on the shimmed PATH, plus `fetch` only when knowledge is `web`. +The system prompt lists the installed skills by name and description, and the model reads their files itself. +The loop, its limits, and the tool set are fixed here so every model faces the same protocol; the recorded harness commit identifies them. +It does not sandbox the network: pair it with the harness --canary-cmd. Exit 75 marks a provider or infrastructure failure. +""" + +from __future__ import annotations + +import json +import os +import re +import shutil +import subprocess +import sys +import time +import urllib.error +import urllib.request +from pathlib import Path + +from common import INFRA_EXIT + +MAX_TURNS = 60 +COMMAND_SECONDS = 120 +OUTPUT_CHARS = 20_000 +MAX_OUTPUT_TOKENS = 16_000 +RETRIES = 3 +SYSTEM = "You are a coding agent working in the current directory. Use the bash tool to inspect and edit files. Finish with a short summary of what you did." +BASH = {"name": "bash", "description": "Run a shell command in the workspace and return its output.", "schema": {"type": "object", "properties": {"command": {"type": "string"}}, "required": ["command"]}} +FETCH = {"name": "fetch", "description": "HTTP GET a URL and return the start of the body.", "schema": {"type": "object", "properties": {"url": {"type": "string"}}, "required": ["url"]}} + + +class ProviderError(Exception): + def __init__(self, message: str, transient: bool): + super().__init__(message) + self.transient = transient + + +def post(url: str, headers: dict, body: dict) -> dict: + request = urllib.request.Request(url, json.dumps(body).encode(), {"content-type": "application/json", **headers}) + for attempt in range(RETRIES + 1): + try: + with urllib.request.urlopen(request, timeout=600) as response: + return json.load(response) + except urllib.error.HTTPError as error: + text = error.read().decode(errors="replace")[:500] + if error.code in (408, 429) or error.code >= 500: + if attempt < RETRIES: + time.sleep(2 ** attempt * 5) + continue + raise ProviderError(f"HTTP {error.code}: {text}", True) + raise ProviderError(f"HTTP {error.code}: {text}", False) + except (urllib.error.URLError, TimeoutError, ConnectionError) as error: + if attempt < RETRIES: + time.sleep(2 ** attempt * 5) + continue + raise ProviderError(str(error), True) + raise AssertionError + + +def usage_row(inp: int, out: int, cache_read: int, cache_write: int, reasoning: int | None) -> dict: + return {"t": time.time(), "input": inp, "output": out, "cache_read": cache_read, "cache_write": cache_write, "reasoning": reasoning, "context": inp + cache_read + cache_write, "context_limit": None} + + +class Anthropic: + def __init__(self, model: str, tools: list[dict], system: str): + self.model, self.system = model, system + self.url = os.environ.get("ANTHROPIC_BASE_URL", "https://api.anthropic.com").rstrip("/") + "/v1/messages" + self.headers = {"x-api-key": os.environ["ANTHROPIC_API_KEY"], "anthropic-version": "2023-06-01"} + self.tools = [{"name": t["name"], "description": t["description"], "input_schema": t["schema"]} for t in tools] + self.messages: list[dict] = [] + + def user(self, text: str) -> None: + self.messages.append({"role": "user", "content": text}) + + def results(self, results: list[tuple[str, str]]) -> None: + self.messages.append({"role": "user", "content": [{"type": "tool_result", "tool_use_id": i, "content": out} for i, out in results]}) + + def step(self) -> tuple[str, list[tuple[str, str, dict]], dict]: + data = post(self.url, self.headers, {"model": self.model, "max_tokens": MAX_OUTPUT_TOKENS, "system": self.system, "tools": self.tools, "messages": self.messages}) + try: + content = data["content"] + text = "".join(b.get("text", "") for b in content if b["type"] == "text") + calls = [(b["id"], b["name"], b["input"]) for b in content if b["type"] == "tool_use"] + except (KeyError, IndexError, TypeError, AttributeError) as error: + raise ProviderError(f"malformed response: {error}", False) from error + self.messages.append({"role": "assistant", "content": content}) + u = data.get("usage") or {} + return text, calls, usage_row(u.get("input_tokens") or 0, u.get("output_tokens") or 0, u.get("cache_read_input_tokens") or 0, u.get("cache_creation_input_tokens") or 0, None) + + +class OpenAI: + def __init__(self, model: str, tools: list[dict], system: str, effort: str | None): + self.model, self.effort = model, effort + self.url = os.environ.get("OPENAI_BASE_URL", "https://api.openai.com/v1").rstrip("/") + "/chat/completions" + self.headers = {"authorization": f"Bearer {os.environ['OPENAI_API_KEY']}"} + self.tools = [{"type": "function", "function": {"name": t["name"], "description": t["description"], "parameters": t["schema"]}} for t in tools] + self.messages: list[dict] = [{"role": "system", "content": system}] + + def user(self, text: str) -> None: + self.messages.append({"role": "user", "content": text}) + + def results(self, results: list[tuple[str, str]]) -> None: + self.messages += [{"role": "tool", "tool_call_id": i, "content": out} for i, out in results] + + def step(self) -> tuple[str, list[tuple[str, str, dict]], dict]: + body = {"model": self.model, "tools": self.tools, "messages": self.messages, **({"reasoning_effort": self.effort} if self.effort else {})} + data = post(self.url, self.headers, body) + try: + message = data["choices"][0]["message"] + calls = [(c["id"], c["function"]["name"], json.loads(c["function"]["arguments"] or "{}")) for c in message.get("tool_calls") or []] + except (KeyError, IndexError, TypeError, AttributeError, json.JSONDecodeError) as error: + raise ProviderError(f"malformed response: {error}", False) from error + self.messages.append(message) + u = data.get("usage") or {} + cached = (u.get("prompt_tokens_details") or {}).get("cached_tokens") or 0 + reasoning = (u.get("completion_tokens_details") or {}).get("reasoning_tokens") or 0 + return message.get("content") or "", calls, usage_row((u.get("prompt_tokens") or 0) - cached, (u.get("completion_tokens") or 0) - reasoning, cached, 0, reasoning) + + +def run_tool(name: str, args: dict, env: dict) -> str: + try: + if name == "bash": + done = subprocess.run(["bash", "--noprofile", "--norc", "-c", args["command"]], capture_output=True, text=True, timeout=COMMAND_SECONDS, env=env, errors="replace") + out = done.stdout + done.stderr + (f"\n[exit {done.returncode}]" if done.returncode else "") + elif name == "fetch": + with urllib.request.urlopen(args["url"], timeout=60) as response: + out = response.read(OUTPUT_CHARS * 2).decode(errors="replace") + else: + return f"unknown tool {name}" + except subprocess.TimeoutExpired: + return f"[timed out after {COMMAND_SECONDS}s]" + except Exception as error: + return f"[{type(error).__name__}: {error}]" + return out if len(out) <= OUTPUT_CHARS else out[:OUTPUT_CHARS] + f"\n[truncated {len(out) - OUTPUT_CHARS} characters]" + + +def skill_listing(skills: list[Path]) -> tuple[str, list[str]]: + lines, names = [], [] + for skill in skills: + target = Path(".agents/skills") / skill.name + shutil.copytree(skill, target) + text = (target / "SKILL.md").read_text() + name = re.search(r"^name:\s*(\S+)", text, re.M) # same first-word rule as the harness's skill_name check + description = re.search(r"^description:\s*(.+)$", text, re.M) + names.append(name.group(1).strip() if name else skill.name) + lines.append(f"- {names[-1]}: {description.group(1).strip() if description else ''} (read {target}/SKILL.md when relevant)") + return ("\n\nSkills available:\n" + "\n".join(lines)) if lines else "", names + + +def main() -> int: + env = os.environ + provider, _, model = env["BENCH_MODEL"].partition("/") + if provider not in ("anthropic", "openai") or not model: + print("BENCH_MODEL must be anthropic/ or openai/", file=sys.stderr) + return 2 + web = env["BENCH_KNOWLEDGE"] == "web" + tools = [BASH] + ([FETCH] if web else []) + listing, loaded = skill_listing([Path(p) for p in env.get("BENCH_SKILL_DIRS", "").split(os.pathsep) if p]) + effort = env.get("BENCH_THINKING") if provider == "openai" else None + system = SYSTEM + listing + chat = Anthropic(model, tools, system) if provider == "anthropic" else OpenAI(model, tools, system, effort) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({ + "agent": "direct", "model": env["BENCH_MODEL"], "effort": effort, "tools": [t["name"] for t in tools], + "protocol": {"maxTurns": MAX_TURNS, "commandSeconds": COMMAND_SECONDS, "outputChars": OUTPUT_CHARS, "maxOutputTokens": MAX_OUTPUT_TOKENS}}, indent=2)) + Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded})) + child_env = {k: v for k, v in env.items() if k not in ("BENCH_HOST_PATH", "ANTHROPIC_API_KEY", "OPENAI_API_KEY")} + chat.user(sys.stdin.read()) + final, code = "", 0 + with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: + transcript.write(json.dumps({"t": time.time(), "type": "system", "text": system}) + "\n") + for _ in range(MAX_TURNS): + try: + text, calls, row = chat.step() + except ProviderError as error: + print(f"provider error: {error}", file=sys.stderr) + code = INFRA_EXIT if error.transient else 1 + break + usage.write(json.dumps(row) + "\n") + transcript.write(json.dumps({"t": time.time(), "type": "assistant", "text": text, "calls": [{"name": n, "input": a} for _, n, a in calls]}) + "\n") + final = text + if not calls: + break + outputs = [(i, run_tool(n, a, child_env)) for i, n, a in calls] + for (_, n, a), (_, out) in zip(calls, outputs): + transcript.write(json.dumps({"t": time.time(), "type": "tool_result", "name": n, "output": out}) + "\n") + chat.results(outputs) + sys.stdout.write(final) + return code + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/agent/adapters/grok.py b/benchmarks/agent/adapters/grok.py new file mode 100644 index 00000000..d6dfef22 --- /dev/null +++ b/benchmarks/agent/adapters/grok.py @@ -0,0 +1,93 @@ +#!/usr/bin/env python3 +"""Adapter for the Grok CLI (`grok -p`): run one benchmark trial and report per-turn usage. + +BENCH_MODEL is a model id from `grok models` and must be set in --agent-cmd. The run uses an isolated GROK_HOME that holds only the +login (auth.json) and the benchmark skills, so no global config, rules, skills, or plugins load. The prompt is sent verbatim. +Subagents are disabled so each assistant message is one model call, and web tools are disabled unless knowledge is `web`. +The tools and skills the agent actually had come from the stream's init line. BENCH_THINKING is the reasoning effort. +The network is not sandboxed: pair this with the harness --canary-cmd. Exit 75 marks a provider or infrastructure failure. +""" + +from __future__ import annotations + +import json +import os +import shutil +import subprocess +import sys +import time +from pathlib import Path + +from common import as_dict, cli_version, drain, INFRA_EXIT, TRANSIENT + + +def usage_row(usage: dict, limit: int | None, now: float) -> dict: + read, write = usage.get("cache_read_input_tokens") or 0, usage.get("cache_creation_input_tokens") or 0 + return {"t": now, "input": usage.get("input_tokens"), "output": usage.get("output_tokens"), "cache_read": read, "cache_write": write, + "reasoning": None, "context": (usage.get("input_tokens") or 0) + read + write, "context_limit": limit} + + +def main() -> int: + env = os.environ + model = env.get("BENCH_MODEL") or sys.exit("BENCH_MODEL is required: set it in --agent-cmd, for example BENCH_MODEL=grok-4.7 python3 adapters/grok.py") + run_dir, real_home = Path(env["BENCH_RUN_DIR"]), Path(env["HOME"]) + grok_home = run_dir / "grok-home" + (grok_home / "skills").mkdir(parents=True) + shutil.copy(real_home / ".grok/auth.json", grok_home / "auth.json") + for skill in (Path(p) for p in env.get("BENCH_SKILL_DIRS", "").split(os.pathsep) if p): + shutil.copytree(skill, grok_home / "skills" / skill.name) + web = env["BENCH_KNOWLEDGE"] == "web" + grok = shutil.which("grok", path=env.get("BENCH_HOST_PATH")) or "grok" + child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "GROK_HOME": str(grok_home), "HOME": str(run_dir / "grok-user"), "GROK_TELEMETRY_ENABLED": "0", "GROK_DISABLE_AUTOUPDATER": "1"} + (run_dir / "grok-user").mkdir() + prompt = run_dir / "grok-prompt.txt" + prompt.write_text(sys.stdin.read()) + cmd = [grok, "--verbatim", "--output-format", "streaming-messages-json", "--permission-mode", "bypassPermissions", "--no-subagents", "-m", model, "--prompt-file", str(prompt)] + if not web: + cmd.append("--disable-web-search") + if env.get("BENCH_THINKING"): + cmd += ["--reasoning-effort", env["BENCH_THINKING"]] + version = cli_version(grok, child_env) + proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=child_env) + stderr_text = drain(proc.stderr) + init: dict = {} + pending: list[dict] = [] + final, error, limit = "", "", None + with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: # line-buffered: a killed run keeps its usage + for line in proc.stdout: + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + if not isinstance(event, dict): + continue + now = time.time() + transcript.write(json.dumps({"t": now, **event}) + "\n") + message = as_dict(event.get("message")) + if event.get("type") == "system" and event.get("subtype") == "init": + init = event + elif event.get("type") == "assistant": + pending.append(message.get("usage") or {}) + text = "".join(as_dict(b).get("text", "") for b in message.get("content", []) if isinstance(b, dict) and b.get("type") == "text") + final = text or final + elif event.get("type") == "result": + final = event.get("result") or final + if event.get("is_error"): + error = json.dumps(event.get("errors") or "error") # only the structured error classifies; the result text is the agent's own + limit = next((m.get("contextWindow") for m in as_dict(event.get("modelUsage")).values()), None) + for u in pending: # the context window is only known from the final result line + usage.write(json.dumps(usage_row(u, limit, time.time())) + "\n") + stderr = stderr_text() + code = proc.wait() + Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": init.get("skills") or []})) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "grok", "version": version, "model": init.get("model") or model, "effort": env.get("BENCH_THINKING"), "tools": init.get("tools"), + "mcpServers": init.get("mcp_servers"), "disabled": ["subagents"] + ([] if web else ["web"])}, indent=2)) + sys.stdout.write(final) + sys.stderr.write(stderr) + if error or code != 0: + return INFRA_EXIT if any(s in (error + stderr).lower() for s in TRANSIENT) else (code or 1) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/agent/adapters/opencode.py b/benchmarks/agent/adapters/opencode.py new file mode 100644 index 00000000..33cf2ad0 --- /dev/null +++ b/benchmarks/agent/adapters/opencode.py @@ -0,0 +1,97 @@ +#!/usr/bin/env python3 +"""Adapter for opencode (`opencode run`): run one benchmark trial and report per-turn usage. + +BENCH_MODEL is `provider/model` as `opencode models` lists it, and must be set in --agent-cmd. The run uses an isolated HOME with +only the opencode credentials copied in, so no global instructions, agents, skills, or plugins load (`--pure`). Skills are installed in the +workspace under `.opencode/skills`; opencode also reads the real home's `~/.claude` and `~/.agents` skills and Claude Code instructions regardless of HOME, +so those scans are disabled by environment variable. Web tools are denied unless knowledge is `web`. BENCH_THINKING is passed as the model variant. +The network is not sandboxed: pair this with the harness --canary-cmd. Exit 75 marks a provider or infrastructure failure. +""" + +from __future__ import annotations + +import json +import os +import shutil +import subprocess +import sys +import time +from pathlib import Path + +from common import as_dict, cli_version, drain, feed_stdin, INFRA_EXIT, TRANSIENT +WEB_PERMISSIONS = {"webfetch": "deny", "websearch": "deny"} + + +def usage_row(tokens: dict, now: float) -> dict: + cache = tokens.get("cache") or {} + read, write = cache.get("read") or 0, cache.get("write") or 0 + return {"t": now, "input": tokens.get("input"), "output": tokens.get("output"), "cache_read": read, "cache_write": write, + "reasoning": tokens.get("reasoning"), "context": (tokens.get("input") or 0) + read + write, "context_limit": None} + + +def available_skills(raw: str) -> list[str]: + """Skill names opencode reports, without its built-in ones.""" + try: + return [s["name"] for s in json.loads(raw) if s.get("location") != ""] + except (json.JSONDecodeError, TypeError): + return [] + + +def main() -> int: + env = os.environ + model = env.get("BENCH_MODEL") or sys.exit("BENCH_MODEL is required: set it in --agent-cmd, for example BENCH_MODEL=openai/gpt-6-luna python3 adapters/opencode.py") + run_dir, workspace, real_home = Path(env["BENCH_RUN_DIR"]), Path.cwd(), Path(env["HOME"]) + home = run_dir / "opencode-home" + (home / ".local/share/opencode").mkdir(parents=True) + (home / ".config/opencode").mkdir(parents=True) + shutil.copy(real_home / ".local/share/opencode/auth.json", home / ".local/share/opencode/auth.json") + for skill in (Path(p) for p in env.get("BENCH_SKILL_DIRS", "").split(os.pathsep) if p): + shutil.copytree(skill, workspace / ".opencode/skills" / skill.name) + web = env["BENCH_KNOWLEDGE"] == "web" + opencode = shutil.which("opencode", path=env.get("BENCH_HOST_PATH")) or "opencode" + child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home), "XDG_CONFIG_HOME": str(home / ".config"), "XDG_DATA_HOME": str(home / ".local/share"), + "OPENCODE_DISABLE_CLAUDE_CODE": "1", "OPENCODE_DISABLE_EXTERNAL_SKILLS": "1", + "OPENCODE_CONFIG_CONTENT": json.dumps({"$schema": "https://opencode.ai/config.json", "autoupdate": False, "share": "disabled", **({} if web else {"permission": WEB_PERMISSIONS})})} + version = cli_version(opencode, child_env) + listing = run_dir / "opencode-skills.json" # a file, not a pipe: `debug skill` truncates large output when stdout is a pipe + with open(listing, "w") as out: + subprocess.run([opencode, "debug", "skill"], stdout=out, env=child_env) + loaded = available_skills(listing.read_text()) + cmd = [opencode, "run", "--pure", "--format", "json", "--auto", "-m", model] + if env.get("BENCH_THINKING"): + cmd += ["--variant", env["BENCH_THINKING"]] + proc = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=child_env) + stderr_text = drain(proc.stderr) + feed_stdin(proc, sys.stdin.read()) + final, error, tools = "", "", set() + with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: # line-buffered: a killed run keeps its usage + for line in proc.stdout: + try: + event = json.loads(line) + except json.JSONDecodeError: + continue + if not isinstance(event, dict): + continue + now, part, etype = time.time(), as_dict(event.get("part")), event.get("type") + transcript.write(json.dumps({"t": now, **event}) + "\n") + if etype == "step_finish" and part.get("tokens"): + usage.write(json.dumps(usage_row(part["tokens"], now)) + "\n") + elif etype == "text": + final = part.get("text") or final + elif etype == "tool_use" and part.get("tool"): + tools.add(part["tool"]) + elif etype == "error": + error = json.dumps(event.get("error")) + stderr = stderr_text() + code = proc.wait() + Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded})) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "opencode", "version": version, "model": model, "effort": env.get("BENCH_THINKING"), "tools": sorted(tools), "toolsNote": "tools the agent used; the CLI does not list its tools", "denied": [] if web else sorted(WEB_PERMISSIONS)}, indent=2)) + sys.stdout.write(final) + sys.stderr.write(stderr) + if error or code != 0: + return INFRA_EXIT if any(s in (error + stderr).lower() for s in TRANSIENT) else (code or 1) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/benchmarks/agent/adapters/pi.py b/benchmarks/agent/adapters/pi.py index 134874af..0e0d2643 100755 --- a/benchmarks/agent/adapters/pi.py +++ b/benchmarks/agent/adapters/pi.py @@ -21,8 +21,7 @@ import time from pathlib import Path -INFRA_EXIT = 75 -TRANSIENT = ("rate limit", "overloaded", "429", "503", "529", "timed out", "timeout", "temporarily", "usage limit", "quota", "insufficient credits", "fetch failed", "websocket error", "connection error") +from common import as_dict, cli_version, drain, feed_stdin, INFRA_EXIT, TRANSIENT SCALE = {"K": 1_000, "M": 1_000_000} @@ -31,7 +30,10 @@ def context_limit(pi: str, model: str, env: dict, extensions: list[str]) -> int command = [pi, "--no-extensions"] for extension in filter(None, extensions): command += ["-e", extension] - listing = subprocess.run([*command, "--list-models", model.split("/")[-1]], env=env, capture_output=True, text=True).stdout + try: + listing = subprocess.run([*command, "--list-models", model.split("/")[-1]], env=env, capture_output=True, text=True, timeout=30).stdout + except (OSError, subprocess.TimeoutExpired): + return None for line in listing.splitlines(): cols = line.split() if len(cols) > 2 and cols[0] == model.split("/")[0] and cols[1] == model.split("/")[-1]: @@ -80,10 +82,10 @@ def main() -> int: cmd += ["--thinking", env["BENCH_THINKING"]] child_env = {**{k: v for k, v in env.items() if k != "BENCH_HOST_PATH"}, "HOME": str(home), "PI_CODING_AGENT_DIR": str(state)} limit = context_limit(pi, model, child_env, extensions) - Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "pi", "model": model, "effort": env.get("BENCH_THINKING"), "tools": ["read", "bash", "edit", "write"], "extensions": [e for e in extensions if e]}, indent=2)) + Path(env["BENCH_AGENT_INFO"]).write_text(json.dumps({"agent": "pi", "version": cli_version(pi), "model": model, "effort": env.get("BENCH_THINKING"), "tools": ["read", "bash", "edit", "write"], "extensions": [e for e in extensions if e]}, indent=2)) proc = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=child_env) - proc.stdin.write(prompt) - proc.stdin.close() + stderr_text = drain(proc.stderr) + feed_stdin(proc, prompt) loaded: list[str] = [] final, error = "", "" with open(env["BENCH_USAGE"], "w", buffering=1) as usage, open(env["BENCH_TRANSCRIPT"], "w", buffering=1) as transcript: # line-buffered: a killed run keeps its usage @@ -92,20 +94,22 @@ def main() -> int: event = json.loads(line) except json.JSONDecodeError: continue + if not isinstance(event, dict): + continue now = time.time() - message = event.get("message") or {} - if event["type"] != "message_update": + message, etype = as_dict(event.get("message")), event.get("type") + if etype != "message_update": transcript.write(json.dumps({"t": now, **event}) + "\n") - if event["type"] == "message_start" and message.get("role") == "system": + if etype == "message_start" and message.get("role") == "system": loaded = skill_names(message) - if event["type"] == "message_end" and message.get("role") == "assistant": + if etype == "message_end" and message.get("role") == "assistant": usage.write(json.dumps(usage_row(message, limit, now)) + "\n") final = message_text(message) or final if message.get("stopReason") == "error": - error = str(message.get("errorMessage") or message_text(message)) + error = str(message.get("errorMessage") or "error") # only the structured error classifies; message text is the agent's own else: error = "" - stderr = proc.stderr.read() + stderr = stderr_text() code = proc.wait() Path(env["BENCH_CONTEXT"]).write_text(json.dumps({"loaded": loaded})) sys.stdout.write(final) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 4d608ce1..dac30b6d 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -4,13 +4,14 @@ from __future__ import annotations import argparse -import functools +import contextlib import hashlib import json import os import platform import random import re +import shlex import shutil import signal import subprocess @@ -21,7 +22,10 @@ from datetime import datetime, timezone from pathlib import Path +import stat + import bench_grade +import bench_leaderboard import bench_report import bench_score import bench_trace @@ -168,7 +172,7 @@ def build_env(cell: dict, args: argparse.Namespace, out: Path, workspace: Path) shim_dir = out / "bin" shim_dir.mkdir() shim = shim_dir / cell["tool"] - shim.write_text(f'#!/bin/sh\nexec "{sys.executable}" "{Path(__file__).resolve()}" shim {cell["tool"]} "$@"\n') + shim.write_text(f'#!/bin/sh\nexec "{sys.executable}" "{(HERE / "bench_trace.py").resolve()}" shim {cell["tool"]} "$@"\n') shim.chmod(0o755) real = args.wright if cell["tool"] == "wright" else str(overpy_launcher(out, os.environ["PATH"])) env.update({f"BENCH_TOOL_REAL_{cell['tool'].upper()}": real, "BENCH_TOOL_TRACE": str(out / "tool-trace.jsonl"), "BENCH_TOOL_SIDECAR": str(out / "tool-calls")}) @@ -206,6 +210,85 @@ def canaries(cell: dict, env: dict, workspace: Path, args: argparse.Namespace) - return None +BENCH_HOME = Path(os.environ.get("WRIGHT_BENCH_HOME", Path.home() / ".local/share/wright-agent-bench")) # the one place for runs (`runs/`), wiki snapshots, and pinned skills +DATA_ROOTS = (BENCH_HOME, Path.home() / ".cache/wright-agent-bench") # hidden from agents; the second is where earlier versions wrote runs +HIDDEN_ROOTS = (Path("/Users"), Path("/Volumes"), Path.home()) # the host's home directories and external drives are hidden unless listed below +ADAPTER_READS = { # what each adapter reads from the real home before the agent starts: credentials, configuration, installation (relative to the home directory) + "devin": [".local/share/devin", ".config/devin"], "pi": [".pi"], "codex": [".codex/auth.json"], "agy": [".gemini/antigravity-cli"], + "opencode": [".local/share/opencode/auth.json"], "grok": [".grok/auth.json"], "claude-code": [".claude.json", ".claude/.credentials.json"], "direct": [], +} + + +CREDENTIALS = { # (real file relative to the home directory, its isolated copy relative to the run directory) per adapter; preflight requires every one + "codex": [(".codex/auth.json", "codex-home/.codex/auth.json")], + "pi": [(".pi/agent/auth.json", "pi-home/.pi/agent/auth.json")], + "devin": [(".local/share/devin/credentials.toml", "devin-home/.local/share/devin/credentials.toml"), + (".config/devin/config.json", "devin-home/.config/devin/config.json")], + "agy": [(".gemini/antigravity-cli/antigravity-oauth-token", "agy-home/.gemini/antigravity-cli/antigravity-oauth-token"), + (".gemini/antigravity-cli/installation_id", "agy-home/.gemini/antigravity-cli/installation_id")], + "opencode": [(".local/share/opencode/auth.json", "opencode-home/.local/share/opencode/auth.json")], + "grok": [(".grok/auth.json", "grok-home/auth.json")], +} +OPTIONAL_CREDENTIALS = { # synced back when present, but not every install needs them (pi's antigravity login) + "pi": [(".pi/agent/antigravity-accounts.json", "pi-home/.pi/agent/antigravity-accounts.json")], +} + + +def sync_credentials_back(pairs: list[tuple[str, str]], run_dir: Path, home: Path | None = None) -> list[str]: + """Copy a login the agent's CLI refreshed inside its isolated home back to the real one. + + Adapters hand the CLI a copy of the user's OAuth login. Refresh tokens rotate, so a refresh in the copy that is then discarded + leaves the user's own login holding a dead token. Only a changed, newer copy is written back, atomically.""" + home = home or Path.home() + updated = [] + for real_rel, isolated_rel in pairs: + real, isolated = home / real_rel, run_dir / isolated_rel + if not isolated.is_file() or not real.is_file() or isolated.read_bytes() == real.read_bytes() or isolated.stat().st_mtime <= real.stat().st_mtime: + continue + temporary = real.with_name(f".{real.name}.bench-sync") + temporary.unlink(missing_ok=True) # a leftover temp keeps its mode; start fresh at 0o600 so a credential never sits at the default umask + temporary.touch(mode=0o600) + temporary.write_bytes(isolated.read_bytes()) + temporary.chmod(real.stat().st_mode & 0o777) + os.replace(temporary, real) + updated.append(real_rel) + return updated + + +def read_policy(args: argparse.Namespace, env: dict) -> tuple[list[Path], list[Path]]: + """What the agent may not read, and the exceptions it needs. + + An allow-list: the host's home directories and drives are hidden, so the agent sees the machine as a clean one holding only its run + directory, the condition's skills, the tool and runtime binaries, and what its adapter needs. Agents otherwise read the whole machine: + the answer keys, other runs, the grader, installed copies of the OverPy source, and checkouts of the repositories under test, so the + benchmark would measure what the agent could find instead of what it was given.""" + run_dir = Path(env["BENCH_RUN_DIR"]).resolve() + selected = [name for name in env["BENCH_SKILLS"].split(",") if name] + hidden = [*HIDDEN_ROOTS, args.out, getattr(args, "out_root", args.out), *DATA_ROOTS, HERE, *(Path(p).expanduser() for p in getattr(args, "deny_read", []) or [])] + if getattr(args, "wiki_dir", None): + hidden.append(Path(args.wiki_dir)) + hidden += [Path(d) for name, d in (args.skill_dirs or {}).items() if name not in selected] + runtime = [Path(args.wright), Path(sys.executable)] + for binary in (shutil.which("node"), shutil.which(ADAPTER_BINARY.get(getattr(args, "adapter", ""), ""))): + if binary: + runtime.append(Path(binary)) + allowed = [run_dir, HERE / "adapters", HERE / "bench_trace.py", Path(sys.prefix), Path(sys.base_prefix), + *(Path(args.skill_dirs[name]) for name in selected), *(Path(p).expanduser() for p in getattr(args, "allow_read", []) or [])] + if env.get("BENCH_TOOL") == "overpy": # only the overpy launcher execs the pinned oracle; other cells must not read the grading authority + allowed.append(HERE / "oracle") + for binary in runtime: # the directory of the binary and of the file its symlink resolves to + allowed += [binary.parent, binary.resolve().parent] + unique = lambda paths: list(dict.fromkeys(p.resolve() for p in paths)) + return unique(hidden), unique(allowed) + + +def sandbox_read_rules(hidden: list[Path], allowed: list[Path]) -> str: + def rule(action: str, path: Path) -> str: + return f"({action} file-read-data ({'literal' if path.is_file() else 'subpath'} {json.dumps(str(path))}))\n" + # the allow-list overrides the hidden roots, and the answer keys are hidden again whatever else is allowed + return "".join(rule("deny", p) for p in hidden) + "".join(rule("allow", p) for p in allowed) + rule("deny", SCENARIOS.resolve()) + + def run_agent(args: argparse.Namespace, env: dict, workspace: Path, prompt: str) -> tuple[int | None, str, str]: command = args.agent_cmd if getattr(args, "file_sandbox", False): @@ -215,26 +298,71 @@ def run_agent(args: argparse.Namespace, env: dict, workspace: Path, prompt: str) temporary = run_dir / "tmp" temporary.mkdir(exist_ok=True) env = {**env, "TMPDIR": str(temporary), "PYTHONDONTWRITEBYTECODE": "1"} + rules = sandbox_read_rules(*read_policy(args, env)) profile = run_dir / "agent.sb" + if profile.exists(): + os.chflags(profile, 0) # an attempt killed mid-run leaves it locked + # the profile lists what is hidden, so the agent must not read it — and it sits inside the writable run dir, so a + # bare read deny is renamed around. the write deny keeps the name bound; UF_IMMUTABLE backs it where path rules + # cannot reach (hardlinking an immutable file fails outright), and the write deny in turn blocks the chflags + # that would clear the flag. + locked = str(profile.resolve()) profile.write_text('(version 1)\n(allow default)\n(deny file-write*)\n' f'(allow file-write* (subpath {json.dumps(str(run_dir.resolve()))}) (subpath "/dev"))\n' '(deny file-read-data (require-all (regex "/(AGENTS|CLAUDE|GEMINI)[.]md$") ' - f'(require-not (subpath {json.dumps(str(workspace.resolve()))}))))\n') + f'(require-not (subpath {json.dumps(str(workspace.resolve()))}))))\n' + rules + + f'(deny file-read-data (literal {json.dumps(locked)}))\n' + + f'(deny file-write* (literal {json.dumps(locked)}))\n') + os.chflags(profile, stat.UF_IMMUTABLE) command = ["sandbox-exec", "-f", str(profile), "/bin/sh", "-c", args.agent_cmd] + else: + profile = None proc = subprocess.Popen( command, shell=isinstance(command, str), cwd=workspace, env=env, text=True, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, start_new_session=True, ) try: - stdout, stderr = proc.communicate(input=prompt, timeout=args.timeout) - return proc.returncode, stdout, stderr - except subprocess.TimeoutExpired: try: - os.killpg(proc.pid, signal.SIGKILL) - except ProcessLookupError: - pass - stdout, stderr = proc.communicate() - return None, stdout, f"{stderr}\ntimeout" if stderr else "timeout" + stdout, stderr = proc.communicate(input=prompt, timeout=args.timeout) + return proc.returncode, stdout, stderr + except subprocess.TimeoutExpired: + try: + os.killpg(proc.pid, signal.SIGKILL) + except ProcessLookupError: + pass + try: + stdout, stderr = proc.communicate(timeout=10) + except subprocess.TimeoutExpired: # a detached grandchild still holds the pipes; the adapter's own transcript has the record + proc.stdout.close() + proc.stderr.close() + stdout, stderr = "", "" + return None, stdout, f"{stderr}\ntimeout" if stderr else "timeout" + finally: + if profile is not None: + with contextlib.suppress(OSError): + os.chflags(profile, 0) + + +def drop(path: Path) -> None: + """Best-effort removal of a path inside the agent-writable run tree. The agent can chflags or chmod its own files — + and a run killed mid-trial can leave agent.sb locked — so a plain rmtree/unlink can hit EPERM. Repairs stay on the + run-dir node itself so a planted symlink cannot redirect them onto a host file; a leaf removal retries right away, + and the next pass reaches whatever a blocked traversal skipped.""" + def unlock(func, node, _exc): + with contextlib.suppress(OSError, AttributeError, NotImplementedError): # chflags is a BSD mechanism + os.chflags(node, 0, follow_symlinks=False) + with contextlib.suppress(OSError, NotImplementedError): + os.chmod(node, 0o700, follow_symlinks=False) + if func in (os.unlink, os.rmdir): + with contextlib.suppress(OSError): + func(node) + for _ in range(8): + if not os.path.lexists(path): + return + if path.is_dir() and not path.is_symlink(): + shutil.rmtree(path, onexc=unlock) + else: + unlock(os.unlink, str(path), None) def context_report(out: Path, skill_names: list[str]) -> dict: @@ -253,16 +381,17 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - check_cell(cell, args) if not applicable(scenario, cell): raise SystemExit(f"condition {cell_label(cell)} does not apply to the {scenario['language']} scenario {scenario['id']}") - shutil.rmtree(out, ignore_errors=True) + drop(out) out.mkdir(parents=True) workspace = out / "workspace" prompt = (scenario["dir"] / "prompt.md").read_text() # the exact pinned prompt: the harness adds no text infra_retries = 0 while True: - shutil.rmtree(workspace, ignore_errors=True) - for stale in ("home", "bin", "real", "tool-trace.jsonl", "tool-calls", "usage.jsonl", "transcript.jsonl", "context.json", "agent-info.json", "snapshots"): - target = out / stale - shutil.rmtree(target, ignore_errors=True) if target.is_dir() else target.unlink(missing_ok=True) + drop(workspace) + for stale in ("home", "bin", "real", "tool-trace.jsonl", "tool-calls", "usage.jsonl", "transcript.jsonl", "context.json", "agent-info.json", "snapshots", "tmp", + "devin-export.json", # devin only writes this on success; a retry that produces none would parse the previous attempt's export + *(child.name for child in out.iterdir() if child.name.endswith(("-home", "-user")))): # adapters keep their isolated homes here; a retry starts from none + drop(out / stale) materialize(scenario, workspace) if cell["knowledge"] == "wiki": shutil.copytree(Path(args.wiki_dir), workspace / "wiki") # a real copy: tools that skip symlinks (rg) must see it @@ -271,16 +400,18 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - if reason: result = base_result(scenario, cell, args, out, 0.0, None) result.update(invalid=reason, status="invalid") - (out / "result.json").write_text(json.dumps(result, indent=2) + "\n") + write_json(out / "result.json", result) return result snapshots = bench_trace.Snapshots(workspace, scenario.get("watch", [scenario["entry"]]), out / "snapshots") snapshots.start() start = time.monotonic() agent_exit, stdout, stderr = run_agent(args, env, workspace, prompt) + sync_credentials_back(getattr(args, "credentials", []), out) seconds = round(time.monotonic() - start, 1) snaps = snapshots.finish() if agent_exit == INFRA_EXIT and infra_retries < args.infra_retries: infra_retries += 1 + time.sleep(getattr(args, "infra_backoff", 0) * infra_retries) # an outage lasts longer than an immediate retry continue break (out / "agent.log").write_text(f"exit={agent_exit}\n--- stdout ---\n{stdout}\n--- stderr ---\n{stderr}\n") @@ -288,6 +419,11 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - result["infraRetries"] = infra_retries result["networkEnforcement"] = "canary-checked" if cell["network"] == "off" and args.canary_cmd else "declared-only" result["fileWriteEnforcement"] = "trial-directory-only" if getattr(args, "file_sandbox", False) else "unrestricted" + if getattr(args, "file_sandbox", False): + hidden, allowed = read_policy(args, env) + result["fileReadEnforcement"] = {"mode": "allow-list", "hidden": [str(p) for p in hidden], "allowed": [str(p) for p in allowed]} + else: + result["fileReadEnforcement"] = "unrestricted" context = context_report(out, [skill_name(Path(args.skill_dirs[name])) for name in cell["skills"]]) result["context"] = context if context.get("unexpected"): @@ -307,7 +443,7 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - first_valid = next((s["t"] for s in result["snapshots"]["series"] if s["valid"]), None) result["usage"] = bench_trace.usage_summary(out / "usage.jsonl", first_valid) result["status"] = run_status(result) - (out / "result.json").write_text(json.dumps(result, indent=2) + "\n") + write_json(out / "result.json", result) return result @@ -333,21 +469,22 @@ def snapshot_validity(scenario: dict, snaps: list[dict], wright: str, out: Path) return {"count": len(series), "firstValidIndex": next((s["i"] for s in series if s["valid"]), None), "regressions": regressions, "series": series} -@functools.lru_cache(maxsize=None) def harness_commit() -> str: proc = subprocess.run(["git", "-C", str(HERE), "rev-parse", "HEAD"], capture_output=True, text=True) dirty = subprocess.run(["git", "-C", str(HERE), "status", "--porcelain", "--", "."], capture_output=True, text=True).stdout.strip() return proc.stdout.strip() + ("+dirty" if dirty else "") -@functools.lru_cache(maxsize=None) -def file_sha256(path: str) -> str: +def file_sha256(path: str | Path) -> str: + """Re-hashed per call: a suite resumes for days in one process and the binary may be rebuilt between trials.""" return hashlib.sha256(Path(path).read_bytes()).hexdigest() -@functools.lru_cache(maxsize=None) -def cached_suite(scenarios: str) -> tuple: - return tuple(bench_grade.suite_identity(Path(scenarios)).items()) +def write_json(path: Path, data: dict) -> None: + """A kill mid-write leaves no partial JSON to crash the next resume: temp file in the same directory, then replace.""" + temporary = path.with_name(f".{path.name}.tmp") + temporary.write_text(json.dumps(data, indent=2) + "\n") + os.replace(temporary, path) def base_result(scenario: dict, cell: dict, args: argparse.Namespace, out: Path, seconds: float, agent_exit: int | None) -> dict: @@ -365,7 +502,7 @@ def base_result(scenario: dict, cell: dict, args: argparse.Namespace, out: Path, "wright": subprocess.run([args.wright, "--version"], capture_output=True, text=True).stdout.strip(), "wrightSha256": file_sha256(args.wright), "harness": harness_commit(), - "suite": dict(cached_suite(str(SCENARIOS))), + "suite": bench_grade.suite_identity(Path(SCENARIOS)), # recomputed per trial: the suite may change during a days-long run "timestamp": datetime.now(timezone.utc).isoformat(timespec="seconds"), "skills": {name: skill_identity(name, Path(args.skill_dirs[name])) for name in cell["skills"]}, **({"wiki": bench_wiki.identity(Path(args.wiki_dir))} if cell["knowledge"] == "wiki" else {}), @@ -417,18 +554,43 @@ def cmd_matrix(args: argparse.Namespace) -> int: print(f"not applicable: {scenario_id} {label}", flush=True) state = {"streak": 0, "interrupted": 0, "unattempted": 0, "failed": 0} lock = threading.Lock() + base = args.config.resolve().parent # relative option paths resolve against the matrix file, so a run's manifest is self-contained + + def option_path(value: str) -> Path: + path = Path(value).expanduser() + return path if path.is_absolute() else (base / path).resolve() + + path_keys = ("wiki_dir", "out", "out_root", "wright") + options = {} + for key, val in config.get("options", {}).items(): + if key in path_keys + ("skill_dirs",) and not val: + continue # a null path option means 'unset', not an override + if key == "skill_dirs": + val = {n: option_path(v) for n, v in val.items()} + elif key in path_keys: + val = option_path(val) + elif key in ("allow_read", "deny_read") and isinstance(val, list): + val = [option_path(v) for v in val] + options[key] = val + merged = {**vars(args), **options} + if merged.get("deny_read") and not merged.get("file_sandbox"): + raise SystemExit("config options: deny_read does nothing without file_sandbox") def work(job: tuple) -> None: scenario_id, agent, cell, trial = job - out = trial_dir(args.out, scenario_id, agent["id"], cell, trial) - if (out / "result.json").is_file(): + out = trial_dir(merged["out"], scenario_id, agent["id"], cell, trial) + finished = out / "result.json" + try: + status = json.loads(finished.read_text()).get("status") if finished.is_file() else None + except json.JSONDecodeError: + status = None # a mid-write kill left a partial file; the trial is unfinished and retried + if status is not None and status != "provider-interrupted": # an interrupted trial is retried on the next run return with lock: if state["streak"] >= STOP_AFTER_INTERRUPTIONS: state["unattempted"] += 1 return - options = {k: ({n: Path(v) for n, v in val.items()} if k == "skill_dirs" else Path(val) if k == "wiki_dir" and val else val) for k, val in config.get("options", {}).items()} - trial_args = argparse.Namespace(**{**vars(args), "agent_id": agent["id"], "agent_cmd": agent["cmd"], **options}) + trial_args = argparse.Namespace(**{**merged, "agent_id": agent["id"], "agent_cmd": agent["cmd"]}) result = run_trial(load_scenario(scenario_id), cell, trial_args, out) with lock: state["streak"] = state["streak"] + 1 if result["status"] == "provider-interrupted" else 0 @@ -443,6 +605,180 @@ def work(job: tuple) -> None: return 3 if state["interrupted"] or state["unattempted"] else 1 if state["failed"] else 0 +ADAPTERS = {"claude-code": "claude_code.py", "pi": "pi.py", "devin": "devin.py", "codex": "codex.py", "agy": "agy.py", "opencode": "opencode.py", "grok": "grok.py", "direct": "direct.py"} +CANONICAL_CELL = {"tool": "wright", "skills": ["wright-skill"], "knowledge": "none", "network": "off"} +CONTROL_CELLS = [ + {"tool": "none", "skills": [], "knowledge": "none", "network": "off"}, + {"tool": "wright", "skills": [], "knowledge": "none", "network": "off"}, + CANONICAL_CELL, + {"tool": "overpy", "skills": [], "knowledge": "none", "network": "off"}, + {"tool": "overpy", "skills": ["opy-skill"], "knowledge": "none", "network": "off"}, +] + + +ADAPTER_BINARY = {"claude-code": "claude", "pi": "pi", "devin": "devin", "codex": "codex", "agy": "agy", "opencode": "opencode", "grok": "grok"} +DIRECT_ENV = ("ANTHROPIC_API_KEY", "ANTHROPIC_BASE_URL", "OPENAI_API_KEY", "OPENAI_BASE_URL") +CONFIG_PATH = Path(os.environ.get("WRIGHT_BENCH_CONFIG", Path.home() / ".config/wright-agent-bench/config.json")) + + +def user_defaults() -> dict: + """Per-user defaults for the long options, so a run is `evaluate --adapter A --model M`: keys wright, out, wiki_dir, env_pass, deny_read, skill_dirs {name: dir}.""" + if not CONFIG_PATH.is_file(): + return {} + config = json.loads(CONFIG_PATH.read_text()) + unknown = set(config) - {"wright", "out", "wiki_dir", "env_pass", "deny_read", "allow_read", "skill_dirs", "models"} + if unknown: + raise SystemExit(f"{CONFIG_PATH}: unknown key(s) {sorted(unknown)}") + for i, entry in enumerate(config.get("models", [])): + if not isinstance(entry, dict) or entry.get("adapter") not in ADAPTERS or not entry.get("model"): + raise SystemExit(f"{CONFIG_PATH}: models[{i}] needs an adapter in {sorted(ADAPTERS)} and a model") + defaults = {k: v for k, v in config.items() if k not in ("skill_dirs", "out", "wiki_dir")} + defaults.update({k: Path(v).expanduser() for k, v in config.items() if k in ("out", "wiki_dir")}) + if "skill_dirs" in config: + defaults["skill_dir"] = [f"{name}={Path(d).expanduser()}" for name, d in config["skill_dirs"].items()] + return defaults + + +def preflight(args: argparse.Namespace, cells: list[dict]) -> list[str]: + """Problems that would waste a run, found before it starts.""" + problems = [] + if not Path(args.wright).is_file(): + problems.append(f"wright binary not found: {args.wright} (pass --wright or put `wright` on PATH)") + binary = ADAPTER_BINARY.get(args.adapter) + if binary and not shutil.which(binary): + problems.append(f"`{binary}` is not on PATH, which adapter '{args.adapter}' needs") + if args.adapter == "direct" and not any(os.environ.get(k) for k in ("ANTHROPIC_API_KEY", "OPENAI_API_KEY")): + problems.append("adapter 'direct' needs ANTHROPIC_API_KEY or OPENAI_API_KEY in the environment") + for name in sorted({s for c in cells for s in c["skills"]}): + if not (Path(args.skill_dirs.get(name, "/nonexistent")) / "SKILL.md").is_file(): + problems.append(f"skill '{name}' needs --skill-dir {name}=DIR (or skill_dirs in {CONFIG_PATH})") + if any(c["tool"] == "overpy" for c in cells) and not bench_grade.oracle_available(): + problems.append("tool 'overpy' needs the pinned oracle: run `agent_bench.py setup-oracle`") + if getattr(args, "file_sandbox", False) and (sys.platform != "darwin" or not shutil.which("sandbox-exec")): + problems.append("the file sandbox needs macOS sandbox-exec; pass --no-file-sandbox for an unprotected run") + for real_rel, _ in CREDENTIALS.get(getattr(args, "adapter", ""), []): + if not (Path.home() / real_rel).is_file(): + problems.append(f"adapter '{args.adapter}' needs ~/{real_rel} (sign in with that program first)") + return problems + + +def wright_mismatch(run_dir: Path, wright: str) -> str | None: + """Why a repeated run must not continue with this Wright binary, or None. + + A run is repeated to finish it later, possibly after `wright` on PATH was upgraded. Trials made with two binaries cannot be scored + together, so the mismatch is refused up front instead of being found when the score card is refused.""" + if not Path(wright).is_file(): + return None # preflight already names the missing binary + current = file_sha256(Path(wright)) + for path in sorted(run_dir.glob("*/*/*/result.json")): + try: + env = json.loads(path.read_text()).get("environment", {}) + except json.JSONDecodeError: + continue # a partial file means the trial never finished; it will be retried + recorded = env.get("wrightSha256") + if recorded and recorded != current: + return (f"{run_dir.name} already has trials made with {env.get('wright')} (sha256 {recorded[:12]}), but {wright} is a different binary. " + f"Pass --wright with the binary that made them, or start a new run with --name.") + return None + + +def cmd_evaluate(args: argparse.Namespace) -> int: + """One command from agent and model to data and document: run the matrix, then write report, score cards, and RESULTS.md. + + Needs no agent harness around it: isolation comes from the harness's own scrubbed environment, so it runs the same from a + terminal or from inside another agent's shell.""" + args.file_sandbox = not args.no_file_sandbox # evaluation hides the answer keys and the rest of the host from the agent by default + if args.deny_read and not args.file_sandbox: + raise SystemExit("--deny-read does nothing without the file sandbox") + args.credentials = CREDENTIALS.get(args.adapter, []) + OPTIONAL_CREDENTIALS.get(args.adapter, []) + args.allow_read = [*(str(Path.home() / rel) for rel in ADAPTER_READS[args.adapter]), *args.allow_read] + if args.adapter == "claude-code" and args.file_sandbox: + print("note: the file sandbox lets claude-code read but not refresh its login; pass --no-file-sandbox when its token may rotate mid-run", flush=True) + script = Path(__file__).parent / "adapters" / ADAPTERS[args.adapter] + if not args.name or args.name in (".", "..") or Path(args.name).name != args.name: + raise SystemExit("--name must be a single directory name") + effort = f"BENCH_THINKING={shlex.quote(args.effort)} " if args.effort else "" + agent_id = model_slug({"adapter": args.adapter, "model": args.model, "effort": args.effort}) + cmd = f"BENCH_MODEL={shlex.quote(args.model)} {effort}{shlex.quote(sys.executable)} {shlex.quote(str(script))}" + wanted = [CANONICAL_CELL] if args.cells == "score" else CONTROL_CELLS + cells = [c for c in wanted if all(s in args.skill_dirs for s in c["skills"])] + dropped = [cell_label(normalize_cell(c)) for c in wanted if c not in cells] + if dropped: + print(f"skipped cells without --skill-dir: {', '.join(dropped)}", flush=True) + if not cells: + raise SystemExit("no cell can run: pass --skill-dir wright-skill=DIR") + args.env_pass = sorted({*args.env_pass, *(DIRECT_ENV if args.adapter == "direct" else ("HOME",))}) # the credentials the adapter copies or reads + problems = preflight(args, [normalize_cell(c) for c in cells]) + if (problem := wright_mismatch(args.out / args.name, args.wright)): + problems.append(problem) + if problems: + raise SystemExit("cannot start:\n " + "\n ".join(problems)) + scenarios = args.scenarios or [s for s in all_scenario_ids() if args.split == "all" or load_scenario(s).get("split") == args.split] + runnable = sum(args.trials for s in scenarios for c in cells if applicable(load_scenario(s), normalize_cell(c))) + print(f"{args.adapter} {args.model}: {len(scenarios)} scenario(s), cells {', '.join(cell_label(normalize_cell(c)) for c in cells)}, {runnable} trial(s) into {args.out / args.name}", flush=True) + if args.dry_run: + return 0 + args.out_root = args.out # sibling evaluation runs must stay unreadable too + args.out = args.out / args.name + args.out.mkdir(parents=True, exist_ok=True) + # every option a trial reads is serialized at its effective value, so `matrix` on this file reproduces the run + config = {"agents": [{"id": agent_id, "cmd": cmd}], "cells": cells, "scenarios": scenarios, "trials": args.trials, "parallel": args.parallel, "seed": args.seed, + "options": {"out": ".", "out_root": "..", "wright": args.wright, "adapter": args.adapter, + "file_sandbox": args.file_sandbox, "env_pass": args.env_pass, "credentials": args.credentials, + # paths the caller gave resolve now, against this cwd — relative ones in the file resolve against the file's directory + "allow_read": [str(Path(p).expanduser().resolve()) for p in args.allow_read], + "deny_read": [str(Path(p).expanduser().resolve()) for p in args.deny_read], + "timeout": args.timeout, "canary_cmd": args.canary_cmd, "check_ancestors": args.check_ancestors, + "infra_retries": args.infra_retries, "infra_backoff": args.infra_backoff, + "skill_dirs": {k: str(Path(v).expanduser().resolve()) for k, v in args.skill_dirs.items()}, + "wiki_dir": str(args.wiki_dir.expanduser().resolve()) if args.wiki_dir else None}} + args.config = args.out / "matrix.json" + args.config.write_text(json.dumps(config, indent=2) + "\n") + status = cmd_matrix(args) + if not list(args.out.glob("*/*/*/result.json")): + return status or 1 + bench_report.main([args.out], args.wright, False, load_scenario, bench_report.BASELINE) + languages = ["workshop", "opy"] + expected = {lang: [s for s in all_scenario_ids() if load_scenario(s)["language"] == lang and load_scenario(s).get("split") == "test"] for lang in languages} + bench_score.main([args.out], languages, expected, None) + (args.out / "RESULTS.md").write_text(f"# Agent benchmark results: {args.name}\n\nAgent `{agent_id}`. Generated by `agent_bench.py evaluate`; `agent_bench.py matrix /matrix.json` resumes it in place.\n\n" + f"## Score\n\n```\n{(args.out / 'score.txt').read_text()}```\n\n{(args.out / 'report.md').read_text()}") + print(f"wrote {args.out / 'RESULTS.md'}") + return status + + +def model_slug(entry: dict) -> str: + effort = entry.get("effort") + return "-".join(filter(None, (entry["adapter"], entry["model"].replace("/", "_"), None if effort and entry["model"].endswith(f"-{effort}") else effort))) # an effort the model id names is not repeated + + +def cmd_suite(args: argparse.Namespace) -> int: + """Evaluate every model of the user's list one after another, then write the publishable results page. + + Safe to repeat: finished runs are skipped, so quota or time limits only postpone the rest. Exit 3 when something is waiting for a rerun.""" + models = [m for m in (args.models or []) if not args.only or any(m["adapter"] == o or f"{m['adapter']}:{m['model']}" == o for o in args.only)] # exact adapter or adapter:model — 'codex:gpt-6' must not swallow 'codex:gpt-6-luna' + if not models: + raise SystemExit(f'no models: add "models": [{{"adapter": "devin", "model": "swe-2-max"}}, ...] to {CONFIG_PATH}') + if not args.suite_name or args.suite_name in (".", "..") or Path(args.suite_name).name != args.suite_name: + raise SystemExit("--suite-name must be a single directory name") + root = args.out / args.suite_name + outcome = {} + for entry in models: + slug = model_slug(entry) + print(f"\n=== {slug}", flush=True) + sub = argparse.Namespace(**{**vars(args), "adapter": entry["adapter"], "model": entry["model"], "effort": entry.get("effort"), "name": slug, "out": root}) + try: + outcome[slug] = {0: "done", 3: "waiting: provider limit or outage, rerun later"}.get(cmd_evaluate(sub), "finished with errors") + except SystemExit as stop: + outcome[slug] = f"skipped: {stop.code}" + print("\n" + "\n".join(f"{slug}: {state}" for slug, state in outcome.items())) + if not args.dry_run: + bench_leaderboard.main(sorted(root.glob("*/")), root / "leaderboard") + if any(state.startswith("waiting") for state in outcome.values()): + return 3 + return 0 if all(state == "done" for state in outcome.values()) else 1 + + def cmd_wiki_snapshot(args: argparse.Namespace) -> int: record = bench_wiki.snapshot(args.base, args.dir, tuple(args.categories)) print(f"{len(record['documents'])} document(s) from {record['source']} into {args.dir}\nsnapshotSha256 {record['snapshotSha256']}") @@ -460,15 +796,14 @@ def cmd_setup_oracle(_: argparse.Namespace) -> int: def main() -> int: - if len(sys.argv) > 1 and sys.argv[1] == "shim": - return bench_trace.shim_main(sys.argv[2:]) parser = argparse.ArgumentParser(description=__doc__) sub = parser.add_subparsers(dest="command", required=True) - for name in ("validate", "run", "matrix"): - p = sub.add_parser(name) + sub.add_parser("suite", help="evaluate every model listed in the user config in turn, then write the results page") + for name in ("validate", "run", "matrix", "evaluate", "suite"): + p = sub.choices[name] if name == "suite" else sub.add_parser(name) p.add_argument("--wright", default=str(ROOT / "target/debug/wright"), help="Wright binary under test") - p.add_argument("--out", type=Path, default=Path.home() / ".cache/wright-agent-bench", help="outside any repository, so agents cannot discover its instruction files") - for name in ("run", "matrix"): + p.add_argument("--out", type=Path, default=BENCH_HOME / "runs", help="outside any repository, so agents cannot discover its instruction files") + for name in ("run", "matrix", "evaluate", "suite"): p = sub.choices[name] p.add_argument("--skill-dir", action="append", default=[], metavar="NAME=DIR", help=f"pinned skill directory for one of {', '.join(SKILLS)}; repeatable") p.add_argument("--wiki-dir", type=Path, help="pinned wiki snapshot, linked as ./wiki for knowledge 'wiki'; content hashes are verified") @@ -476,8 +811,12 @@ def main() -> int: p.add_argument("--no-ancestor-check", dest="check_ancestors", action="store_false", help="skip the check for instruction files above the workspace") p.add_argument("--canary-cmd", help="shell command that must fail in the agent environment when the network is 'off'") p.add_argument("--timeout", type=int, default=1800) - p.add_argument("--file-sandbox", action="store_true", help="macOS: restrict agent and descendant file writes to the trial directory") + p.add_argument("--deny-read", nargs="*", default=[], metavar="PATH", help="extra directories hidden from the agent (the home directories and drives already are); needs the file sandbox") + p.add_argument("--allow-read", nargs="*", default=[], metavar="PATH", help="paths the agent's CLI needs inside the hidden home directories (credentials, installation); `evaluate` adds its adapter's") p.add_argument("--infra-retries", type=int, default=2, help="retries when the agent exits 75 (provider or infrastructure failure)") + p.add_argument("--infra-backoff", type=int, default=60, help="seconds before the first retry; each further retry waits one more multiple") + for p in (sub.choices["run"], sub.choices["matrix"]): + p.add_argument("--file-sandbox", action="store_true", help="macOS: restrict agent and descendant file writes to the trial directory, and hide the scenarios, other runs, the wiki, unselected skills, and --deny-read paths") run = sub.choices["run"] run.add_argument("scenario", choices=all_scenario_ids()) run.add_argument("--agent-cmd", required=True, help="shell command; the task prompt arrives on stdin, cwd is the workspace, BENCH_* describes the condition") @@ -488,6 +827,33 @@ def main() -> int: run.add_argument("--network", choices=("off", "on"), default="off") run.add_argument("--trials", type=int, default=1) sub.choices["matrix"].add_argument("config", type=Path, help="JSON: agents[{id,cmd}], cells[{tool,skills,knowledge,network}], scenarios, trials, parallel, seed, options") + ev = sub.choices["evaluate"] + ev.add_argument("--adapter", choices=sorted(ADAPTERS), required=True, help="agent adapter; `direct` is the built-in loop that needs no agent harness") + ev.add_argument("--model", required=True, help="BENCH_MODEL, in the form the adapter expects") + ev.add_argument("--effort", help="BENCH_THINKING, where the adapter supports it") + ev.add_argument("--name", default=time.strftime("%Y%m%d-%H%M%S"), help="run directory under --out") + su = sub.choices["suite"] + su.add_argument("--suite-name", default="results", help="directory under --out holding every model's run and the results page") + su.add_argument("--only", nargs="*", metavar="ADAPTER[:MODEL]", help="evaluate only these entries of the models list") + su.add_argument("--cells", choices=("score", "controls"), default="score") + su.add_argument("--split", choices=("test", "train", "all"), default="test") + su.add_argument("--scenarios", nargs="*", choices=all_scenario_ids()) + su.add_argument("--trials", type=int, default=3) + su.add_argument("--parallel", type=int, default=1) + su.add_argument("--seed", type=int, default=1) + su.add_argument("--dry-run", action="store_true") + su.add_argument("--no-file-sandbox", action="store_true") + lb = sub.add_parser("leaderboard", help="write the publishable results page (Markdown, HTML, JSON) from evaluation run directories") + lb.add_argument("dirs", nargs="+", type=Path) + lb.add_argument("--page-out", type=Path, help="directory for the page; `leaderboard` inside the first directory's parent by default") + ev.add_argument("--cells", choices=("score", "controls"), default="score", help="score: the canonical cell only; controls: also baseline and language-appropriate controls") + ev.add_argument("--split", choices=("test", "train", "all"), default="test") + ev.add_argument("--scenarios", nargs="*", choices=all_scenario_ids()) + ev.add_argument("--trials", type=int, default=3) + ev.add_argument("--parallel", type=int, default=1, help="trials at a time; sequential by default so provider limits are not hit, and a run can continue across sessions") + ev.add_argument("--seed", type=int, default=1) + ev.add_argument("--dry-run", action="store_true", help="check the setup and print what would run, without running it") + ev.add_argument("--no-file-sandbox", action="store_true", help="run without the macOS file sandbox: the agent can then read the scenario answer keys") sub.add_parser("setup-oracle", help="install the pinned upstream OverPy oracle") skill = sub.add_parser("wiki-skill", help="build the progressive-disclosure workshop-wiki skill from a wiki snapshot") skill.add_argument("--snapshot", type=Path, required=True) @@ -495,7 +861,7 @@ def main() -> int: skill.add_argument("--catalog", type=Path, required=True, help="workshop-rs catalog.json, for Workshop names and ids") skill.add_argument("--opy-manifest", type=Path, required=True, help="opy-rs manifest.json, for upstream OverPy spellings") wiki = sub.add_parser("wiki-snapshot", help="fetch the Workshop wiki Markdown mirror into a pinned local snapshot") - wiki.add_argument("--dir", type=Path, default=Path.home() / ".cache/wright-agent-bench-wiki") + wiki.add_argument("--dir", type=Path, default=BENCH_HOME / "wiki") wiki.add_argument("--base", default=bench_wiki.BASE) wiki.add_argument("--categories", nargs="+", default=list(bench_wiki.CATEGORIES), help="wiki categories to crawl (add tutorials for the second tier)") report = sub.add_parser("report", help="summarize result.json files") @@ -503,9 +869,18 @@ def main() -> int: report.add_argument("--regrade", action="store_true", help="re-grade stored workspaces twice and flag unstable graders") report.add_argument("--wright", default=str(ROOT / "target/debug/wright")) report.add_argument("--reference", default=bench_report.BASELINE, help="condition label the paired comparison is made against") + compare = sub.add_parser("compare", help="one table from the score.json of several evaluation runs, warning when they are not comparable") + compare.add_argument("dirs", nargs="+", type=Path) score = sub.add_parser("score", help="compute the Wright Agent Score card of each language track from canonical test runs") score.add_argument("dirs", nargs="+", type=Path) score.add_argument("--language", choices=("workshop", "opy"), action="append", help="track to score; both when omitted") + defaults = user_defaults() + for choice in sub.choices.values(): + known = {a.dest for a in choice._actions} + choice.set_defaults(**{k: v for k, v in defaults.items() if k in known}) + for name in ("evaluate", "suite"): + sub.choices[name].set_defaults(wright=defaults.get("wright") or shutil.which("wright") or str(ROOT / "target/debug/wright")) + sub.choices["suite"].set_defaults(models=defaults.get("models")) args = parser.parse_args() if hasattr(args, "wright"): args.wright = str(Path(args.wright).resolve()) @@ -518,6 +893,8 @@ def main() -> int: if name not in SKILLS or not directory: raise SystemExit(f"--skill-dir expects NAME=DIR with NAME one of {', '.join(SKILLS)}: {item}") args.skill_dirs[name] = Path(directory) + if getattr(args, "deny_read", None) and not getattr(args, "file_sandbox", False) and args.command not in ("evaluate", "suite"): + raise SystemExit("--deny-read needs --file-sandbox") if args.command == "validate": return 0 if validate(args.wright, args.out) else 1 if args.command == "setup-oracle": @@ -528,11 +905,16 @@ def main() -> int: return cmd_wiki_skill(args) if args.command == "report": return bench_report.main(args.dirs, args.wright, args.regrade, lambda s: load_scenario(s), args.reference) + if args.command == "leaderboard": + return bench_leaderboard.main(args.dirs, args.page_out or args.dirs[0].parent / "leaderboard") + if args.command == "compare": + print(bench_score.compare(args.dirs)) + return 0 if args.command == "score": languages = args.language or ["workshop", "opy"] expected = {lang: [s for s in all_scenario_ids() if load_scenario(s)["language"] == lang and load_scenario(s).get("split") == "test"] for lang in languages} return bench_score.main(args.dirs, languages, expected, None) - return cmd_run(args) if args.command == "run" else cmd_matrix(args) + return {"run": cmd_run, "matrix": cmd_matrix, "evaluate": cmd_evaluate, "suite": cmd_suite}[args.command](args) if __name__ == "__main__": diff --git a/benchmarks/agent/bench_grade.py b/benchmarks/agent/bench_grade.py index 6e07a4fc..e0b6c0b4 100644 --- a/benchmarks/agent/bench_grade.py +++ b/benchmarks/agent/bench_grade.py @@ -13,7 +13,7 @@ ORACLE = HERE / "oracle" GRADER_FILES = ("bench_grade.py", "oracle/compile.js", "oracle/package-lock.json") SUITE_VERSION = "v1" -UNSAFE_IGNORED = ("wiki", ".agents", ".devin") # linked wiki and skills installed through the agent's own mechanism +UNSAFE_IGNORED = ("wiki", ".agents", ".devin", ".opencode") # linked wiki and skills installed through the agent's own mechanism def wright_json(wright: str, args: list[str]) -> tuple[int, dict]: @@ -85,6 +85,14 @@ def authorities(wright: str, source: Path, scratch: Path) -> dict: return result +def missing_entry_authorities(entry: Path) -> dict: + """The agent produced no entry file: every authority rejects it, and the oracle is only unavailable when it is not installed.""" + result: dict = {"wrightCompile": {"status": "error", "error": "entry file missing"}} + if entry.suffix == ".opy" and oracle_available(): + result["oracle"] = {"status": "error", "error": "entry file missing"} + return result + + def compiled_text(state: dict, source: str) -> str | None: """Compiled Workshop text from `wright` or the upstream `oracle`, when that authority succeeded.""" auth = state["authorities"] @@ -206,7 +214,7 @@ def grade(scenario: dict, workspace: Path, wright: str, scratch: Path | None = N entry = workspace / scenario["entry"] scratch = scratch or workspace.parent / f"{workspace.name}-grading" shutil.rmtree(scratch, ignore_errors=True) - state: dict = {"scratch": scratch, "authorities": authorities(wright, entry, scratch) if entry.is_file() else {"wrightCompile": {"status": "error"}}} + state: dict = {"scratch": scratch, "authorities": authorities(wright, entry, scratch) if entry.is_file() else missing_entry_authorities(entry)} _, lint = wright_json(wright, ["lint", str(entry)]) state["lint"] = (lint.get("result") or {}).get("findings") or [] checks = [run_check(c, workspace, entry, wright, state) for c in scenario["checks"]] diff --git a/benchmarks/agent/bench_leaderboard.py b/benchmarks/agent/bench_leaderboard.py new file mode 100644 index 00000000..88681b72 --- /dev/null +++ b/benchmarks/agent/bench_leaderboard.py @@ -0,0 +1,189 @@ +"""Publishable results page (#467): one table of Wright Agent Scores for every evaluated agent, as Markdown, a self-contained HTML page, and JSON.""" + +from __future__ import annotations + +import html +import json +import re +from collections import Counter +from datetime import date +from pathlib import Path + +import bench_report + +TRACKS = (("workshop", "Workshop"), ("opy", "OverPy")) + + +def suite_shape(data: dict) -> tuple[int | None, int | None]: + """(scenarios per language, trials per scenario) when every run used the same suite shape, else (None, None).""" + counts = {(t["scenarios"], t["trials"]) for e in data["entries"] for t in e["tracks"].values()} + return next(iter(counts)) if len(counts) == 1 else (None, None) + + +def reading(data: dict) -> list[str]: + scenarios, trials = suite_shape(data) + tasks = f"{scenarios} realistic Workshop tasks" if scenarios else "the held-out tasks" + times = f"{trials} times" if trials else "a fixed number of times" + return [ + f"The score is the share of tasks an agent completed with a valid, safe result, averaged over {tasks} per language, each tried {times}.", + f"The bar shows the score; the range in brackets is the 95% confidence interval. With {scenarios or 'few'} tasks the range is wide: agents marked **tied with top** cannot be told apart from the first row.", + "It is a reference for how an agent behaves with Wright and its guide, not a measure of general ability, and a model's score depends on the agent program that runs it.", + ] + + +def limits(data: dict) -> list[str]: + scenarios, _ = suite_shape(data) + networks = {n for e in data["entries"] for n in e["network"]} + network = ("The harness verified that the network was unreachable on every run." if networks == {"canary-checked"} + else "Network access was switched off by instruction only for at least some runs; the harness did not block it.") + return [ + network, + "A run that hit the time limit counts as a failure. Runs cut off by provider outages are retried, not counted.", + f"The suite has {scenarios or 'few'} test tasks per language, so differences of a few points mean nothing.", + ] + + +def load_entries(dirs: list[Path]) -> list[dict]: + """One entry per evaluation run directory that has a score.json.""" + entries = [] + for directory in dirs: + path = directory / "score.json" + if not path.is_file(): + continue + cards = {c["language"]: c for c in json.loads(path.read_text())["cards"] if "refused" not in c} + if not cards: + continue + first = next(iter(cards.values())) + ident = first["identity"] + info = next((r.get("agentInfo") or {} for r in bench_report.load([directory]) if r.get("agentInfo")), {}) + entries.append({ + "run": directory.name, + "harness": info.get("agent") or ident["agent"], "harnessVersion": info.get("version"), + "model": info.get("model") or ident.get("model"), "effort": info.get("effort") or ident.get("effort"), + "tracks": {lang: {"score": c["score"], "ci": c["ci95"], "trials": c["trialsPerScenario"], "scenarios": c["scenarios"], "passPowK": c["passPowK"], "provisional": c["provisional"], "exclusions": c["exclusions"]} for lang, c in cards.items()}, + "environment": {"wrightSha256": ident["wrightSha256"], "wright": ident["wright"], "skills": ident["skills"], "suite": first["suite"].get("hash")}, + "network": first["networkEnforcement"], "harnessCommit": first["harness"], + }) + return entries + + +def mean_score(entry: dict) -> float: + scores = [t["score"] for t in entry["tracks"].values()] + return sum(scores) / len(scores) + + +def overlaps(a: list[float], b: list[float]) -> bool: + return a[1] >= b[0] and a[0] <= b[1] + + +def standing(entry: dict, leader: dict) -> str: + """`top`, `tied with top` when every track's interval overlaps the leader's, else `below top`.""" + if entry is leader: + return "top" + shared = [lang for lang in entry["tracks"] if lang in leader["tracks"]] + return "tied with top" if shared and all(overlaps(entry["tracks"][l]["ci"], leader["tracks"][l]["ci"]) for l in shared) else "below top" + + +def build(dirs: list[Path]) -> dict: + entries = load_entries(dirs) + if not entries: + return {"entries": [], "excluded": [], "environment": None, "date": date.today().isoformat()} + coverage = max(len(e["tracks"]) for e in entries) + covered, partial = [e for e in entries if len(e["tracks"]) == coverage], [e for e in entries if len(e["tracks"]) < coverage] + key = lambda e: json.dumps(e["environment"], sort_keys=True) + main_key = Counter(key(e) for e in covered).most_common(1)[0][0] + ranked = sorted([e for e in covered if key(e) == main_key], key=mean_score, reverse=True) + for entry in ranked: + entry["standing"] = standing(entry, ranked[0]) + other = [{"run": e["run"], "reason": "made against a different Wright binary, skills, or task suite"} for e in covered if key(e) != main_key] + other += [{"run": e["run"], "reason": f"covers {len(e['tracks'])} of {coverage} language tracks"} for e in partial] + return {"entries": ranked, "excluded": other, "environment": ranked[0]["environment"], "date": date.today().isoformat()} + + +def label(entry: dict) -> str: + version = re.sub(r" \([0-9a-f]{7,}\)$", "", entry["harnessVersion"] or "") # the build hash adds nothing for a reader + return version if version.startswith(entry["harness"]) else " ".join(filter(None, (entry["harness"], version))) + + +def bar(score: float, width: int = 10) -> str: + filled = round(score / 100 * width) + return "█" * filled + "░" * (width - filled) + + +def markdown(data: dict) -> str: + env = data["environment"] + lines = ["# Wright Agent Score", "", f"How well coding agents work on real Overwatch Workshop projects with Wright. Results of {data['date']}.", ""] + if not data["entries"]: + return "\n".join(lines + ["No results yet.", ""]) + lines += ["| # | Agent | Model | Effort | " + " | ".join(f"{n} score" for _, n in TRACKS) + " | Against the top |", "| --- | --- | --- | --- | " + " | ".join("---" for _ in TRACKS) + " | --- |"] + cell = lambda s: str(s if s is not None else "not recorded").replace("|", "\\|") # a raw | would break the table row + for rank, e in enumerate(data["entries"], 1): + cells = [] + for lang, _ in TRACKS: + t = e["tracks"].get(lang) + cells.append(f"`{bar(t['score'])}` **{t['score']:.0f}** ({t['ci'][0]:.0f}–{t['ci'][1]:.0f})" + (" ⚠️" if t["provisional"] else "") if t else "n/a") + lines.append(f"| {rank} | {cell(label(e))} | {cell(e['model'])} | {cell(e['effort'] or 'default')} | " + " | ".join(cells) + f" | {e['standing']} |") + lines += ["", "## How to read this", "", *(f"- {s}" for s in reading(data)), "", "## Limits", "", *(f"- {s}" for s in limits(data))] + if any(t["provisional"] for e in data["entries"] for t in e["tracks"].values()): + lines.append("- ⚠️ marks a score that is provisional because the run did not cover every task or had unequal trials.") + skills = ", ".join(f"{k} `{v[:8]}`" for k, v in (env["skills"] or {}).items()) or "none" + commits = ", ".join(f"`{c[:8]}`" for c in sorted({c for e in data["entries"] for c in e["harnessCommit"]})) or "not recorded" + lines += ["", "## What was run", "", f"- Wright: {env['wright'] or 'not recorded'} (sha256 `{(env['wrightSha256'] or 'not recorded')[:12]}`)", f"- Skills: {skills}", f"- Task suite: `{(env['suite'] or 'not recorded')[:12]}`", + f"- Harness commit: {commits}", "- Scores come from the benchmark in `benchmarks/agent`; the run directories hold every result.json."] + if data["excluded"]: + lines += ["", "## Not comparable", "", *(f"- `{o['run']}`: {o['reason']}." for o in data["excluded"])] + return "\n".join(lines) + "\n" + + +def page(data: dict) -> str: + esc = html.escape + rows = [] + for rank, e in enumerate(data["entries"], 1): + cells = [] + for lang, _ in TRACKS: + t = e["tracks"].get(lang) + if not t: + cells.append("n/a") + continue + lo, hi = t["ci"] + cells.append(f'
' + f'{t["score"]:.0f} {lo:.0f}–{hi:.0f}{" ⚠️" if t["provisional"] else ""}') + rows.append(f'{rank}{esc(label(e))}{esc(str(e["model"] or "not recorded"))}{esc(e["effort"] or "default")}{"".join(cells)}{esc(e["standing"])}') + env = data["environment"] or {} + skills = ", ".join(f"{k} {v[:8]}" for k, v in (env.get("skills") or {}).items()) or "none" + head = "".join(f"{n} score" for _, n in TRACKS) + body = (f"{head}{''.join(rows)}
#AgentModelEffortAgainst the top
" + if rows else "

No results yet.

") + li = lambda items: "".join(f"
  • {esc(s.replace('**', ''))}
  • " for s in items) + excluded = ("

    Not comparable

      " + "".join(f"
    • {esc(o['run'])}: {esc(o['reason'])}
    • " for o in data["excluded"]) + "
    " if data["excluded"] else "") + prose = (f"

    How to read this

      {li(reading(data))}

    Limits

      {li(limits(data))}
    " if data["entries"] else "") + sha12 = lambda v: esc(str(v or "not recorded")[:12]) + return f""" +Wright Agent Score +
    +

    Wright Agent Score

    How well coding agents work on real Overwatch Workshop projects with Wright. Results of {esc(data.get("date", ""))}.

    +{body} +{prose}{excluded} +

    What was run

    Wright {esc(str(env.get("wright") or "not recorded"))} · sha256 {sha12(env.get("wrightSha256"))} · skills {esc(skills)} · task suite {sha12(env.get("suite"))}

    +
    +""" + + +def main(dirs: list[Path], out: Path) -> int: + data = build(dirs) + out.mkdir(parents=True, exist_ok=True) + (out / "LEADERBOARD.md").write_text(markdown(data)) + (out / "leaderboard.html").write_text(page(data)) + (out / "leaderboard.json").write_text(json.dumps(data, indent=2) + "\n") + print(markdown(data)) + print(f"wrote {out / 'LEADERBOARD.md'}, {out / 'leaderboard.html'}, {out / 'leaderboard.json'}") + return 0 if data["entries"] else 1 diff --git a/benchmarks/agent/bench_report.py b/benchmarks/agent/bench_report.py index 162de014..f37f2105 100644 --- a/benchmarks/agent/bench_report.py +++ b/benchmarks/agent/bench_report.py @@ -32,7 +32,7 @@ def load(dirs: list[Path]) -> list[dict]: for base in dirs: for path in sorted(base.rglob("result.json")): result = json.loads(path.read_text()) - if str(result.get("contract", "")).startswith("wright-agent-bench/"): + if str(result.get("contract", "")).startswith("wright-agent-bench/") and all(k in result for k in ("status", "language", "condition", "scenario", "agent", "environment")): result["_dir"] = path.parent match = re.search(r"-(\d+)$", path.parent.name) result["_trial"] = int(match.group(1)) if match else 0 @@ -147,6 +147,22 @@ def regrade_notes(runs: list[dict], wright: str, load_scenario) -> list[str]: return [f"GRADER: unstable verdict on {len(unstable)} workspace(s): {unstable}"] if unstable else ["GRADER: consistent on every regraded workspace."] +def setup_rows(runs: list[dict]) -> list[str]: + """What each agent was actually given: CLI version, model, tools, and loaded skills, as the adapters observed them.""" + out = ["", "## Agent setup", "", "What the adapters observed, not what was requested. `not recorded` means the CLI does not expose it.", "", + "| agent | condition | CLI | model | effort | tools | loaded skills |", "| --- | --- | --- | --- | --- | --- | --- |"] + for agent in sorted({r["agent"]["id"] for r in runs}): + for cell in sorted({label(r) for r in runs if r["agent"]["id"] == agent}): + group = [r for r in runs if r["agent"]["id"] == agent and label(r) == cell] + infos = [r.get("agentInfo") or {} for r in group] + first = next((i for i in infos if i), {}) + tools = first.get("tools") or first.get("toolsObserved") + tool_text = "not recorded" if tools is None else f"{len(tools)}: {', '.join(tools[:12])}{' ...' if len(tools) > 12 else ''}" if tools else "none" + loaded = sorted({s for r in group for s in ((r.get("context") or {}).get("loaded") or [])}) + out.append(f"| {agent} | {cell} | {first.get('version') or 'not recorded'} | {first.get('model') or 'not recorded'} | {first.get('effort') or 'not recorded'} | {tool_text} | {', '.join(loaded) or 'none'} |") + return out + + def render(results: list[dict], regrade: list[str] | None = None, reference: str = BASELINE) -> tuple[str, dict]: invalid = [r for r in results if r["status"] == "invalid"] infrastructure = [r for r in results if r["status"] == "provider-interrupted"] @@ -163,6 +179,7 @@ def render(results: list[dict], regrade: list[str] | None = None, reference: str summary["cells"][f"{agent}|{cell}"] = row out.append(f"| {agent} | {cell} | {row['n']} | {rate_runs(group)} | {row['passed']}/{row['n']} | {row['usedTool']}/{row['n']} | " f"{fmt(row['tokens'])} | {fmt(row['tokensPerUsable'])} | {fmt(row['peakContext'])} | {fmt(row['seconds'], 1)} |") + out += setup_rows(runs) out += ["", "## By scenario", "", "| scenario | agent | condition | usable |", "| --- | --- | --- | --- |"] groups: dict[tuple, list[dict]] = defaultdict(list) for r in runs: diff --git a/benchmarks/agent/bench_score.py b/benchmarks/agent/bench_score.py index 7e1e140b..aa10a2f7 100644 --- a/benchmarks/agent/bench_score.py +++ b/benchmarks/agent/bench_score.py @@ -14,7 +14,7 @@ BOOTSTRAP_DRAWS = 10_000 BOOTSTRAP_SEED = 467 METHOD = f"two-stage percentile bootstrap (scenarios, then trials within a scenario), {BOOTSTRAP_DRAWS} draws, seed {BOOTSTRAP_SEED}" -EXCLUDED = ("provider-interrupted", "invalid", "agent-error") # reported separately; a timeout is the agent's own outcome and counts +VALID = ("completed", "timeout") # a timeout is the agent's own outcome and counts; every other status is excluded and reported separately TRACKS = {"workshop": "Wright Workshop Agent Score", "opy": "Wright OPY Agent Score"} @@ -38,6 +38,12 @@ def pass_power_k(usable: int, valid: int, k: int) -> float: return comb(usable, k) / comb(valid, k) if valid >= k else 0.0 +def read_enforcement(run: dict) -> str | None: + """The file-read policy in one word: 'allow-list' is recorded with its hidden/allowed paths, which differ per trial.""" + enforcement = run.get("fileReadEnforcement") + return enforcement.get("mode") if isinstance(enforcement, dict) else enforcement + + def identity_of(run: dict) -> dict: env = run["environment"] info = run.get("agentInfo") or {} @@ -47,16 +53,18 @@ def identity_of(run: dict) -> dict: "suite": (env.get("suite") or {}).get("hash"), "agent": run["agent"]["id"], "model": info.get("model"), "effort": info.get("effort"), "protocol": run.get("protocol"), + "fileReadEnforcement": read_enforcement(run), "fileWriteEnforcement": run.get("fileWriteEnforcement"), + "networkEnforcement": run.get("networkEnforcement"), # what the agent could reach is part of the environment being scored } def card(results: list[dict], language: str, expected: list[str]) -> dict: """The score card of one language track, or a refusal when the runs are not one comparable environment.""" - track = [r for r in results if r["language"] == language and r["condition"]["label"] == CANONICAL and r.get("split") == "test"] + track = [r for r in results if r.get("language") == language and (r.get("condition") or {}).get("label") == CANONICAL and r.get("split") == "test"] excluded = defaultdict(list) valid = [] for run in track: - (excluded[run["status"]] if run["status"] in EXCLUDED else valid).append(run) + (valid if run.get("status") in VALID else excluded[run.get("status") or "missing-status"]).append(run) if not valid: return {"contract": CONTRACT, "track": TRACKS[language], "refused": "no valid canonical test runs for this language"} identities = {json.dumps(identity_of(r), sort_keys=True) for r in valid} @@ -99,8 +107,9 @@ def card(results: list[dict], language: str, expected: list[str]) -> dict: "identity": identity, "suite": (valid[0]["environment"].get("suite") or {}), "harness": sorted({r["environment"].get("harness") for r in valid if r["environment"].get("harness")}), - "networkEnforcement": sorted({r.get("networkEnforcement") for r in valid}), - "fileWriteEnforcement": sorted({r.get("fileWriteEnforcement") for r in valid}), + "networkEnforcement": sorted({r.get("networkEnforcement") or "not recorded" for r in valid}), + "fileWriteEnforcement": sorted({r.get("fileWriteEnforcement") or "not recorded" for r in valid}), + "fileReadEnforcement": sorted({read_enforcement(r) or "not recorded" for r in valid}), "exclusions": {status: len(runs) for status, runs in sorted(excluded.items())}, "perScenario": [{"scenario": s, "usable": sum(v), "valid": len(v), "rate": round(rates[s], 3)} for s, v in sorted(per.items())], "byFamily": {f: round(100 * sum(v) / len(v), 1) for f, v in sorted(families.items())}, @@ -126,6 +135,7 @@ def render(c: dict) -> str: f"Suite: {c['suite'].get('version')} {c['suite'].get('hash')}", f"Wright: {ident['wright']} sha256 {ident['wrightSha256']}", f"Skills: {', '.join(f'{n} {h}' for n, h in ident['skills'].items()) or 'none'}", f"Harness: {', '.join(c['harness']) or 'not recorded'}", f"Network: {', '.join(x or 'not recorded' for x in c['networkEnforcement'])}", f"File writes: {', '.join(x or 'not recorded' for x in c['fileWriteEnforcement'])}", + f"File reads: {', '.join(x or 'not recorded' for x in c['fileReadEnforcement'])}", f"Excluded: {c['exclusions'] or 'none'}", ] if c["provisional"]: @@ -137,6 +147,34 @@ def render(c: dict) -> str: return "\n".join(lines) + "\n" +COMPARABLE = ("wrightSha256", "skills", "suite", "fileReadEnforcement", "fileWriteEnforcement", "networkEnforcement") # what must match for two cards to be read side by side; agent, model, and effort are what is being compared + + +def compare(dirs: list[Path]) -> str: + """One table from the score cards of several evaluation runs, with a warning when they were not made against the same Wright, skills, and suite.""" + rows, identities = [], {} + for directory in dirs: + path = directory / "score.json" + if not path.is_file(): + rows.append(("", directory.name, "no score", "no score.json in this directory")) + continue + for card_ in json.loads(path.read_text())["cards"]: + if "refused" in card_: + rows.append((card_["track"], directory.name, "no score", card_["refused"])) + continue + ident = card_["identity"] + identities.setdefault(card_["track"], []).append((directory.name, {k: ident.get(k) for k in COMPARABLE})) # cards written before a field existed compare as 'not recorded' + label = " ".join(filter(None, (ident.get("agent"), ident.get("effort") and f"effort {ident['effort']}"))) + note = "provisional: " + "; ".join(card_["provisional"]) if card_["provisional"] else "" + rows.append((card_["track"], directory.name, f"{card_['score']} [{card_['ci95'][0]}-{card_['ci95'][1]}]", f"{label}; {card_['trialsPerScenario']} trial(s) x {card_['scenarios']} scenarios; excluded {card_['exclusions'] or 'none'}. {note}".strip())) + lines = ["| track | run | score [95% CI] | agent and notes |", "| --- | --- | --- | --- |", *(f"| {t} | {d} | {s} | {n} |" for t, d, s, n in sorted(rows))] + for track, entries in identities.items(): + differing = sorted({k for _, a in entries for _, b in entries for k in COMPARABLE if a[k] != b[k]}) + if differing: + lines.append(f"\nWARNING {track}: runs differ in {', '.join(differing)}, so these scores are not directly comparable.") + return "\n".join(lines) + "\n" + + def main(dirs: list[Path], languages: list[str], expected_by_language: dict[str, list[str]], out: Path | None) -> int: from bench_report import load results = load(dirs) diff --git a/benchmarks/agent/bench_trace.py b/benchmarks/agent/bench_trace.py index a46c3f8e..9a50f8af 100644 --- a/benchmarks/agent/bench_trace.py +++ b/benchmarks/agent/bench_trace.py @@ -298,3 +298,7 @@ def total(row: dict) -> int: "peakContextShare": round(peak["context"] / limit, 4) if limit and peak.get("context") else None, "toFirstValid": to_first, } + + +if __name__ == "__main__": + sys.exit(shim_main(sys.argv[2:])) # the tool shims run this file directly: `bench_trace.py shim args...` diff --git a/benchmarks/agent/test_adapters.py b/benchmarks/agent/test_adapters.py index f2e42d8c..b5a5dc5f 100644 --- a/benchmarks/agent/test_adapters.py +++ b/benchmarks/agent/test_adapters.py @@ -11,6 +11,13 @@ import pi import codex import agy +import direct +import opencode +import grok +import os +import tempfile +import threading +from http.server import BaseHTTPRequestHandler, HTTPServer class PiAdapterTest(unittest.TestCase): @@ -46,6 +53,7 @@ def test_recovered_provider_error_does_not_fail_completed_task(self): patch.object(pi.Path, "is_file", return_value=False), patch.object(pi.Path, "write_text"), patch.object(pi, "context_limit", return_value=None), + patch.object(pi, "cli_version", return_value="pi 0"), patch.object(pi.subprocess, "Popen", return_value=proc), patch("builtins.open", side_effect=lambda *a, **kw: io.StringIO()), ): @@ -89,11 +97,38 @@ def test_config_reads_no_other_tools_and_denies_web_unless_asked(self): self.assertFalse(any(closed["read_config_from"].values())) self.assertEqual(closed["agent"]["model"], "swe-2-max") self.assertIn("web_search", closed["permissions"]["deny"]) + self.assertIn("web_search", closed["disabled_tools"]) + self.assertIn("mcp_call_tool", closed["disabled_tools"]) + self.assertNotIn("web_search", opened["disabled_tools"]) self.assertNotIn("web_search", opened["permissions"]["deny"]) self.assertIn("mcp_call_tool", opened["permissions"]["deny"]) self.assertNotIn("allow", closed["permissions"]) +class DevinTransientTest(unittest.TestCase): + def test_empty_model_catalog_is_a_provider_failure_but_a_wrong_model_is_not(self): + self.assertTrue(devin.transient("Error: Unknown model: 'swe-2-max'\nAvailable:\n", "")) + self.assertFalse(devin.transient("Error: Unknown model: 'nope'\nAvailable:\n swe-2-max\n swe-2\n", "")) + self.assertTrue(devin.transient("", "429 rate limit")) + + def test_agent_text_on_stdout_does_not_classify_as_a_provider_failure(self): + self.assertFalse(devin.transient("I could not finish: my test run timed out and the quota for retries is used up", "")) + + +class AgyTransientTest(unittest.TestCase): + def test_a_dropped_connection_is_a_provider_failure_even_with_an_answer_written(self): + self.assertTrue(agy.transient('API error (attempt 1): request failed: Post "https://x/v1internal:streamGenerateContent": EOF')) + self.assertTrue(agy.transient("quota exceeded")) + self.assertFalse(agy.transient("the agent wrote an invalid file")) + + +class DevinEffortTest(unittest.TestCase): + def test_effort_is_read_from_the_model_id(self): + self.assertEqual(devin.split_effort("swe-2-max"), ("swe-2", "max")) + self.assertEqual(devin.split_effort("claude-opus-5-5-medium"), ("claude-opus-5-5", "medium")) + self.assertEqual(devin.split_effort("swe-2"), ("swe-2", None)) + + class NativeAdapterUsageTest(unittest.TestCase): def test_codex_inclusive_counts_are_split_without_counting_reasoning_twice(self): row = codex.usage_row({"input_tokens": 100, "cached_input_tokens": 60, "output_tokens": 20, "reasoning_output_tokens": 12}, 1.0, 272000) @@ -112,5 +147,105 @@ def test_codex_builtin_skills_are_separate_from_observed_project_skills(self): self.assertEqual(codex.loaded_skills(text), (["wright"], ["openai-docs"])) +class GrokAdapterTest(unittest.TestCase): + def test_usage_row_buckets_are_disjoint_and_carry_the_context_limit(self): + row = grok.usage_row({"input_tokens": 12908, "output_tokens": 17, "cache_read_input_tokens": 1536, "cache_creation_input_tokens": 0}, 256000, 1.0) + self.assertEqual((row["input"], row["cache_read"], row["output"], row["context"], row["context_limit"]), (12908, 1536, 17, 14444, 256000)) + + +class OpencodeAdapterTest(unittest.TestCase): + def test_usage_row_keeps_cache_apart_from_input(self): + row = opencode.usage_row({"total": 6679, "input": 1042, "output": 5, "reasoning": 0, "cache": {"write": 0, "read": 5632}}, 1.0) + self.assertEqual((row["input"], row["cache_read"], row["output"], row["context"]), (1042, 5632, 5, 6674)) + + def test_builtin_skills_are_not_loaded_context(self): + raw = json.dumps([{"name": "customize-opencode", "location": ""}, {"name": "wright", "location": "/w/.agents/skills/wright/SKILL.md"}]) + self.assertEqual(opencode.available_skills(raw), ["wright"]) + self.assertEqual(opencode.available_skills("not json"), []) + + +class DirectAdapterTest(unittest.TestCase): + def serve(self, replies): + seen = [] + + class Handler(BaseHTTPRequestHandler): + def do_POST(self): + seen.append(json.loads(self.rfile.read(int(self.headers["content-length"])))) + status, body = replies[min(len(seen), len(replies)) - 1] + self.send_response(status) + self.end_headers() + self.wfile.write(json.dumps(body).encode()) + + def log_message(self, *args): + pass + + server = HTTPServer(("127.0.0.1", 0), Handler) + threading.Thread(target=server.serve_forever, daemon=True).start() + self.addCleanup(server.shutdown) + return f"http://127.0.0.1:{server.server_port}", seen + + def run_direct(self, model, base_var, replies): + base, seen = self.serve(replies) + with tempfile.TemporaryDirectory(dir=Path(__file__).parent) as tmp: + tmp = Path(tmp) + skill = tmp / "demo" + skill.mkdir() + (skill / "SKILL.md").write_text("---\nname: demo\ndescription: A demo skill\n---\nbody\n") + env = {"BENCH_MODEL": model, "BENCH_KNOWLEDGE": "none", "BENCH_SKILL_DIRS": str(skill), "ANTHROPIC_API_KEY": "k", "OPENAI_API_KEY": "k", base_var: base, + **{f"BENCH_{n}": str(tmp / n.lower()) for n in ("USAGE", "TRANSCRIPT", "CONTEXT", "AGENT_INFO")}, "PATH": os.environ["PATH"]} + cwd = os.getcwd() + os.makedirs(tmp / "work") + os.chdir(tmp / "work") + out = io.StringIO() + try: + with patch.dict(direct.os.environ, env, clear=True), patch.object(direct.sys, "stdin", io.StringIO("task")), patch.object(direct.sys, "stdout", out): + code = direct.main() + finally: + os.chdir(cwd) + read = lambda n: (tmp / n).read_text() + return code, out.getvalue(), seen, [json.loads(l) for l in read("usage").splitlines()], json.loads(read("context")), json.loads(read("agent_info")) + + def test_anthropic_loop_runs_a_tool_and_records_usage(self): + use = {"content": [{"type": "tool_use", "id": "t1", "name": "bash", "input": {"command": "echo hi"}}], "usage": {"input_tokens": 10, "output_tokens": 2, "cache_read_input_tokens": 5}} + done = {"content": [{"type": "text", "text": "done"}], "usage": {"input_tokens": 20, "output_tokens": 3}} + code, final, seen, usage, context, info = self.run_direct("anthropic/m", "ANTHROPIC_BASE_URL", [(200, use), (200, done)]) + self.assertEqual((code, final, len(seen)), (0, "done", 2)) + self.assertEqual(seen[1]["messages"][-1]["content"][0]["content"].strip(), "hi") + self.assertIn("demo: A demo skill", seen[0]["system"]) + self.assertEqual((usage[0]["input"], usage[0]["cache_read"], usage[0]["context"]), (10, 5, 15)) + self.assertEqual((context["loaded"], info["agent"], [t["name"] for t in seen[0]["tools"]]), (["demo"], "direct", ["bash"])) + + def test_openai_usage_splits_cached_and_reasoning_tokens(self): + reply = {"choices": [{"message": {"role": "assistant", "content": "ok"}}], "usage": {"prompt_tokens": 100, "completion_tokens": 20, "prompt_tokens_details": {"cached_tokens": 60}, "completion_tokens_details": {"reasoning_tokens": 12}}} + code, final, _, usage, _, _ = self.run_direct("openai/m", "OPENAI_BASE_URL", [(200, reply)]) + self.assertEqual((code, final), (0, "ok")) + self.assertEqual((usage[0]["input"], usage[0]["cache_read"], usage[0]["output"], usage[0]["reasoning"]), (40, 60, 8, 12)) + + def test_client_error_is_not_an_infrastructure_failure(self): + code, *_ = self.run_direct("anthropic/m", "ANTHROPIC_BASE_URL", [(400, {"error": "bad"})]) + self.assertEqual(code, 1) + + def test_a_malformed_payload_is_a_clean_provider_error_not_a_traceback(self): + for model, base_var, replies in (("anthropic/m", "ANTHROPIC_BASE_URL", [(200, {"unexpected": "shape"})]), + ("openai/m", "OPENAI_BASE_URL", [(200, {"choices": []})])): + with self.subTest(model=model): + code, *_ = self.run_direct(model, base_var, replies) + self.assertEqual(code, 1) + + +class PipeTest(unittest.TestCase): + def test_drain_keeps_a_chatty_stderr_from_deadlocking_the_stdout_read(self): + # a child that floods stderr past the pipe buffer blocks unless someone drains it concurrently with stdout + import subprocess + import common + child = subprocess.Popen( + [sys.executable, "-c", "import sys; sys.stderr.write('x' * 262144); sys.stderr.flush(); print('ok')"], + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + stderr_text = common.drain(child.stderr) + self.assertEqual(child.stdout.read(), "ok\n") + child.wait() + self.assertEqual(stderr_text(), "x" * 262144) + + if __name__ == "__main__": unittest.main() diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 5bd0d966..62443ede 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -1,11 +1,14 @@ import argparse +import contextlib import hashlib import json import os import shutil +import stat import sys import tempfile import unittest +from unittest.mock import patch from pathlib import Path import agent_bench @@ -32,7 +35,7 @@ def setUp(self): def trial(self, agent_cmd: str, tool: str = "wright", skills: tuple = (), knowledge: str = "none", scenario: str = SCENARIO, **options) -> dict: args = argparse.Namespace(**{ "wright": str(Path(WRIGHT).resolve()), "agent_id": "fake", "agent_cmd": agent_cmd, "timeout": 60, "infra_retries": 2, - "env_pass": [], "canary_cmd": None, "skill_dirs": {}, "wiki_dir": None, "check_ancestors": False, **options, + "env_pass": [], "canary_cmd": None, "skill_dirs": {}, "wiki_dir": None, "check_ancestors": False, "out": self.out, **options, }) cell = agent_bench.normalize_cell({"tool": tool, "skills": list(skills), "knowledge": knowledge, "network": "off"}) return agent_bench.run_trial(agent_bench.load_scenario(scenario), cell, args, self.out / f"{scenario}-{tool}") @@ -158,6 +161,91 @@ def test_matrix_skips_inapplicable_cells_and_stops_after_repeated_interruptions( self.assertIn("not applicable: repair-runaway-loop", printed.getvalue()) self.assertEqual(len(list((self.out / "m").rglob("result.json"))), 2) # the third and fourth jobs were left unattempted self.assertIn("left unattempted", printed.getvalue()) + interrupted = {p for p in (self.out / "m").rglob("result.json") if json.loads(p.read_text())["status"] == "provider-interrupted"} + self.assertEqual(len(interrupted), 2) + config.write_text(config.read_text().replace('"cmd": "exit 75"', '"cmd": "exit 0"')) # the provider is back + with contextlib.redirect_stdout(io.StringIO()): + agent_bench.cmd_matrix(args) + self.assertTrue(all(json.loads(p.read_text())["status"] != "provider-interrupted" for p in (self.out / "m").rglob("result.json"))) + self.assertEqual(len(list((self.out / "m").rglob("result.json"))), 4) # the interrupted two were rerun and the rest ran + + def test_a_partial_result_from_a_killed_run_is_retried_instead_of_crashing(self): + import io + cell = agent_bench.normalize_cell({"tool": "none", "skills": [], "knowledge": "none", "network": "off"}) + config = self.out / "matrix.json" + config.write_text(json.dumps({ + "agents": [{"id": "fake", "cmd": "exit 0"}], + "cells": [cell], "scenarios": [SCENARIO], "trials": 1, "parallel": 1, "seed": 1, + "options": {"timeout": 30, "infra_retries": 0}, + })) + out = agent_bench.trial_dir(self.out / "m", SCENARIO, "fake", cell, 1) + out.mkdir(parents=True) + (out / "result.json").write_text('{"status": "com') # a kill mid-write left this + args = argparse.Namespace(config=config, out=self.out / "m", wright=str(Path(WRIGHT).resolve()), skill_dirs={}, wiki_dir=None, env_pass=[], + check_ancestors=False, canary_cmd=None, timeout=30, infra_retries=0, file_sandbox=False) + with contextlib.redirect_stdout(io.StringIO()): + code = agent_bench.cmd_matrix(args) + self.assertEqual(code, 0) + self.assertEqual(json.loads((out / "result.json").read_text())["status"], "completed") + + def test_evaluate_serializes_the_run_options_so_its_matrix_reproduces_the_run(self): + import contextlib + import io + script = agent_bench.HERE / "adapters" / "_test_fake_adapter.py" + script.write_text("import os, sys\nsys.stdin.read()\nopen('probe.txt', 'w').write(os.environ.get('BENCH_PROBE', '') + '|' + os.environ.get('HOME', ''))\n") + self.addCleanup(script.unlink, True) + self.skill_dir("wright-skill") # reached as a relative path below, resolved against the evaluate cwd + (self.out / "wiki-snap").mkdir() + args = argparse.Namespace( # relative paths must serialize resolved: the file resolves its own against its directory + adapter="fake", model="m", effort=None, name="eval", out=self.out, wright=str(Path(WRIGHT).resolve()), + skill_dirs={"wright-skill": Path("skills/wright-skill")}, wiki_dir=Path("wiki-snap"), env_pass=["BENCH_PROBE"], + allow_read=["allow-this"], deny_read=[], canary_cmd=None, check_ancestors=False, + timeout=30, infra_retries=3, infra_backoff=0, no_file_sandbox=True, dry_run=False, + cells="score", split="test", scenarios=[SCENARIO], trials=1, parallel=1, seed=1) + run_dir = self.out / "eval" + with patch.dict(agent_bench.ADAPTERS, {"fake": "_test_fake_adapter.py"}), patch.dict(agent_bench.ADAPTER_READS, {"fake": []}), \ + patch.dict(os.environ, {"BENCH_PROBE": "present"}), contextlib.chdir(self.out), contextlib.redirect_stdout(io.StringIO()): + code = agent_bench.cmd_evaluate(args) + self.assertEqual(code, 0) + options = json.loads((run_dir / "matrix.json").read_text())["options"] + self.assertEqual((options["out"], options["out_root"], options["adapter"]), (".", "..", "fake")) + self.assertEqual((options["wright"], options["timeout"], options["file_sandbox"]), (str(Path(WRIGHT).resolve()), 30, False)) + self.assertIn("BENCH_PROBE", options["env_pass"]) + self.assertIn("HOME", options["env_pass"]) + self.assertEqual(options["skill_dirs"]["wright-skill"], str((self.out / "skills/wright-skill").resolve())) + self.assertEqual(options["wiki_dir"], str((self.out / "wiki-snap").resolve())) + self.assertEqual(options["deny_read"], []) + self.assertEqual(options["infra_retries"], 3) + self.assertIn(str((self.out / "allow-this").resolve()), options["allow_read"]) + trial = run_dir / SCENARIO / "fake-m" / "wright+wright-skill_none_off-1" + self.assertEqual((trial / "workspace" / "probe.txt").read_text(), f"present|{Path.home()}") + elsewhere = self.out / "elsewhere" # flag values that would all be wrong; every serialized option must win + margs = argparse.Namespace(config=run_dir / "matrix.json", out=elsewhere, wright="/missing", skill_dirs={}, wiki_dir=None, + env_pass=[], check_ancestors=True, canary_cmd=None, timeout=99, infra_retries=5, infra_backoff=5, + file_sandbox=True, deny_read=[], allow_read=[]) + (trial / "result.json").unlink() + with patch.dict(os.environ, {"BENCH_PROBE": "present"}), contextlib.redirect_stdout(io.StringIO()): + code = agent_bench.cmd_matrix(margs) + self.assertEqual(code, 0) + self.assertFalse(elsewhere.exists()) # the run's own directory, not --out, receives the rerun trial + result = json.loads((trial / "result.json").read_text()) + self.assertEqual((result["protocol"]["timeoutSeconds"], result["fileWriteEnforcement"]), (30, "unrestricted")) + self.assertEqual((trial / "workspace" / "probe.txt").read_text(), f"present|{Path.home()}") # env_pass reached the agent again + + def test_evaluate_reports_a_missing_wright_binary_instead_of_crashing(self): + import io + self.skill_dir("wright-skill") + args = argparse.Namespace( + adapter="fake", model="m", effort=None, name="eval", out=self.out, wright=str(self.out / "no-such-wright"), + skill_dirs={"wright-skill": self.out / "skills" / "wright-skill"}, wiki_dir=None, env_pass=[], + allow_read=[], deny_read=[], canary_cmd=None, check_ancestors=False, + timeout=30, infra_retries=0, infra_backoff=0, no_file_sandbox=True, dry_run=False, + cells="score", split="test", scenarios=[SCENARIO], trials=1, parallel=1, seed=1) + with patch.dict(agent_bench.ADAPTERS, {"fake": "x.py"}), patch.dict(agent_bench.ADAPTER_READS, {"fake": []}), \ + contextlib.redirect_stdout(io.StringIO()), self.assertRaises(SystemExit) as stop: + agent_bench.cmd_evaluate(args) + self.assertIn("cannot start", str(stop.exception.code)) + self.assertIn("wright binary not found", str(stop.exception.code)) def test_scenarios_are_solvable_and_not_vacuous(self): self.assertTrue(agent_bench.validate(WRIGHT, self.out / "validate")) @@ -189,6 +277,190 @@ def test_host_instructions_are_unreadable_but_workspace_instructions_are_allowed self.assertEqual(host.read_text(), "host-only instructions") self.assertEqual((self.out / f"{SCENARIO}-wright/workspace/AGENTS.md").read_text(), "workspace instructions") + @unittest.skipUnless(sys.platform == "darwin" and shutil.which("sandbox-exec"), "macOS file sandbox") + def test_agent_cannot_read_answer_keys_other_runs_or_denied_paths(self): + import shlex + denied = self.out / "checkout" + denied.mkdir() + (denied / "secret.txt").write_text("sibling repository") + answer = agent_bench.SCENARIOS / SCENARIO / "reference" / "mode.ws" + probes = {"answer": answer, "denied": denied / "secret.txt"} + code = ("from pathlib import Path\nimport json\nout = {}\n" + + "".join(f"try:\n Path({str(p)!r}).read_text(); out[{k!r}] = 'read'\nexcept PermissionError:\n out[{k!r}] = 'blocked'\n" for k, p in probes.items()) + + "Path('probe.json').write_text(json.dumps(out))\nPath('own.txt').write_text('ok'); assert Path('own.txt').read_text() == 'ok'\n") + result = self.trial(f'{shlex.quote(sys.executable)} -c {shlex.quote(code)}', file_sandbox=True, deny_read=[str(denied)]) + workspace = self.out / f"{SCENARIO}-wright/workspace" + self.assertEqual(json.loads((workspace / "probe.json").read_text()), {"answer": "blocked", "denied": "blocked"}) + self.assertIn(str(denied.resolve()), result["fileReadEnforcement"]["hidden"]) + + def test_user_defaults_are_read_from_the_config_file(self): + config = self.out / "config.json" + config.write_text(json.dumps({"skill_dirs": {"wright-skill": "/s/wright"}, "deny_read": ["~/Repos"], "wright": "/bin/wright"})) + with patch.object(agent_bench, "CONFIG_PATH", config): + defaults = agent_bench.user_defaults() + self.assertEqual((defaults["skill_dir"], defaults["deny_read"], defaults["wright"]), (["wright-skill=/s/wright"], ["~/Repos"], "/bin/wright")) + config.write_text(json.dumps({"skil_dirs": {}})) + with patch.object(agent_bench, "CONFIG_PATH", config), self.assertRaises(SystemExit): + agent_bench.user_defaults() + + def test_preflight_names_what_is_missing_before_a_run_starts(self): + args = argparse.Namespace(wright=str(self.out / "nope"), adapter="devin", skill_dirs={}) + with patch.object(agent_bench.shutil, "which", return_value=None), patch.dict(agent_bench.CREDENTIALS, {"devin": [(".missing-bench-cred/auth.json", "x")]}): + problems = agent_bench.preflight(args, [agent_bench.normalize_cell({"tool": "wright", "skills": ["wright-skill"], "knowledge": "none", "network": "off"})]) + self.assertEqual(len(problems), 4) + self.assertTrue(any("wright binary not found" in p for p in problems) and any("`devin` is not on PATH" in p for p in problems) and any("--skill-dir wright-skill" in p for p in problems) and any("~/.missing-bench-cred/auth.json" in p for p in problems)) + + def test_preflight_names_a_missing_adapter_login_and_an_unusable_sandbox(self): + args = argparse.Namespace(wright=str(Path(WRIGHT).resolve()), adapter="grok", skill_dirs={}, file_sandbox=True, credentials=[]) + creds = {"grok": [(".missing-bench-cred/auth.json", "x"), (".missing-bench-cred/secondary", "y")]} + with patch.object(agent_bench.shutil, "which", side_effect=lambda b: f"/bin/{b}" if b != "sandbox-exec" else None), patch.dict(agent_bench.CREDENTIALS, creds, clear=True): + problems = agent_bench.preflight(args, [agent_bench.normalize_cell({"tool": "wright", "skills": [], "knowledge": "none", "network": "off"})]) + self.assertTrue(any("sandbox-exec" in p for p in problems)) + self.assertTrue(any("~/.missing-bench-cred/auth.json" in p for p in problems)) + self.assertTrue(any("~/.missing-bench-cred/secondary" in p for p in problems)) # every file the adapter needs is named + + def test_read_policy_hides_the_host_and_allows_only_what_the_run_needs(self): + root = self.out + skills = {"wright-skill": root / "s1", "opy-skill": root / "s2"} + args = argparse.Namespace(out=root / "run-a", out_root=root, wright=WRIGHT, skill_dirs=skills, wiki_dir=None, deny_read=[], allow_read=[str(root / "creds")], adapter="devin") + env = {"BENCH_RUN_DIR": str(root / "run-a" / "t"), "BENCH_SKILLS": "wright-skill", "BENCH_TOOL": "wright"} + hidden, allowed = agent_bench.read_policy(args, env) + self.assertTrue(all(p in hidden for p in (Path("/Users"), root.resolve(), agent_bench.HERE))) + self.assertTrue(all(d.resolve() in hidden for d in agent_bench.DATA_ROOTS)) + self.assertIn(skills["opy-skill"].resolve(), hidden) + self.assertNotIn(skills["wright-skill"].resolve(), hidden) + for needed in (root / "run-a" / "t", skills["wright-skill"], root / "creds", agent_bench.HERE / "adapters", agent_bench.HERE / "bench_trace.py", Path(WRIGHT).resolve().parent): + self.assertIn(needed.resolve(), allowed) + self.assertNotIn(agent_bench.HERE / "bench_grade.py", allowed) + self.assertNotIn((agent_bench.HERE / "oracle").resolve(), allowed) # the grading authority is readable only where the tool needs it + _, allowed = agent_bench.read_policy(args, {**env, "BENCH_TOOL": "overpy"}) + self.assertIn((agent_bench.HERE / "oracle").resolve(), allowed) + + @unittest.skipUnless(sys.platform == "darwin" and shutil.which("sandbox-exec"), "macOS file sandbox") + def test_the_agent_sees_only_its_run_the_allowed_paths_and_the_shim(self): + import shlex + other = self.out / "elsewhere" + other.mkdir() + (other / "note.txt").write_text("a file the agent was not given") + granted = self.out / "granted" + granted.mkdir() + (granted / "note.txt").write_text("a file the adapter needs") + probes = {"unlisted": other / "note.txt", "granted": granted / "note.txt", "grader": agent_bench.HERE / "bench_grade.py", "shim": agent_bench.HERE / "bench_trace.py", + "answer": agent_bench.SCENARIOS / SCENARIO / "reference" / "mode.ws", "oracle": agent_bench.HERE / "oracle" / "compile.js", + "profile": self.out / f"{SCENARIO}-wright" / "agent.sb"} # the sandbox profile itself, which lists the hidden paths + code = ("import json\nfrom pathlib import Path\nout = {}\n" + + "".join(f"try:\n Path({str(p)!r}).read_text(); out[{k!r}] = 'read'\nexcept PermissionError:\n out[{k!r}] = 'blocked'\n" for k, p in probes.items()) + # a read deny on the profile's own path is not enough: inside the writable run dir the agent can move or + # link the file to a name the deny does not cover, so the probes must try the rename and link themselves + + f"try:\n moved = Path('agent-moved.sb')\n Path({str(probes['profile'])!r}).rename(moved)\n moved.read_text(); out['profile-renamed'] = 'read'\nexcept PermissionError:\n out['profile-renamed'] = 'blocked'\n" + + f"try:\n linked = Path('agent-linked.sb')\n linked.hardlink_to({str(probes['profile'])!r})\n linked.read_text(); out['profile-linked'] = 'read'\nexcept PermissionError:\n out['profile-linked'] = 'blocked'\n" + + "Path('probe.json').write_text(json.dumps(out))\n") + result = self.trial(f'{shlex.quote(sys.executable)} -c {shlex.quote(code)}', file_sandbox=True, allow_read=[str(granted)]) + seen = json.loads((self.out / f"{SCENARIO}-wright/workspace/probe.json").read_text()) + self.assertEqual(seen, {"unlisted": "blocked", "granted": "read", "grader": "blocked", "shim": "read", "answer": "blocked", "oracle": "blocked", + "profile": "blocked", "profile-renamed": "blocked", "profile-linked": "blocked"}) # wright cells must not read the grading authority, and no cell reads its own sandbox profile + self.assertEqual(result["fileReadEnforcement"]["mode"], "allow-list") + + @unittest.skipUnless(sys.platform == "darwin", "file flags are the macOS enforcement") + def test_a_locked_profile_left_by_a_killed_run_does_not_block_the_next_trial(self): + out = self.out / f"{SCENARIO}-wright" + locked = {"profile": out / "agent.sb", "agent-file": out / "workspace" / "agent-locked.txt"} + for name, path in locked.items(): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(f"left by a run killed before cleanup ({name})") + os.chflags(path, stat.UF_IMMUTABLE) # the harness locks agent.sb; the agent can lock anything inside its run dir + try: + result = self.trial("true") + self.assertNotIn("invalid", result) + finally: + for path in locked.values(): # if drop left them, free the test's own cleanup + with contextlib.suppress(OSError): + os.chflags(path, 0) + + def test_the_tool_shim_runs_without_the_grader(self): + shim = (agent_bench.HERE / "bench_trace.py").read_text() + self.assertNotIn("import bench_grade", shim) + self.assertIn("shim_main(sys.argv[2:])", shim) + + def test_a_refreshed_login_is_written_back_to_the_real_one_only_when_newer(self): + home, run = self.out / "home", self.out / "run" + (home / ".grok").mkdir(parents=True) + run.mkdir() + real, isolated = home / ".grok/auth.json", run / "grok-home/auth.json" + real.write_text("old-token") + os.chmod(real, 0o600) + isolated.parent.mkdir() + isolated.write_text("old-token") + pairs = [(".grok/auth.json", "grok-home/auth.json")] + self.assertEqual(agent_bench.sync_credentials_back(pairs, run, home), []) # unchanged + isolated.write_text("refreshed-token") + os.utime(isolated, (real.stat().st_mtime + 10, real.stat().st_mtime + 10)) + self.assertEqual(agent_bench.sync_credentials_back(pairs, run, home), [".grok/auth.json"]) + self.assertEqual((real.read_text(), oct(real.stat().st_mode & 0o777)), ("refreshed-token", "0o600")) + real.write_text("newer-real-token") # the user logged in again meanwhile: never overwrite a newer real login + os.utime(isolated, (real.stat().st_mtime - 10, real.stat().st_mtime - 10)) + self.assertEqual(agent_bench.sync_credentials_back(pairs, run, home), []) + self.assertEqual(real.read_text(), "newer-real-token") + + def test_suite_runs_each_model_in_turn_continues_past_a_wait_or_a_skip_and_writes_the_page(self): + models = [{"adapter": "devin", "model": "swe-2-max"}, {"adapter": "codex", "model": "gpt-6-luna", "effort": "xhigh"}, {"adapter": "pi", "model": "m"}] + args = argparse.Namespace(models=models, only=None, out=self.out, suite_name="s", dry_run=False) + calls = [] + + def evaluate(sub): + calls.append((sub.adapter, sub.name, sub.effort, sub.out)) + if sub.adapter == "codex": + return 3 + if sub.adapter == "pi": + raise SystemExit("cannot start: pi missing") + return 0 + + with patch.object(agent_bench, "cmd_evaluate", evaluate), patch.object(agent_bench.bench_leaderboard, "main") as page: + status = agent_bench.cmd_suite(args) + self.assertEqual(status, 3) # codex is waiting for a rerun + self.assertEqual([c[1] for c in calls], ["devin-swe-2-max", "codex-gpt-6-luna-xhigh", "pi-m"]) + self.assertTrue(all(c[3] == self.out / "s" for c in calls)) + page.assert_called_once() + with patch.object(agent_bench, "cmd_evaluate", evaluate): + only = agent_bench.cmd_suite(argparse.Namespace(models=models, only=["devin"], out=self.out, suite_name="s2", dry_run=True)) + self.assertEqual(only, 0) + calls.clear() # --only matches the adapter alone or adapter:model exactly — a prefix must not pick another model + with patch.object(agent_bench, "cmd_evaluate", evaluate): + only = agent_bench.cmd_suite(argparse.Namespace(models=models, only=["codex:gpt-6-luna"], out=self.out, suite_name="s4", dry_run=True)) + self.assertEqual(only, 3) # only codex ran, and it is waiting for a rerun + self.assertEqual([c[1] for c in calls], ["codex-gpt-6-luna-xhigh"]) + with self.assertRaisesRegex(SystemExit, "no models"): + agent_bench.cmd_suite(argparse.Namespace(models=models, only=["codex:gpt-6"], out=self.out, suite_name="s5", dry_run=True)) + with patch.object(agent_bench, "cmd_evaluate", return_value=1): + failed = agent_bench.cmd_suite(argparse.Namespace(models=[{"adapter": "devin", "model": "m"}], only=None, out=self.out, suite_name="s3", dry_run=True)) + self.assertEqual(failed, 1) # a model that finished with errors fails the suite, it does not pass silently + + models = [{"adapter": "devin", "model": "m"}] + for bad in ("..", "a/b", ""): + with self.assertRaises(SystemExit): + agent_bench.cmd_suite(argparse.Namespace(models=models, only=None, out=self.out, suite_name=bad, dry_run=True)) + + def test_a_repeated_run_refuses_a_different_wright_binary(self): + run = self.out / "r" + trial = run / "s" / "agent" / "cell-1" + trial.mkdir(parents=True) + (trial / "result.json").write_text(json.dumps({"environment": {"wright": "wright 0.7.0", "wrightSha256": "a" * 64}})) + other = self.out / "wright-other" + other.write_text("a different binary") + message = agent_bench.wright_mismatch(run, str(other)) + self.assertIn("wright 0.7.0", message) + self.assertIn("--wright", message) + self.assertIsNone(agent_bench.wright_mismatch(self.out / "fresh", str(other))) # a new run has nothing to disagree with + (trial / "result.json").write_text(json.dumps({"environment": {"wright": "x", "wrightSha256": agent_bench.file_sha256(other)}})) + self.assertIsNone(agent_bench.wright_mismatch(run, str(other))) + self.assertIsNone(agent_bench.wright_mismatch(run, str(self.out / "no-such-binary"))) # preflight names the missing binary; hashing it must not crash first + (trial / "result.json").write_text('{"environment": {"wrightSha') # a mid-write kill left a partial file: the trial is unfinished, not evidence + self.assertIsNone(agent_bench.wright_mismatch(run, str(other))) + + def test_an_effort_the_model_id_already_names_is_not_repeated_in_the_run_name(self): + self.assertEqual(agent_bench.model_slug({"adapter": "agy", "model": "gemini-3.8-flash-high", "effort": "high"}), "agy-gemini-3.8-flash-high") + self.assertEqual(agent_bench.model_slug({"adapter": "codex", "model": "gpt-6-luna", "effort": "xhigh"}), "codex-gpt-6-luna-xhigh") + def test_tools_differ_only_in_availability(self): agent = f"cp {reference()}/* . && (wright check mode.ws >/dev/null 2>&1 || echo no-wright > missing-wright.txt)" none = self.trial(agent, tool="none") @@ -271,6 +543,12 @@ def test_adapter_usage_reaches_result(self): self.assertEqual(result["usage"]["totalTokens"], 6) self.assertIn("unexpected loaded context", result["invalid"]) + def test_a_retry_starts_without_the_previous_attempts_adapter_home(self): + agent = ('if [ -f "$BENCH_RUN_DIR/tried" ]; then test ! -e "$BENCH_RUN_DIR/devin-home" || exit 1; exit 0; ' + 'else touch "$BENCH_RUN_DIR/tried"; mkdir -p "$BENCH_RUN_DIR/devin-home/.local"; exit 75; fi') + result = self.trial(agent) + self.assertEqual((result["agent"]["exit"], result["infraRetries"]), (0, 1)) + def test_infrastructure_failures_are_retried(self): agent = 'if [ -f "$BENCH_RUN_DIR/tried" ]; then exit 0; else touch "$BENCH_RUN_DIR/tried"; exit 75; fi' self.assertEqual(self.trial(agent)["infraRetries"], 1) @@ -307,6 +585,16 @@ def test_oracle_check_never_passes_silently(self): self.assertFalse(graded["checks"][0]["passed"]) # a raw Workshop entry has no upstream OverPy verdict self.assertRegex(graded["grader"]["hash"], r"^[0-9a-f]{64}$") + @unittest.skipUnless(bench_grade.oracle_available(), "run `agent_bench.py setup-oracle`") + def test_missing_entry_is_an_agent_failure_not_an_unavailable_grader(self): + workspace = self.out / "empty" + workspace.mkdir() + scenario = agent_bench.load_scenario("widow-headshots") + graded = bench_grade.grade(scenario, workspace, WRIGHT) + self.assertEqual(graded["authorities"]["oracle"]["status"], "error") + self.assertNotIn("grader-unavailable", graded["usableReason"]) + self.assertIn("checks-failed", graded["usableReason"]) + @unittest.skipUnless(bench_grade.oracle_available(), "run `agent_bench.py setup-oracle`") def test_oracle_disagreement_is_reported(self): source = self.out / "n.opy" @@ -318,6 +606,47 @@ def test_oracle_disagreement_is_reported(self): self.assertEqual(auth["disagreement"]["kind"], "wright-accepts-oracle-rejects") +@unittest.skipUnless(sys.platform == "darwin", "file flags are the macOS enforcement") +class DropTest(unittest.TestCase): + def setUp(self): + (agent_bench.ROOT / "target").mkdir(exist_ok=True) + self.out = Path(tempfile.mkdtemp(dir=agent_bench.ROOT / "target")).resolve() + self.addCleanup(shutil.rmtree, self.out, True) + + def test_repairs_never_reach_through_a_symlink_to_a_host_file(self): + host_file = self.out / "host-file.txt" + host_file.write_text("host data the run must not mutate") + host_file.chmod(0o400) + os.chflags(host_file, stat.UF_IMMUTABLE) + link = self.out / "run" / "stale-link" + link.parent.mkdir(parents=True) + link.symlink_to(host_file) + try: + agent_bench.drop(link) + self.assertFalse(os.path.lexists(link)) + self.assertTrue(os.lstat(host_file).st_flags & stat.UF_IMMUTABLE, "the link redirected the flag repair onto the host file") + self.assertEqual(host_file.stat().st_mode & 0o777, 0o400) + finally: + with contextlib.suppress(OSError): + os.chflags(host_file, 0) + host_file.chmod(0o600) + + def test_nested_chmodded_directories_come_down_a_level_per_pass(self): + root = self.out / "run" / "denied" + inner = root / "inner" + inner.mkdir(parents=True) + (inner / "left.txt").write_text("agent-owned") + for directory in (inner, root): + directory.chmod(0) # the agent can make its own directories untraversable — deepest first, the parent must stay resolvable + try: + agent_bench.drop(self.out / "run") + self.assertFalse(os.path.lexists(self.out / "run")) + finally: + for directory in (root, inner): + with contextlib.suppress(OSError): + directory.chmod(0o700) + + class DetectorTest(unittest.TestCase): def call(self, argv, exit_code=0, t=0.0, **extra): return {"tool": "wright", "type": "call", "t": t, "argv": argv, "exit": exit_code, "seconds": 0.1, "stdoutBytes": 40, "stderrBytes": 0, "stderrHead": "", "envelope": None, **extra} @@ -395,6 +724,15 @@ def test_headroom_and_invalid_runs_are_reported(self): self.assertIn("HEADROOM", text) self.assertIn("INVALID: 1 run(s) excluded", text) + def test_report_shows_what_the_agent_was_actually_given(self): + run = self.result("wright+wright-skill/none/off", 1, True, 100) + run["agentInfo"] = {"version": "tool 1.2", "model": "m-1", "effort": "high", "tools": ["bash", "read"]} + run["context"] = {"loaded": ["wright"]} + bare = self.result("none/none/off", 1, True, 100) + text, _ = bench_report.render([run, bare]) + self.assertIn("| m | wright+wright-skill/none/off | tool 1.2 | m-1 | high | 2: bash, read | wright |", text) + self.assertIn("| m | none/none/off | not recorded | not recorded | not recorded | not recorded | none |", text) + def test_provider_failures_do_not_count_as_agent_failures(self): good = self.result("none/none/off", 1, True, 100) provider_failure = self.result("none/none/off", 2, False, 0) diff --git a/benchmarks/agent/test_leaderboard.py b/benchmarks/agent/test_leaderboard.py new file mode 100644 index 00000000..f80ccbe8 --- /dev/null +++ b/benchmarks/agent/test_leaderboard.py @@ -0,0 +1,69 @@ +import json +import shutil +import tempfile +import unittest +from pathlib import Path + +import bench_leaderboard +import bench_score +from test_score import SCENARIOS, runs + +ROOT = Path(__file__).resolve().parents[2] + + +class LeaderboardTest(unittest.TestCase): + def setUp(self): + self.root = Path(tempfile.mkdtemp(dir=ROOT / "target")) + self.addCleanup(shutil.rmtree, self.root, True) + + def write(self, name, usable, sha="a" * 64, model="m", tracks=("workshop", "opy")): + directory = self.root / name + directory.mkdir() + cards = [bench_score.card([{**r, "language": lang} for r in runs(SCENARIOS[:usable], sha=sha, model=model)], lang, SCENARIOS) for lang in tracks] + (directory / "score.json").write_text(json.dumps({"contract": bench_score.CONTRACT, "cards": cards})) + return directory + + def test_entries_are_ranked_and_tied_with_the_top_when_intervals_overlap(self): + data = bench_leaderboard.build([self.write("low", 1), self.write("high", 8), self.write("close", 7)]) + names = [e["run"] for e in data["entries"]] + self.assertEqual(names[0], "high") + standing = {e["run"]: e["standing"] for e in data["entries"]} + self.assertEqual((standing["high"], standing["close"], standing["low"]), ("top", "tied with top", "below top")) + + def test_runs_against_a_different_wright_are_listed_as_not_comparable(self): + data = bench_leaderboard.build([self.write("a", 6), self.write("b", 5), self.write("other", 6, sha="d" * 64)]) + self.assertEqual({e["run"] for e in data["entries"]}, {"a", "b"}) + self.assertEqual([o["run"] for o in data["excluded"]], ["other"]) + self.assertIn("Not comparable", bench_leaderboard.markdown(data)) + + def test_a_run_covering_fewer_tracks_is_not_ranked(self): + data = bench_leaderboard.build([self.write("full", 5), self.write("half", 8, tracks=("workshop",))]) + self.assertEqual([e["run"] for e in data["entries"]], ["full"]) + self.assertEqual([o["run"] for o in data["excluded"]], ["half"]) + self.assertIn("1 of 2 language tracks", bench_leaderboard.markdown(data)) + + def test_markdown_and_page_give_the_headline_a_reader_needs(self): + data = bench_leaderboard.build([self.write("only", 6)]) + text, page = bench_leaderboard.markdown(data), bench_leaderboard.page(data) + for needed in ("Wright Agent Score", "Workshop score", "OverPy score", "top", "How to read this", "Limits"): + self.assertIn(needed, text) + self.assertIn("█", text) + self.assertIn('class="bar"', page) + self.assertIn("width:75.0%", page) # 6 of 8 scenarios + self.assertNotIn(" str: - return hashlib.sha256("".join(f"{p.relative_to(skill_dir)}{hashlib.sha256(p.read_bytes()).hexdigest()}" for p in sorted(skill_dir.rglob("*.md"))).encode()).hexdigest() + return hashlib.sha256("".join(f"{p.relative_to(skill_dir)}{hashlib.sha256(p.read_bytes()).hexdigest()}" for p in sorted(skill_dir.rglob("*")) + if p.is_file() and p.name != "BUILD.json").encode()).hexdigest() # the build record is written after the hash is taken def identity(skill_dir: Path) -> dict: diff --git a/docs/README.md b/docs/README.md index 24446a57..3e52e09e 100644 --- a/docs/README.md +++ b/docs/README.md @@ -41,6 +41,8 @@ Issue contract. transport mappings for coding agents and embedding consumers. - [Agent benchmark](agent-benchmark.md): product-level benchmark contract for general coding agents working with Wright. +- [Agent benchmark how-to](agent-benchmark-howto.md): run the benchmark and + publish the results page. - [Agent benchmark comparison spec](specs/SPEC-414-agent-benchmark-comparison.md): proposed multi-condition, multi-model benchmark and tool-call analysis. - [Language services & LSP](language-services.md): editor-neutral language diff --git a/docs/agent-benchmark-howto.md b/docs/agent-benchmark-howto.md new file mode 100644 index 00000000..52c10dad --- /dev/null +++ b/docs/agent-benchmark-howto.md @@ -0,0 +1,105 @@ +# Run the agent benchmark + +Run it from a shell, one command for every model, and get a results page you can publish. Background and the full contract are in [agent-benchmark.md](agent-benchmark.md). + +## Set up once + +1. Install `wright` on `PATH`, and the agent programs you want to test, each logged in. +2. Install the OverPy reference compiler the grader uses: + ```sh + python3 benchmarks/agent/agent_bench.py setup-oracle + ``` +3. Write `~/.config/wright-agent-bench/config.json` (the `WRIGHT_BENCH_CONFIG` variable overrides the path): + ```json + { + "skill_dirs": { + "wright-skill": "~/skills/skills/wright", + "opy-skill": "~/skills/skills/overpy" + }, + "models": [ + {"adapter": "devin", "model": "swe-2-max"}, + {"adapter": "codex", "model": "gpt-6-luna", "effort": "xhigh"}, + {"adapter": "pi", "model": "openai-codex/gpt-6-luna", "effort": "xhigh"}, + {"adapter": "grok", "model": "grok-4.7"} + ] + } + ``` + `skill_dirs` points at the `wrightkit/skills` checkout. Keep the Wright binary, the skills, and the scenarios unchanged for as long as you want results to be comparable. + +## Where things live + +Everything is under one directory, `~/.local/share/wright-agent-bench` (the `WRIGHT_BENCH_HOME` variable moves it): + +| Path | Holds | +| --- | --- | +| `runs/` | every evaluation run and the results page (`runs/results/leaderboard/`) | +| `wiki/` | local wiki snapshots (not published) | +| `skills-*/` | pinned copies of the skills under test | + +Runs made by earlier versions are in `~/.cache/wright-agent-bench`; nothing writes there any more. + +## Run everything + +```sh +python3 benchmarks/agent/agent_bench.py suite --dry-run # checks the setup, runs nothing +python3 benchmarks/agent/agent_bench.py suite +``` + +It evaluates the models one after another, each on 16 tasks tried 3 times, one trial at a time so provider limits are not hit. It is safe to stop and repeat: finished trials are skipped, and a model that hits a quota or an outage waits for the next run (the command ends with exit code 3 and says which). Run it again later, on another day if needed, and the same command carries on. + +The results are in `~/.local/share/wright-agent-bench/runs/results/leaderboard/`: + +| File | For | +| --- | --- | +| `LEADERBOARD.md` | pasting into a README or an issue | +| `leaderboard.html` | one self-contained page to host or share | +| `leaderboard.json` | other tools | + +Use `--only devin codex:gpt-6-luna` to run some entries only. + +## Run one model + +```sh +python3 benchmarks/agent/agent_bench.py evaluate --adapter codex --model gpt-6-luna --effort xhigh --name codex-luna-xhigh +``` + +Put several runs side by side, or rebuild the page from chosen runs: + +```sh +python3 benchmarks/agent/agent_bench.py compare ~/.local/share/wright-agent-bench/runs/{run-a,run-b} +python3 benchmarks/agent/agent_bench.py leaderboard ~/.local/share/wright-agent-bench/runs/{run-a,run-b} +``` + +Runs made against a different Wright binary, skills, or task suite are listed as not comparable instead of being ranked. + +## The agent programs + +| Program | `adapter` | `model` | `effort` | State | +| --- | --- | --- | --- | --- | +| Devin CLI | `devin` | `swe-2-max` (the effort is part of the name) | read from the name | works | +| Codex CLI | `codex` | `gpt-6-luna` | `low` to `xhigh` | works | +| pi | `pi` | `openai-codex/gpt-6-luna` (`pi --list-models`) | `low` to `xhigh` | works | +| Grok CLI | `grok` | `grok-4.7` (`grok models`) | optional | works | +| opencode | `opencode` | `openai/gpt-6-luna` (`opencode models`) | optional | works once its login is valid | +| Antigravity CLI | `agy` | `gemini-3.8-flash-high` | optional | not yet checked with the file sandbox | +| Claude Code | `claude-code` | `sonnet` | none | not yet checked with the file sandbox | +| Built-in loop, no program | `direct` | `anthropic/` or `openai/` | `openai/` only | needs an API key in the environment | + +The agent program runs the model, so the same model scores differently under different programs. The page shows both. + +## What the page tells you + +- The score is the share of tasks an agent finished with a valid, safe result. A bar shows it; the bracketed range is the 95% interval. +- With 8 tasks per language the range is wide. A row marked "tied with top" cannot be told apart from the first. +- It is a reference for how an agent behaves with Wright and its guide, not a measure of general ability. Network access is off by instruction only, not blocked. + +## When something goes wrong + +| What you see | Meaning | +| --- | --- | +| Exit code 3, "waiting" | quota or outage; run the same command later | +| `cannot start: ...` | `--dry-run` names what is missing (binary, login, skill directory, oracle) | +| An agent exits at once with an auth error | its login expired; sign in again with that program | +| An agent cannot start under the file sandbox | add the paths it needs: `--allow-read PATH` or `allow_read` in the config | + +The agent only sees its own run directory, the condition's skills, the tool binaries, and what its program needs to start; the home directories, drives, and benchmark data are hidden from it, so it cannot find the answers or other runs. diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index 65989ea6..ac4c410b 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -2,6 +2,7 @@ - Contracts: `wright-agent-bench/v3` (a run result) and `wright-agent-score/v1` (a score card) - Harness: [`benchmarks/agent/agent_bench.py`](../benchmarks/agent/agent_bench.py) +- Run it from a shell: [agent-benchmark-howto.md](agent-benchmark-howto.md) - Design and requirements: [`SPEC-414`](specs/SPEC-414-agent-benchmark-comparison.md) The benchmark answers one product question: can a general coding agent, with no @@ -10,6 +11,12 @@ complete realistic Workshop work correctly? It measures Wright's discoverable semantic surface; it is not a model leaderboard, and one stochastic run is not evidence of correctness (use `--trials`). +A score is a credible reference, not a measure of an agent's ability. It holds for one +Wright, skill set, suite, agent, model, and protocol, and it is meant to show what +the prompt, skills, and tools did in that setup. Every report therefore lists, from +what the adapter observed, the CLI version, model, tools, and loaded skills each +agent actually had (`Agent setup`). + ## Allowed agent context - The scenario workspace: the seed project only. @@ -120,6 +127,9 @@ keep host configuration out of the run, and it reports what loaded through | [`devin.py`](../benchmarks/agent/adapters/devin.py) | Devin CLI | `BENCH_MODEL` (for example `swe-2-max`). Runs with an isolated `HOME` holding only the Devin credentials, a config that reads no other tool's rules or skills, and MCP tools denied. Managed plugin skills are listed apart from `loaded`. | | [`codex.py`](../benchmarks/agent/adapters/codex.py) | Codex CLI | `BENCH_MODEL` and `BENCH_THINKING`. Isolated `HOME`/`CODEX_HOME`, user config and exec rules ignored, workspace skills installed under `.agents/skills`. Session token events supply per-model-call usage and observed skill context; built-ins are listed apart. Web search, Apps, and plugin discovery are disabled; any observed MCP call invalidates the trial. | | [`agy.py`](../benchmarks/agent/adapters/agy.py) | Antigravity CLI | `BENCH_MODEL` (for example `gemini-3.8-flash-high`) and `BENCH_THINKING`. Isolated `HOME` with only authentication files, workspace skills under `.agents/skills`, MCP/browser access denied and URL reads denied outside `web`. Observed built-in web tools under network `off` stop and invalidate the trial; URL permissions do not cover search. Streaming step usage is recorded. The CLI does not export observed loaded skills or context limits; these remain unreported. | +| [`opencode.py`](../benchmarks/agent/adapters/opencode.py) | opencode | `BENCH_MODEL=provider/model` (as `opencode models` lists it) and `BENCH_THINKING` as the model variant. Isolated `HOME` and XDG directories holding only the credentials; `--pure`; Claude Code instructions and the real home's external skills are disabled by environment variable because opencode reads them regardless of `HOME`; skills installed under `.opencode/skills`; web tools denied outside `web`. Per-step usage comes from `step_finish` events; the tool list is what the agent used. | +| [`grok.py`](../benchmarks/agent/adapters/grok.py) | Grok CLI | `BENCH_MODEL` (a `grok models` id) and `BENCH_THINKING` as reasoning effort. Isolated `GROK_HOME` holding only the login and the skills; prompt sent `--verbatim`; subagents disabled; web search disabled outside `web`. Tools, skills, and the context window come from the stream's init and result lines. | +| [`direct.py`](../benchmarks/agent/adapters/direct.py) | none (built-in loop) | See below. | The pi adapter also uses an isolated `HOME` containing only its authentication files. Explicit provider extensions remain referenced by path, not copied with @@ -135,7 +145,7 @@ Pass `--env-pass HOME` when the agent authenticates from the real home directory Two checks protect the context. The workspace must not sit below a directory that holds instruction files (`AGENTS.md`, `CLAUDE.md`, and similar), because agents discover them by walking up; the default `--out` is -`~/.cache/wright-agent-bench` for that reason, and a violation marks the run +`~/.local/share/wright-agent-bench/runs` for that reason, and a violation marks the run `invalid` (`--no-ancestor-check` disables it). Network `off` is enforced only when `--canary-cmd` is given and fails inside the agent environment; without it the result records `networkEnforcement: declared-only`, which is what the shell @@ -146,8 +156,15 @@ processes: filesystem writes are restricted to that trial's output directory (plus device streams), and `TMPDIR` points inside it. Unsupported hosts fail instead of silently running without protection. This protects host files; it blocks reads of `AGENTS.md`, `CLAUDE.md`, and `GEMINI.md` outside the trial workspace -to prevent tool-path rule discovery from contaminating context. Other host reads -remain possible, and network isolation is not enforced. Model +to prevent tool-path rule discovery from contaminating context. It also hides from +the agent the scenarios (their reference solutions), every other run under `--out`, +the wiki snapshot, skills outside the condition, and any `--deny-read PATH`, such as +the checkouts of the repositories under test. The adapter, harness code, the `wright` +binary directory, and the condition's skills stay readable. The result lists the +denied paths in `fileReadEnforcement`. Without this, agents find the answer keys and the +owner repositories on the host (seen in practice), so `evaluate` turns the sandbox on by +default (`--no-file-sandbox` disables it). Other host reads remain possible, and +network isolation is not enforced. Model account usage, CPU and disk consumption remain shared with the host. Provider failures returned as exit 75 are listed separately and excluded from outcome metrics. The harness never edits the task prompt: network `off` and the workspace @@ -217,9 +234,14 @@ python3 benchmarks/agent/agent_bench.py score target/agent-bench # Wright Agen ``` [`matrix.example.json`](../benchmarks/agent/matrix.example.json) is the Tier 1 matrix. `matrix.json` lists `agents` (`{id, cmd}`), `cells`, optional `scenarios`, -`trials`, `parallel`, `seed` (run order is shuffled by it), and `options` -(`skill_dirs` as `{name: dir}`, `wiki_dir`, `env_pass`, ...). Cells not applicable to a scenario's language are skipped; the matrix stops after two consecutive provider interruptions and exits 3 when any occurred. Finished runs are skipped, so an -interrupted matrix resumes. +`trials`, `parallel`, `seed` (run order is shuffled by it), and `options`. Options +are the trial-time settings a run needs, overriding their command-line counterparts: `out` and `out_root` (relative paths resolve against the matrix +file's directory, so `evaluate`'s `out: "."` makes the file's own directory the run directory), `wright`, `adapter`, `file_sandbox`, `env_pass`, `credentials`, +`allow_read`/`deny_read`, `timeout`, `canary_cmd`, `check_ancestors`, `infra_retries`/`infra_backoff`, `skill_dirs` as `{name: dir}`, `wiki_dir`. Every +path-valued option (`out`, `out_root`, `wright`, `skill_dirs`, `wiki_dir`, `allow_read`, `deny_read`) follows the same rule: relative resolves against the matrix file's +directory, and `evaluate` writes its own path options already resolved so the file reproduces the run from any cwd. Cells not +applicable to a scenario's language are skipped; the matrix stops after two consecutive provider interruptions and exits 3 when any occurred. Finished runs +are skipped, so an interrupted matrix resumes; `evaluate` writes its effective options into `matrix.json`, so `matrix /matrix.json` resumes that run in place. For a local offline-declared pilot, use [`matrix.pilot.example.json`](../benchmarks/agent/matrix.pilot.example.json): one @@ -254,7 +276,7 @@ with the workspace, `agent.log`, snapshots, and the Wright trace beside it. | `friction`, `expectations` | Usage errors, unknown subcommands, help lookups, retries, malformed `serve` requests, unparsed `serve` responses, identical repeats; expectation E01-E12 verdicts | | `snapshots` | Strict validity of each snapshot of the entry, first valid index, and valid-to-invalid regressions | | `usage`, `context` | Turns, tokens by kind, peak context (and its share of the limit), tokens to first valid; loaded context | -| `invalid`, `infraRetries`, `networkEnforcement` | Present when the run was excluded or retried; whether network `off` was checked by a canary or only declared | +| `invalid`, `infraRetries`, `fileReadEnforcement`, `fileWriteEnforcement`, `networkEnforcement` | Present when the run was excluded or retried; how file reads (`allow-list` with the hidden and allowed paths, or `unrestricted`), file writes (`trial-directory-only` or `unrestricted`), and network `off` (`canary-checked` or `declared-only`) were enforced | `toolUse` is recorded per CLI invocation through the shim; `wright serve` sessions are teed line by line into the trace. Comparing cells for the same scenario shows what @@ -286,9 +308,72 @@ over scenarios, and gives a two-stage bootstrap 95% interval (10,000 draws, seed 467) and Pass^k as a secondary figure. Provider-interrupted, invalid, and agent-error runs are published as exclusions; timeouts count. The score is refused if the runs differ in Wright binary, skill hashes, suite hash, agent, -model, effort, or protocol, and it is marked provisional with fewer than eight +model, effort, protocol, or file-read/file-write/network enforcement — a run +where the agent could read answer keys does not score beside a sandboxed one — +and it is marked provisional with fewer than eight held-out scenarios, missing scenarios, or unequal trials. The card discloses -`networkEnforcement` (`declared-only` or `canary-checked`). +`fileReadEnforcement`, `fileWriteEnforcement`, and `networkEnforcement` modes. + +## Evaluating without an agent harness + +Quick start. Put your paths in `~/.config/wright-agent-bench/config.json` once +(`WRIGHT_BENCH_CONFIG` overrides the location): + +```json +{"skill_dirs": {"wright-skill": "~/skills/skills/wright", "opy-skill": "~/skills/skills/overpy"}, + "deny_read": ["~/Repos", "~/.agents", "~/.claude"], + "wiki_dir": "~/.local/share/wright-agent-bench/wiki"} +``` + +then run one command per agent. `--dry-run` checks the setup (wright binary, the +agent CLI, credentials, skill directories, the oracle) and prints the plan without +running anything; `wright` is taken from `PATH` unless `--wright` or the config says +otherwise, and the credentials each adapter needs are passed through automatically. + +```sh +python3 benchmarks/agent/agent_bench.py evaluate --adapter devin --model swe-2-max --dry-run +python3 benchmarks/agent/agent_bench.py evaluate --adapter devin --model swe-2-max +``` + +`agent_bench.py evaluate --adapter ADAPTER --model MODEL --skill-dir wright-skill=DIR` +is the one-command entry: it writes `matrix.json`, runs the cells, then writes +`report.md`, `summary.json`, `score.json`, `score.txt`, and `RESULTS.md` into +`/`. `--cells score` runs the canonical cell only; `--cells controls` +adds the baseline, `wright` without the skill, and, for OverPy scenarios, the +`overpy` controls (cells whose skill has no `--skill-dir` are skipped). It runs the +same from a terminal or from inside another agent's shell, because isolation comes +from the harness's scrubbed environment, not from its parent. It runs locally; CI +does not run it. `evaluate` passes `HOME` through and the adapter copies the +credentials it needs into an isolated home; preflight names the missing login when +one is absent. + +`--adapter direct` is the built-in loop (`adapters/direct.py`) that needs no agent +harness: it calls a model API with one `bash` tool (and `fetch` only for +knowledge `web`), lists the installed skills by name and description, and records +exact usage and a full transcript. `BENCH_MODEL` is `anthropic/` +(`ANTHROPIC_API_KEY`, optional `ANTHROPIC_BASE_URL`) or `openai/` +(`OPENAI_API_KEY`, optional `OPENAI_BASE_URL`, so OpenAI-compatible endpoints work). +Pass the key variables with `--env-pass`. Its turn, time, and output limits are +recorded in `agentInfo.protocol`. It does not sandbox the network, so pair it with +`--canary-cmd`. Scores from `direct` and from product harnesses measure different +things and are not mixed. + +### Running different models at different times + +Each `evaluate` writes its own directory and scores only its own runs, so models +and agents can be run whenever quota allows, in any order. Put the runs side by side +with + +```sh +python3 benchmarks/agent/agent_bench.py compare ~/.local/share/wright-agent-bench/runs/{devin-swe2-stage1,codex-luna-xhigh,pi-luna-xhigh} +``` + +which prints one table of scores, intervals, trials, and exclusions, and warns when +the runs differ in the Wright binary, skill contents, or suite (scenarios, grader, +oracle lock). Scores are comparable when it prints no warning. Keep the Wright +binary, skills, and scenarios unchanged between runs; changes to the rest of the +harness are disclosed in each card's `Harness` line and do not block comparison. +An interrupted run resumes by repeating the same command with the same `--name`. ## Cadence