From 75819b51d41cddb1eae84e5b7ec655c6ea6d2708 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 01:25:07 +0800 Subject: [PATCH 1/2] feat(bench): count correction rounds and pair reports against any reference Refs #466. Two metrics additions for the effectiveness benchmark: - correctionRounds: failed-validation -> workspace-edit rounds per trial. The condition tool's validating op (wright check/lint/analyze/compile as CLI calls or serve ops on any transport, overpy compile under the opy cell) reporting exit 1 followed by an edit counts once; consecutive failures merge, a pass resets, and usage errors, refusals, crashes, and protocol errors are neutral. Reported as a corr column and per-cell mean. - Repeatable --reference on report and evaluate: emits one paired section per named reference label, so lift can be measured against docs, skill, or any other control cell, not only none/none/off. serve_result_payload normalizes op payloads across stdio, jsonrpc, and MCP (content[] text blocks; isError refusals carry no exit and are neutral). Verified end to end with scripted agents through the real harness: a failed check then repair yields correctionRounds 1 under wright and 0 under none, and both requested paired sections render. --- benchmarks/agent/agent_bench.py | 14 +++-- benchmarks/agent/bench_report.py | 18 ++++--- benchmarks/agent/bench_trace.py | 58 ++++++++++++++++++-- benchmarks/agent/test_agent_bench.py | 79 ++++++++++++++++++++++++++++ docs/agent-benchmark.md | 6 ++- 5 files changed, 159 insertions(+), 16 deletions(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index feba1920..0a5378e9 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -99,6 +99,12 @@ def baseline_path(path: str) -> str: return os.pathsep.join(kept) +def references_of(args: argparse.Namespace) -> list[str]: + """The paired-comparison reference labels: `action="append"` collects them; a programmatic caller may pass a bare string.""" + value = getattr(args, "reference", None) or bench_report.BASELINE + return value if isinstance(value, list) else [value] + + def normalize_cell(raw: dict) -> dict: return {"tool": raw["tool"], "level": raw.get("level") or "bin", "skills": sorted(raw.get("skills") or []), "knowledge": raw["knowledge"], "network": raw["network"]} @@ -455,6 +461,7 @@ def run_trial(scenario: dict, cell: dict, args: argparse.Namespace, out: Path) - entry = workspace / scenario["entry"] final_sha = hashlib.sha256(entry.read_bytes()).hexdigest() if entry.is_file() else None result["friction"] = bench_trace.friction(events) + result["correctionRounds"] = bench_trace.correction_rounds(events, snaps) result["expectations"] = bench_trace.detect_expectations(events, snaps, scenario, final_sha) result["snapshots"] = snapshot_validity(scenario, snaps, args.wright, out) first_valid = next((s["t"] for s in result["snapshots"]["series"] if s["valid"]), None) @@ -747,7 +754,7 @@ def cmd_evaluate(args: argparse.Namespace) -> int: status = cmd_matrix(args) if not list(args.out.glob("*/*/*/result.json")): return status or 1 - bench_report.main([args.out], args.wright, False, load_scenario, bench_report.BASELINE) + bench_report.main([args.out], args.wright, False, load_scenario, references_of(args)) languages = ["workshop", "opy"] expected = {lang: [s for s in all_scenario_ids() if load_scenario(s)["language"] == lang and load_scenario(s).get("split") == "test"] for lang in languages} bench_score.main([args.out], languages, expected, None) @@ -863,6 +870,7 @@ def main() -> int: ev.add_argument("--trials", type=int, default=3) ev.add_argument("--parallel", type=int, default=1, help="trials at a time; sequential by default so provider limits are not hit, and a run can continue across sessions") ev.add_argument("--seed", type=int, default=1) + ev.add_argument("--reference", action="append", help="condition label the report's paired comparison is made against; repeatable (default: none/none/off)") ev.add_argument("--dry-run", action="store_true", help="check the setup and print what would run, without running it") ev.add_argument("--no-file-sandbox", action="store_true", help="run without the macOS file sandbox: the agent can then read the scenario answer keys") sub.add_parser("setup-oracle", help="install the pinned upstream OverPy oracle") @@ -879,7 +887,7 @@ def main() -> int: report.add_argument("dirs", nargs="+", type=Path) report.add_argument("--regrade", action="store_true", help="re-grade stored workspaces twice and flag unstable graders") report.add_argument("--wright", default=str(ROOT / "target/debug/wright")) - report.add_argument("--reference", default=bench_report.BASELINE, help="condition label the paired comparison is made against") + report.add_argument("--reference", action="append", help="condition label a paired comparison is made against; repeatable for lift against several references (default: none/none/off)") compare = sub.add_parser("compare", help="one table from the score.json of several evaluation runs, warning when they are not comparable") compare.add_argument("dirs", nargs="+", type=Path) score = sub.add_parser("score", help="compute the Wright Agent Score card of each language track from canonical test runs") @@ -915,7 +923,7 @@ def main() -> int: if args.command == "wiki-skill": return cmd_wiki_skill(args) if args.command == "report": - return bench_report.main(args.dirs, args.wright, args.regrade, lambda s: load_scenario(s), args.reference) + return bench_report.main(args.dirs, args.wright, args.regrade, lambda s: load_scenario(s), references_of(args)) if args.command == "leaderboard": return bench_leaderboard.main(args.dirs, args.page_out or args.dirs[0].parent / "leaderboard") if args.command == "compare": diff --git a/benchmarks/agent/bench_report.py b/benchmarks/agent/bench_report.py index 998e4a2e..1cc77a64 100644 --- a/benchmarks/agent/bench_report.py +++ b/benchmarks/agent/bench_report.py @@ -85,6 +85,7 @@ def group_rows(runs: list[dict]) -> dict: "peakContext": mean(peaks) if peaks else None, "seconds": mean(r["agent"]["seconds"] for r in runs), "usedTool": sum(1 for r in runs if any(u["invocations"] for u in r.get("toolUse", {}).values())), + "correctionRounds": mean_of([r.get("correctionRounds") for r in runs]), } @@ -238,13 +239,13 @@ def setup_rows(runs: list[dict]) -> list[str]: return out -def render(results: list[dict], regrade: list[str] | None = None, reference: str = BASELINE) -> tuple[str, dict]: +def render(results: list[dict], regrade: list[str] | None = None, references: list[str] | None = None) -> tuple[str, dict]: invalid = [r for r in results if r["status"] == "invalid"] infrastructure = [r for r in results if r["status"] == "provider-interrupted"] runs = [r for r in results if r["status"] not in ("invalid", "provider-interrupted")] out = ["# Agent benchmark report", "", f"{len(runs)} valid run(s), {len(invalid)} invalid, {len(infrastructure)} infrastructure failures excluded.", ""] summary: dict = {"cells": {}, "infrastructureFailures": len(infrastructure)} - out += ["## Outcome by agent and condition", "", "| agent | condition | runs | usable | passed | used a tool | tokens/run | tokens per usable | peak context | s/run |", "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |"] + out += ["## Outcome by agent and condition", "", "| agent | condition | runs | usable | passed | used a tool | tokens/run | tokens per usable | peak context | s/run | corr |", "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |"] for agent in sorted({r["agent"]["id"] for r in runs}): for cell in sorted({label(r) for r in runs}): group = [r for r in runs if r["agent"]["id"] == agent and label(r) == cell] @@ -253,7 +254,7 @@ def render(results: list[dict], regrade: list[str] | None = None, reference: str row = group_rows(group) summary["cells"][f"{agent}|{cell}"] = row out.append(f"| {agent} | {cell} | {row['n']} | {rate_runs(group)} | {row['passed']}/{row['n']} | {row['usedTool']}/{row['n']} | " - f"{fmt(row['tokens'])} | {fmt(row['tokensPerUsable'])} | {fmt(row['peakContext'])} | {fmt(row['seconds'], 1)} |") + f"{fmt(row['tokens'])} | {fmt(row['tokensPerUsable'])} | {fmt(row['peakContext'])} | {fmt(row['seconds'], 1)} | {fmt(row['correctionRounds'], 1)} |") out += setup_rows(runs) out += ["", "## By scenario", "", "| scenario | agent | condition | usable |", "| --- | --- | --- | --- |"] groups: dict[tuple, list[dict]] = defaultdict(list) @@ -283,9 +284,10 @@ def render(results: list[dict], regrade: list[str] | None = None, reference: str "search/read shell command (a `bash` call invoking the wright CLI counts once, as a wright call).", "", "| agent | cell | level | runs | usable | passed | search/read | wright calls | bash calls | tool calls | turns | tokens/run |", "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |", *level_lines] - pairs = paired(runs, reference) - if pairs: - out += ["", f"## Paired against `{reference}` (same scenario, agent, trial)", "", "| agent | comparison | pairs | usable gained/lost | tokens where both usable |", "| --- | --- | --- | --- | --- |", *pairs] + for reference in references or [BASELINE]: + pairs = paired(runs, reference) + if pairs: + out += ["", f"## Paired against `{reference}` (same scenario, agent, trial)", "", "| agent | comparison | pairs | usable gained/lost | tokens where both usable |", "| --- | --- | --- | --- | --- |", *pairs] exp = expectations(runs) if exp: out += ["", "## Expectation rates (pass/(pass+fail); n/a and unavailable excluded)", "", "| condition | expectations |", "| --- | --- |", *exp] @@ -324,13 +326,13 @@ def render(results: list[dict], regrade: list[str] | None = None, reference: str return "\n".join(out) + "\n", summary -def main(dirs: list[Path], wright: str, regrade: bool, load_scenario, reference: str = BASELINE) -> int: +def main(dirs: list[Path], wright: str, regrade: bool, load_scenario, references: list[str] | None = None) -> int: results = load(dirs) if not results: print("no results found") return 1 notes = regrade_notes([r for r in results if r["status"] != "invalid"], wright, load_scenario) if regrade else None - text, summary = render(results, notes, reference) + text, summary = render(results, notes, references) (dirs[0] / "report.md").write_text(text) write_json(dirs[0] / "summary.json", summary) print(text) diff --git a/benchmarks/agent/bench_trace.py b/benchmarks/agent/bench_trace.py index 41e2df3e..bf287d56 100644 --- a/benchmarks/agent/bench_trace.py +++ b/benchmarks/agent/bench_trace.py @@ -15,7 +15,7 @@ TOKEN_BYTES = 4 # estimation only: bytes per token for Wright output attribution DECISION_COMMANDS = ("check", "lint", "analyze", "inspect") JSONRPC_METHODS = ("compile", "check", "analyze", "inspect") # serve.rs's direct methods — `lint` exists only as a CLI command -VALIDATING = ("check", "lint", "analyze", "compile") +VALIDATING = {"wright": ("check", "lint", "analyze", "compile"), "overpy": ("compile",)} # ops whose exit 1 means "ran fine, found problems" OUTPUT_FORMAT_FLAGS = ("--format", "-f") @@ -222,6 +222,58 @@ def serve_ops(events: list[dict]) -> list[str]: return [request["op"] for _, request, _ in serve_pairs(events) if request["op"]] +def serve_result_payload(response: dict | None) -> dict | None: + """The op's result payload from a serve response, unwrapped across transports. + + stdio and jsonrpc put the payload in `result`; an MCP tool result carries the same payload as JSON text + inside `result.content[]`. Service-level errors carry `error` instead and yield None.""" + if response is None: + return None + try: + message = json.loads(response["line"]) + except (json.JSONDecodeError, TypeError, KeyError): + return None + result = message.get("result") if isinstance(message, dict) else None + if isinstance(result, dict) and isinstance(result.get("content"), list): # an MCP tool result + text = "".join( + block.get("text") or "" for block in result["content"] if isinstance(block, dict) and block.get("type") == "text" + ) + try: + result = json.loads(text) if text else None + except json.JSONDecodeError: + return None + return result if isinstance(result, dict) else None + + +def correction_rounds(events: list[dict], snapshots: list[dict]) -> int: + """`failed validation -> edit` rounds (#466): a validating call or serve op that reported problems + (`exit` 1) followed by a workspace edit. Only clean exits mark a verdict: usage errors, refusals, + crashes, and protocol errors are all neutral — they neither count nor clear a pending failure. + The condition's tool sets the ops that count (`overpy compile` under the `opy` cell). + + Consecutive failures before one edit count as one round; an edit after a passing validation does not. + Edit markers use the poller's detection time, so ordering inside one poll interval may merge rounds.""" + markers = [(s["t"], "edit") for s in snapshots] + for call in (e for e in events if e["type"] == "call"): + if command_of(call["argv"]) in VALIDATING.get(call.get("tool"), ()) and call["exit"] in (0, 1): + markers.append((call["t"] + call.get("seconds", 0), call["exit"] == 1)) + for _event, request, response in serve_pairs(tool_events(events, "wright")): + if request["op"] in VALIDATING["wright"]: + payload = serve_result_payload(response) + exit_code = payload.get("exit") if payload else None # an MCP refusal payload has `code`, no `exit` + if exit_code in (0, 1): + markers.append((response["t"], exit_code == 1)) + rounds, pending = 0, False + for _t, kind in sorted(markers, key=lambda m: m[0]): + if kind == "edit": + if pending: + rounds += 1 + pending = False + else: + pending = kind + return rounds + + def transcript_events(path: Path): """Parsed transcript events (dicts only); empty when the transcript does not exist.""" if not path.is_file(): @@ -412,8 +464,8 @@ def detect_expectations(events: list[dict], snapshots: list[dict], scenario: dic if last_edit is None: result["E04"] = expectation("na", "no edits observed") else: - after = [c for c in calls if command_of(c["argv"]) in VALIDATING and c["t"] + c["seconds"] >= last_edit] - after += [e for e, request in ((e, p) for e, p, _ in pairs) if request["op"] in VALIDATING and e["t"] >= last_edit] + after = [c for c in calls if command_of(c["argv"]) in VALIDATING["wright"] and c["t"] + c["seconds"] >= last_edit] + after += [e for e, request in ((e, p) for e, p, _ in pairs) if request["op"] in VALIDATING["wright"] and e["t"] >= last_edit] matched = [c for c in after if c.get("envelope") and c["envelope"].get("inputIdentity") == final_sha256] result["E04"] = expectation("pass" if after else "fail", f"{len(after)} validation(s) after last edit; {len(matched)} match the final content") withheld = [c for c in calls if ((c.get("envelope") or {}).get("selection") or {}).get("withheld")] diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 9082629e..59ab9271 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -732,6 +732,54 @@ def test_friction_reports_unparseable_serve_responses_without_losing_errors(self def serve(self, line, direction="req", transport="stdio", session=7, t=0.0): return {"tool": "wright", "type": "serve", "dir": direction, "t": t, "line": line, "transport": transport, "session": session} + def test_correction_rounds_counts_failed_validation_then_edit(self): + events = [ + self.call(["check", "mode.ws"], exit_code=1, t=1.0), + self.call(["check", "mode.ws"], exit_code=1, t=3.0), + self.call(["check", "mode.ws"], exit_code=0, t=5.0), + ] + snapshots = [{"t": 2.0}, {"t": 4.0}, {"t": 6.0}] + # fail -> edit, fail -> edit: two rounds; the pass at t=5 and the edit at t=6 are not one. + self.assertEqual(bench_trace.correction_rounds(events, snapshots), 2) + # consecutive failures before one edit are one round + self.assertEqual(bench_trace.correction_rounds(events[:2], snapshots[:1]), 1) + # a pass between failure and edit is not a correction + self.assertEqual(bench_trace.correction_rounds([events[0], events[2]], [{"t": 6.0}]), 0) + self.assertEqual(bench_trace.correction_rounds([], snapshots), 0) + # usage errors and crashes are neutral: the fail -> edit round still counts, and they are not corrections + errored = self.call(["check", "mode.ws", "--bogus"], exit_code=2, t=4.0) + crashed = self.call(["check", "mode.ws"], exit_code=4, t=4.5) + self.assertEqual(bench_trace.correction_rounds([events[0], errored, crashed], [{"t": 6.0}]), 1) + self.assertEqual(bench_trace.correction_rounds([errored, crashed], [{"t": 6.0}]), 0) + # `overpy compile` is the validating op of the opy cell + opy = [self.call(["compile", "-i", "mode.opy"], exit_code=1, t=1.0, tool="overpy")] + self.assertEqual(bench_trace.correction_rounds(opy, [{"t": 2.0}]), 1) + + def test_correction_rounds_covers_serve_ops_and_ignores_refusals(self): + request = '{"op": "check"}' + failed = '{"result": {"ok": false, "exit": 1, "diagnostics": []}}' + passed = '{"result": {"ok": true, "exit": 0, "diagnostics": []}}' + refused = '{"error": {"code": -32602, "message": "bad params"}}' + events = [ + self.serve(request, t=1.0), self.serve(failed, direction="res", t=1.1), + self.serve(request, t=3.0), self.serve(passed, direction="res", t=3.1), + self.serve(request, t=5.0), self.serve(refused, direction="res", t=5.1), + self.serve(request, t=7.0), # unanswered: not a completed validation + ] + # t=1 fail -> t=2 edit; t=3 pass clears; t=5 refusal is not a correction signal; t=7 pending with no edit. + self.assertEqual(bench_trace.correction_rounds(events, [{"t": 2.0}, {"t": 6.0}, {"t": 8.0}]), 1) + + def test_correction_rounds_unwraps_mcp_tool_payloads(self): + request = '{"jsonrpc": "2.0", "id": 1, "method": "tools/call", "params": {"name": "wright_check", "arguments": {}}}' + result = '{"jsonrpc": "2.0", "id": 1, "result": {"content": [{"type": "text", "text": "{\\"ok\\": false, \\"exit\\": 1}"}]}}' + events = [self.serve(request, transport="mcp", t=1.0), self.serve(result, direction="res", transport="mcp", t=1.1)] + self.assertEqual(bench_trace.correction_rounds(events, [{"t": 2.0}]), 1) + # an isError refusal (`{code, message}` payload) is neutral: it neither fails nor clears a pending correction + refusal = '{"jsonrpc": "2.0", "id": 2, "result": {"isError": true, "content": [{"type": "text", "text": "{\\"code\\": \\"refused\\"}"}]}}' + neutral = [self.serve(request, transport="mcp", t=3.0), self.serve(refusal, direction="res", transport="mcp", t=3.1)] + self.assertEqual(bench_trace.correction_rounds(neutral, [{"t": 4.0}]), 0) + self.assertEqual(bench_trace.correction_rounds(events + neutral, [{"t": 2.0}, {"t": 4.0}]), 1) + def test_serve_request_mirrors_the_servers_answer_rule(self): notification = '{"jsonrpc":"2.0","method":"notifications/initialized"}' self.assertFalse(bench_trace.serve_request("", "mcp")["expects"]) # blank lines are skipped silently @@ -847,6 +895,37 @@ def test_wilson_interval(self): self.assertAlmostEqual(high, 0.763, places=2) self.assertEqual(bench_report.wilson(0, 0), (0.0, 0.0)) + def test_paired_lift_against_each_named_reference(self): + runs = [ + self.result("none/none/off", 1, False, 100), + self.result("none/wiki/off", 1, True, 200), + self.result("wright/none/off", 1, True, 80), + ] + text, _ = bench_report.render(runs, references=["none/none/off", "none/wiki/off"]) + self.assertIn("Paired against `none/none/off`", text) + self.assertIn("Paired against `none/wiki/off`", text) + self.assertIn("wright/none/off vs none/none/off", text) + self.assertIn("wright/none/off vs none/wiki/off", text) + self.assertIn("none/wiki/off vs none/none/off", text) + text, _ = bench_report.render(runs) # the default pairs only against the baseline + self.assertIn("Paired against `none/none/off`", text) + self.assertNotIn("Paired against `none/wiki/off`", text) + text, _ = bench_report.render(runs, references=["missing/cell/here"]) + self.assertNotIn("Paired against", text) + + def test_correction_rounds_reach_the_cell_summary(self): + run = self.result("wright/none/off", 1, True, 100) + run["correctionRounds"] = 2.0 + text, summary = bench_report.render([run]) + self.assertEqual(summary["cells"]["m|wright/none/off"]["correctionRounds"], 2.0) + self.assertIn("| corr |", text) + + def test_references_of_normalizes_flag_and_config_forms(self): + ns = argparse.Namespace(reference=["a/b/c", "d/e/f"]) + self.assertEqual(agent_bench.references_of(ns), ["a/b/c", "d/e/f"]) + self.assertEqual(agent_bench.references_of(argparse.Namespace(reference="a/b/c")), ["a/b/c"]) + self.assertEqual(agent_bench.references_of(argparse.Namespace(reference=None)), [bench_report.BASELINE]) + def test_paired_efficiency_counts_only_both_usable_and_failures_cost(self): runs = [self.result("none/none/off", 1, True, 1000), self.result("wright/none/off", 1, True, 600), self.result("none/none/off", 2, False, 900), self.result("wright/none/off", 2, True, 700)] diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index 75e7a067..5a8ebf6b 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -235,7 +235,7 @@ python3 benchmarks/agent/agent_bench.py run --agent-id LABEL --agent- --tool wright --skills wright-skill --skill-dir wright-skill=DIR \ --knowledge none --network off --trials 5 python3 benchmarks/agent/agent_bench.py matrix matrix.json # agents x cells x scenarios x trials -python3 benchmarks/agent/agent_bench.py report target/agent-bench [--regrade] [--reference none/none/off] +python3 benchmarks/agent/agent_bench.py report target/agent-bench [--regrade] [--reference none/none/off ...] python3 benchmarks/agent/agent_bench.py score target/agent-bench # Wright Agent Score cards ``` @@ -281,6 +281,7 @@ with the workspace, `agent.log`, snapshots, and the Wright trace beside it. | `toolUse` | Per tool (`wright`, `overpy`): invocations by subcommand, failures, exits of 3 or 4 (candidate owner or environment gaps), and estimated output tokens per command. Under level `mcp`, each `tools/call` counts as an invocation of the Wright operation its tool name carries; the `initialize`/`tools/list` handshake is not an invocation but its response bytes (the tool schemas) are included in `mcp:tools/list`'s estimated output tokens | | `toolCalls` | Model-visible tool calls by name from the adapter's normalized transcript (`bash`, `fetch`, `wright_*` under `mcp`), when the adapter writes one | | `friction`, `expectations` | Usage errors, unknown subcommands, help lookups, retries, malformed `serve` requests, unparsed `serve` responses, identical repeats; expectation E01-E12 verdicts | +| `correctionRounds` | Failed-validation → workspace-edit rounds: the condition tool's validating op (`check`/`lint`/`analyze`/`compile`, `overpy compile` under the `opy` cell) reporting `exit` 1 followed by an edit; consecutive failures before one edit count once, a pass resets the sequence, and refusals are not corrections | | `snapshots` | Strict validity of each snapshot of the entry, first valid index, and valid-to-invalid regressions | | `usage`, `context` | Turns, tokens by kind, peak context (and its share of the limit), tokens to first valid; loaded context | | `invalid`, `infraRetries`, `fileReadEnforcement`, `fileWriteEnforcement`, `networkEnforcement` | Present when the run was excluded or retried; how file reads (`allow-list` with the hidden and allowed paths, or `unrestricted`), file writes (`trial-directory-only` or `unrestricted`), and network `off` (`canary-checked` or `declared-only`) were enforced | @@ -299,7 +300,8 @@ output use four bytes per token; provider-reported usage is authoritative. `agent_bench.py report` writes `report.md` and `summary.json`: usable and passed counts with scenario-clustered 95% intervals (resampling scenarios, then trials), tokens per run and per usable result, peak -context, paired comparison against `--reference` (default `none/none/off`) (same scenario, agent, and +context, mean correction rounds per condition, paired comparison against each `--reference` (repeatable for +lift against several named references; default `none/none/off`) (same scenario, agent, and trial; token comparison only where both are usable), per-scenario and per-split tables, expectation rates, friction, output size per command, and diagnostics. Diagnostics flag headroom (baseline usable rate of at least 95%), infrastructure From cfb480855f6d97b992d6c479b763ed93ad177779 Mon Sep 17 00:00:00 2001 From: Teakowa <27560638+Teakowa@users.noreply.github.com> Date: Sun, 4 Oct 2026 01:44:22 +0800 Subject: [PATCH 2/2] refactor(bench): name the generated wiki skill workshop-skill Refs #466. The generated progressive-disclosure skill was named workshop-wiki while the condition vocabulary, cells, and docs all call it workshop-skill, so the generated artifact never installed under the name the harness and result records expect. SKILL_NAME, the output-directory requirement, BUILD.json identity, help text, examples, docs, and test fixtures now agree on workshop-skill. --- benchmarks/agent/agent_bench.py | 2 +- benchmarks/agent/matrix.example.json | 2 +- benchmarks/agent/matrix.pilot.example.json | 2 +- benchmarks/agent/test_agent_bench.py | 10 +++++----- benchmarks/agent/test_wiki_skill.py | 4 ++-- benchmarks/agent/wiki_skill.py | 4 ++-- docs/agent-benchmark.md | 6 +++--- 7 files changed, 15 insertions(+), 15 deletions(-) diff --git a/benchmarks/agent/agent_bench.py b/benchmarks/agent/agent_bench.py index 0a5378e9..dd004b3b 100644 --- a/benchmarks/agent/agent_bench.py +++ b/benchmarks/agent/agent_bench.py @@ -874,7 +874,7 @@ def main() -> int: ev.add_argument("--dry-run", action="store_true", help="check the setup and print what would run, without running it") ev.add_argument("--no-file-sandbox", action="store_true", help="run without the macOS file sandbox: the agent can then read the scenario answer keys") sub.add_parser("setup-oracle", help="install the pinned upstream OverPy oracle") - skill = sub.add_parser("wiki-skill", help="build the progressive-disclosure workshop-wiki skill from a wiki snapshot") + skill = sub.add_parser("wiki-skill", help="build the progressive-disclosure workshop-skill from a wiki snapshot") skill.add_argument("--snapshot", type=Path, required=True) skill.add_argument("--out-dir", type=Path, required=True, help="new skill directory (not overwritten)") skill.add_argument("--catalog", type=Path, required=True, help="workshop-rs catalog.json, for Workshop names and ids") diff --git a/benchmarks/agent/matrix.example.json b/benchmarks/agent/matrix.example.json index 4be79d96..7b667911 100644 --- a/benchmarks/agent/matrix.example.json +++ b/benchmarks/agent/matrix.example.json @@ -72,7 +72,7 @@ ], "skill_dirs": { "wright-skill": "/abs/path/skills/skills/wright", - "workshop-skill": "/abs/path/local/workshop-wiki" + "workshop-skill": "/abs/path/local/workshop-skill" } } } diff --git a/benchmarks/agent/matrix.pilot.example.json b/benchmarks/agent/matrix.pilot.example.json index 5311b38c..745fc68b 100644 --- a/benchmarks/agent/matrix.pilot.example.json +++ b/benchmarks/agent/matrix.pilot.example.json @@ -56,7 +56,7 @@ "timeout": 3600, "skill_dirs": { "wright-skill": "/abs/path/skills/skills/wright", - "workshop-skill": "/abs/path/local/workshop-wiki" + "workshop-skill": "/abs/path/local/workshop-skill" } } } diff --git a/benchmarks/agent/test_agent_bench.py b/benchmarks/agent/test_agent_bench.py index 59ab9271..a1de9377 100644 --- a/benchmarks/agent/test_agent_bench.py +++ b/benchmarks/agent/test_agent_bench.py @@ -54,20 +54,20 @@ def test_wiki_snapshot_is_copied_and_identified(self): self.assertNotIn("wiki", " ".join(result["unsafeEdits"])) def test_skills_are_installed_identified_and_the_only_ones_expected(self): - skill = self.out / "workshop-wiki" + skill = self.out / "workshop-skill" skill.mkdir() - (skill / "SKILL.md").write_text("---\nname: workshop-wiki\n---\n") + (skill / "SKILL.md").write_text("---\nname: workshop-skill\n---\n") skill_hash = wiki_skill.content_hash(skill) with self.assertRaisesRegex(SystemExit, "requires --skill"): self.trial("true", skills=("workshop-skill",)) - expected = self.trial("echo '{\"loaded\": [\"workshop-wiki\"]}' > \"$BENCH_CONTEXT\"; echo \"$BENCH_SKILL_DIRS\" > dirs.txt", skills=("workshop-skill",), skill_dirs={"workshop-skill": skill}) + expected = self.trial("echo '{\"loaded\": [\"workshop-skill\"]}' > \"$BENCH_CONTEXT\"; echo \"$BENCH_SKILL_DIRS\" > dirs.txt", skills=("workshop-skill",), skill_dirs={"workshop-skill": skill}) self.assertNotIn("invalid", expected) self.assertEqual(expected["condition"]["label"], "wright+workshop-skill/none/off") self.assertEqual(expected["environment"]["skills"]["workshop-skill"]["sha256"], skill_hash) self.assertEqual((self.out / f"{SCENARIO}-wright/workspace/dirs.txt").read_text().strip(), str(skill.resolve())) - stray = self.trial("echo '{\"loaded\": [\"workshop-wiki\", \"other\"]}' > \"$BENCH_CONTEXT\"", skills=("workshop-skill",), skill_dirs={"workshop-skill": skill}) + stray = self.trial("echo '{\"loaded\": [\"workshop-skill\", \"other\"]}' > \"$BENCH_CONTEXT\"", skills=("workshop-skill",), skill_dirs={"workshop-skill": skill}) self.assertIn("unexpected loaded context", stray["invalid"]) - (skill / "BUILD.json").write_text(json.dumps({"name": "workshop-wiki", "skillSha256": "0" * 64})) + (skill / "BUILD.json").write_text(json.dumps({"name": "workshop-skill", "skillSha256": "0" * 64})) with self.assertRaisesRegex(SystemExit, "skill content mismatch"): self.trial("true", skills=("workshop-skill",), skill_dirs={"workshop-skill": skill}) diff --git a/benchmarks/agent/test_wiki_skill.py b/benchmarks/agent/test_wiki_skill.py index 699b7f32..232b50cf 100644 --- a/benchmarks/agent/test_wiki_skill.py +++ b/benchmarks/agent/test_wiki_skill.py @@ -40,7 +40,7 @@ def setUp(self): records.append({"slug": slug, "categories": cats, "title": title, "updatedAt": "2026-07-10T18:15:21.022Z", "contentHash": "abc", "sha256": hashlib.sha256(raw).hexdigest()}) identity_text = "\n".join(f"{d['slug']} {d['sha256']}" for d in sorted(records, key=lambda d: d["slug"])) (snap / "SNAPSHOT.json").write_text(json.dumps({"snapshotSha256": hashlib.sha256(identity_text.encode()).hexdigest(), "documents": records})) - self.snap, self.out = snap, self.tmp / "workshop-wiki" + self.snap, self.out = snap, self.tmp / "workshop-skill" def build(self, upstream: str = UPSTREAM) -> dict: return wiki_skill.build(self.snap, self.out, CATALOG, MANIFEST, upstream) @@ -48,7 +48,7 @@ def build(self, upstream: str = UPSTREAM) -> dict: def test_layout_and_progressive_disclosure(self): stats = self.build() skill = (self.out / "SKILL.md").read_text() - self.assertIn("name: workshop-wiki", skill) + self.assertIn("name: workshop-skill", skill) self.assertLess(len(skill.splitlines()), 40) self.assertEqual(stats["articles"], 5) categories = (self.out / "references/categories.md").read_text() diff --git a/benchmarks/agent/wiki_skill.py b/benchmarks/agent/wiki_skill.py index eeac33e7..be67fe44 100644 --- a/benchmarks/agent/wiki_skill.py +++ b/benchmarks/agent/wiki_skill.py @@ -1,4 +1,4 @@ -"""Generate the `workshop-wiki` skill from a pinned wiki snapshot (#414, SPEC-414). +"""Generate the `workshop-skill` skill from a pinned wiki snapshot (#414, SPEC-414). The skill is community knowledge that works without Wright: a short SKILL.md, one index per category, and one file per article. It is built deterministically from `SNAPSHOT.json`; nothing is written by hand except SKILL.md. Spellings @@ -18,7 +18,7 @@ import bench_wiki KIND = {"actions": "action", "values": "value", "events": "event", "constants": "constant", "references": "reference"} -SKILL_NAME = "workshop-wiki" +SKILL_NAME = "workshop-skill" DESCRIPTION = ( "Use when you are unsure of the exact name, parameters, or behavior of an Overwatch Workshop action, value, event, " "or constant, in raw Workshop script or OverPy, or hit a Workshop quirk such as timing, event semantics, or a HUD or " diff --git a/docs/agent-benchmark.md b/docs/agent-benchmark.md index 5a8ebf6b..d09d1516 100644 --- a/docs/agent-benchmark.md +++ b/docs/agent-benchmark.md @@ -61,7 +61,7 @@ The default categories are actions, values, events, constants, and references; add `tutorials` through `--categories` for a separate second-tier experiment. The mirror's manifest is incomplete and is not the crawl source. -The `workshop-skill` is built by `agent_bench.py wiki-skill`. It is a separate skill (generated name `workshop-wiki`) with a short `SKILL.md`, category +The `workshop-skill` is built by `agent_bench.py wiki-skill`. It is a separate skill (generated name `workshop-skill`) with a short `SKILL.md`, category indexes, and individual articles. It takes a pinned snapshot, the workshop-rs catalog, and the opy-rs manifest; OverPy spellings are included only when found in the pinned upstream oracle. The generated skill is community guidance, not canonical semantic @@ -71,12 +71,12 @@ is not counted as an edit, so edits there are not detected. ```sh python3 benchmarks/agent/agent_bench.py wiki-skill \ - --snapshot /abs/path/pinned-wiki --out-dir /abs/path/local/workshop-wiki \ + --snapshot /abs/path/pinned-wiki --out-dir /abs/path/local/workshop-skill \ --catalog /abs/path/workshop-rs/crates/workshop-rs/src/catalog/data/catalog.json \ --opy-manifest /abs/path/opy-rs/crates/opy-rs/src/manifest/data/manifest.json ``` -The output directory must be named `workshop-wiki` and must not exist. Run +The output directory must be named `workshop-skill` and must not exist. Run `setup-oracle` first. Snapshots and derived skills are local benchmark material; do not commit or distribute them. The [Workshop.codes Terms of Service](https://workshop.codes/tos) apply to the source content; generating a skill grants no additional permission.