From 620dfccd4192ceb97b0e439038d1c4914add3873 Mon Sep 17 00:00:00 2001 From: Mengye Ren Date: Tue, 29 Sep 2026 20:43:20 -0400 Subject: [PATCH 1/2] Refuse out-of-scope launches and submits instead of ending the run Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 11 ++ docs/design/lifecycle.md | 16 ++- src/outerloop/orchestrator.py | 81 +++++++++----- tests/test_attempt_review.py | 62 ++++++----- tests/test_inbox.py | 46 ++++++++ tests/test_orchestrator.py | 192 +++++++++++++++++++++++++++++++++- 6 files changed, 352 insertions(+), 56 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 21e9f009..110dd4fc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,17 @@ Versions follow [SemVer](https://semver.org). ## [Unreleased] +- Launch, submit, and stale-submit checkpoint scope violations now refuse once + and resume the author with the offending paths in the kernel inbox. Refusal + seals nothing, runs no jobs or measurements, and spends no request budget. + A second scope violation without an accepted request remains terminal, as + does the authoritative measurement scope check. +- Upgrading: no action or backfill needed. Scope refusals use the existing + kernel note payload and refusal keys; old inboxes, ended runs, and in-flight + runs remain readable without rewriting delivered messages. The next request + uses the new admission behavior. No run-record fields change; rollback is + safe and restores terminal scope admission checks. + - Deployment author overrides select a backend/model per target or agent slot, bind it to each run, and leave panel and CI reviewer inheritance on the fleet author. Per-run board details identify the effective author and overrides. diff --git a/docs/design/lifecycle.md b/docs/design/lifecycle.md index 2f67ba70..25857e82 100644 --- a/docs/design/lifecycle.md +++ b/docs/design/lifecycle.md @@ -228,7 +228,7 @@ comment never does. | Kept | Why it is the kernel's | | --- | --- | | the paired measurement, the private seed, the floor, the suite, the cached baseline rule, the zero-change rule (an unchanged tree cannot be credited), the verdict bound to its sealed tree, base and contract | the number must be nobody's claim | -| scope on the diff before anything is sealed, launched or measured | the out-of-scope edit could be to the ruler | +| scope on the diff before anything is sealed, launched or measured; one uncharged refusal before a consecutive violation ends the run | the out-of-scope edit could be to the ruler | | containment, the lane from the contract, `--nice` on launches, always queue, cancel on end | the session cannot hold GPUs or credentials | | launch, sleep and GPU-hour counts; refusal on exhaustion with the numbers | the meter is the only bound on spend | | the publish: open or fast-forward the PR head to the sealed tree, the ledger row rule, disarm before a head moves, refuse when a human pushed or the contract moved; under `merge: auto`, record the blessed head and let the sweep merge a clean, quiet PR at that head only, never arm GitHub auto-merge; otherwise humans merge | credit, merge authority, and nobody's work overwritten | @@ -236,6 +236,20 @@ comment never does. | standing: which comments are messages, the bot's own markers, the task label; the issue claim and its release; one delivery per message | authorization and liveness | | leases, the sweep, deadline floors, the stuck cap, the outage latch, the tamper guard, the report on every ending, the line seal at every terminal | liveness and audit | +Scope admission for launch, submit, and stale-submit checkpoints refuses the +first out-of-scope tree with a kernel inbox note listing up to ten paths. +Nothing is sealed, launched, measured, or charged for the refused request; +the author can retry once the tree only changes contract-allowed paths. +Scope has its own in-memory refusal bound, independent of malformed-request +and budget refusals, so an unrelated refusal does not remove this recovery +opportunity. An accepted request resets the bound; a second scope violation +without acceptance ends as `scope-violation`. A session that cannot resume +also ends as `scope-violation`; endpoint loss during refusal delivery ends as +`session-outage` without sealing the rejected tree. The note remains in the +inbox for operators and later readers. The authoritative scope re-check in +`measure_and_decide`, including wake re-entry, remains terminal. Judges still +receive the run record. + ## What becomes the author's, and what is deleted | Today | After | diff --git a/src/outerloop/orchestrator.py b/src/outerloop/orchestrator.py index a2a57d13..9d977b3e 100644 --- a/src/outerloop/orchestrator.py +++ b/src/outerloop/orchestrator.py @@ -1488,11 +1488,13 @@ def _ack(messages: list[Message]) -> None: baseline_note = "" measured: tuple[str, ...] = () refused_once = False + # Independent of malformed/budget refusals; only acceptance resets this bound. + scope_refused_once = False # Reuse a negative only for the same measurement base and candidate tree. failed_gate: tuple[str, str, AttemptResult] | None = judged tree = tree_of or (lambda sha: sha) - def _resume(message: Message) -> AttemptResult | None: + def _resume(message: Message, *, allow_checkpoint: bool = True) -> AttemptResult | None: """Deliver pending messages; return an ending only if the resume fails.""" nonlocal session if on_meter is not None: @@ -1517,7 +1519,16 @@ def _resume(message: Message) -> AttemptResult | None: wake_result = run_role( spec, harness, prompt, workspace, resume_session_id=session.session_id ) - except EndpointUnavailable: + except EndpointUnavailable as exc: + if not allow_checkpoint: + # A rejected tree cannot be sealed even to preserve an outage park. + return AttemptResult( + outcome="session-outage", + baseline=baseline, + session=session, + note=f"scope refusal delivery unavailable: {exc}", + run_seed=run_seed, + ) # A prior leg already ran: keep its native context and budget meters. raise RunParked( phase="author-sleep", @@ -1693,19 +1704,51 @@ def _not_run_note(request: SyscallRequest | None) -> str: request, gpus=bench.gpus, max_concurrent_gpus=contract.budgets.max_concurrent_gpus ) if not problem: - if stale_submit: - violations = scope_validator(list(changed_paths()), contract) - if violations: + # Admission precedes every seal, dispatch and budget charge, including + # a stale submit's jobless checkpoint. Keep the measurement backstop. + measured = tuple(changed_paths()) + violations = scope_validator(list(measured), contract) + if violations: + where = ( + " at checkpoint" if stale_submit else "" if request.submit else " at launch" + ) + problem = f"out-of-scope paths{where}: {', '.join(sorted(violations)[:10])}" + if scope_refused_once or not _can_resume(): return AttemptResult( outcome="scope-violation", baseline=baseline, session=session, - note=( - "out-of-scope paths at checkpoint: " - + ", ".join(sorted(violations)[:10]) - ), + note=problem, run_seed=run_seed, + panel_transcript="\n\n".join(panel_sections), + panel_rounds=panel_reads, ) + scope_refused_once = True + presealed = "" + failed = _resume( + Message( + 0, + "note", + "kernel", + inbox_thread, + time.time(), + f"refusal:{session.session_id}:{inbox_seq}", + { + "text": ( + "Your syscall request was REFUSED. Nothing was snapshotted, " + "launched, measured or charged. The request can be made again " + "once the tree only changes paths the contract allows." + ), + "quoted_text": problem, + }, + origin=inbox_dir.name, + ), + allow_checkpoint=False, + ) + if failed is not None: + return failed + continue + if stale_submit: sha = snapshot() sleeps_used += 1 if preflight.status == "stale": @@ -1778,6 +1821,7 @@ def _not_run_note(request: SyscallRequest | None) -> str: # the walltime THIS submit declares (else the contract's): # a resubmit without a declaration reverts to the default, # never inheriting a prior park's. + scope_refused_once = False submitted = request sleeps_used += 1 evals_charge = evals_gpu_hours( @@ -1793,22 +1837,6 @@ def _not_run_note(request: SyscallRequest | None) -> str: break # a launch park: its launches are dispatched right below, so # they are charged now - # Scope BEFORE the snapshot, same invariant as the candidate - # path below: an out-of-scope tree is never snapshotted OR - # executed — the out-of-scope edit could be to the ruler - # itself, and a launch runs code from this tree in an external - # job. Same ending as the candidate path. - violations = scope_validator(list(changed_paths()), contract) - if violations: - return AttemptResult( - outcome="scope-violation", - baseline=baseline, - session=session, - note=( - f"out-of-scope paths at launch: {', '.join(sorted(violations)[:10])}" - ), - run_seed=run_seed, - ) sha = snapshot() assert launcher is not None try: @@ -1873,7 +1901,8 @@ def _not_run_note(request: SyscallRequest | None) -> str: return AttemptResult( outcome="no-improvement", session=session, note="ended without a submit" ) - measured = tuple(changed_paths()) + if submitted is None: + measured = tuple(changed_paths()) # Scope BEFORE the snapshot: an out-of-scope tree is never snapshotted # OR measured — the out-of-scope edit could be to the ruler itself. This # early exit keeps the snapshot off a rejected tree; measure_and_decide diff --git a/tests/test_attempt_review.py b/tests/test_attempt_review.py index 7aac6d25..4596a90d 100644 --- a/tests/test_attempt_review.py +++ b/tests/test_attempt_review.py @@ -450,9 +450,14 @@ def test_review_launch_checks_committed_edits(review_run, monkeypatch): class CommittingHarness(ResumingHarness): def run(self, brief_text, workspace, resume_session_id=None): - (workspace / "docs/roadmap.md").write_text("out of scope") - _git(workspace, "add", "docs/roadmap.md") - _git(workspace, "-c", "user.name=t", "-c", "user.email=t@t", "commit", "-qm", "edit") + if not self.calls: + (workspace / "docs/roadmap.md").write_text("out of scope") + _git(workspace, "add", "docs/roadmap.md") + _git( + workspace, "-c", "user.name=t", "-c", "user.email=t@t", "commit", "-qm", "edit" + ) + else: + assert "REFUSED" in brief_text and "out-of-scope paths" in brief_text assert ( main(["launch", "--name", "probe", "--minutes", "1", "--", "true"], root=workspace) == 0 @@ -460,15 +465,17 @@ def run(self, brief_text, workspace, resume_session_id=None): assert main(["sleep"], root=workspace) == 0 return super().run(brief_text, workspace, resume_session_id) + author = CommittingHarness() out = wake_review( root, "tsp-r1", - CommittingHarness(), + author, cast(GitHubClient, FakeGitHub(comments=[member(101, "experiment")])), bot_login=BOT, now=NOW, dispatch=DispatchSettings(compute=LocalCompute(), image="", account="", partition=""), ) + assert len(author.calls) == 2 assert out.action == "scope-violation" assert ( "out-of-scope paths at launch: docs/roadmap.md" @@ -1593,28 +1600,31 @@ def launch(sha, request): class FoldingHarness(ResumingHarness): def run(self, brief_text, workspace, resume_session_id=None): - _git(workspace, "reset", "--mixed", "origin/main") - _git(workspace, "checkout", "origin/main", "--", "BENCHMARKS.md") - (workspace / "src/pilot/solvers/tsp.py").write_text("author's edit\n") - reference = ( - head - if rollback_to == "head" - else _git(workspace, "rev-parse", "origin/main").strip() - ) - if moved_again: - if rollback_to: - (seed / "docs/roadmap.md").write_text("reviewed ruler B2\n") - (seed / "BENCHMARKS.md").write_text("main's next ledger\n") - _git(seed, "add", "-A") - _git(seed, "-c", "user.name=t", "-c", "user.email=t@t", "commit", "-qm", "B2") - _git(seed, "push", str(bare), "main") - _git(workspace, "fetch", "origin") - if rollback_to: + if not self.calls: _git(workspace, "reset", "--mixed", "origin/main") _git(workspace, "checkout", "origin/main", "--", "BENCHMARKS.md") - _git(workspace, "checkout", reference, "--", "docs/roadmap.md") - if edit_ledger: - (workspace / "BENCHMARKS.md").write_text("author's ledger\n") + (workspace / "src/pilot/solvers/tsp.py").write_text("author's edit\n") + reference = ( + head + if rollback_to == "head" + else _git(workspace, "rev-parse", "origin/main").strip() + ) + if moved_again: + if rollback_to: + (seed / "docs/roadmap.md").write_text("reviewed ruler B2\n") + (seed / "BENCHMARKS.md").write_text("main's next ledger\n") + _git(seed, "add", "-A") + _git(seed, "-c", "user.name=t", "-c", "user.email=t@t", "commit", "-qm", "B2") + _git(seed, "push", str(bare), "main") + _git(workspace, "fetch", "origin") + if rollback_to: + _git(workspace, "reset", "--mixed", "origin/main") + _git(workspace, "checkout", "origin/main", "--", "BENCHMARKS.md") + _git(workspace, "checkout", reference, "--", "docs/roadmap.md") + if edit_ledger: + (workspace / "BENCHMARKS.md").write_text("author's ledger\n") + else: + assert "REFUSED" in brief_text and "out-of-scope paths" in brief_text assert ( main(["launch", "--name", "probe", "--minutes", "1", "--", "true"], root=workspace) == 0 @@ -1623,8 +1633,10 @@ def run(self, brief_text, workspace, resume_session_id=None): return super().run(brief_text, workspace, resume_session_id) github = FakeGitHub(pr={"state": "open", "head": {"sha": head}}) - outcome = wake_review(root, "tsp-r1", FoldingHarness(), github) + author = FoldingHarness() + outcome = wake_review(root, "tsp-r1", author, github) refused = bool(edit_ledger or rollback_to) + assert len(author.calls) == (2 if refused else 1) assert outcome.action == ("scope-violation" if refused else "parked") assert bool(launched) is not refused if rollback_to: diff --git a/tests/test_inbox.py b/tests/test_inbox.py index a765e0eb..1fa976b2 100644 --- a/tests/test_inbox.py +++ b/tests/test_inbox.py @@ -1202,3 +1202,49 @@ def interrupted(*args, **kwargs): assert len(pending(directory, 0)) == 2 assert path.read_text() == original assert append(directory, Message(**fixture["message"])).seq == 1 + + +@pytest.mark.parametrize("version", [1, 2]) +def test_legacy_refusal_note_survives_read_and_interrupted_append(tmp_path, monkeypatch, version): + import json + from dataclasses import asdict + + import outerloop.inbox as inbox + + # The existing refusal payload, before scope admission could refuse. + old = replace( + message("note", key="refusal:s1:0"), + seq=1, + payload={ + "text": "Your syscall request was REFUSED and nothing was launched.", + "quoted_text": "launch budget exhausted", + }, + ) + raw = asdict(old) + if version == 1: + for field in ("message_id", "context_id", "to", "in_reply_to"): + raw.pop(field) + else: + raw["v"] = 2 + directory = tmp_path / "inbox" + directory.mkdir() + path = directory / "000001.json" + original = json.dumps(raw) + path.write_text(original) + for _ in range(2): + rendered = render_inbox(pending(tmp_path, 0), budgets="budget") + assert old.payload["text"] in rendered + assert old.payload["quoted_text"] in rendered + assert path.read_text() == original + new = message("note", key="refusal:s1:1", text="Your syscall request was REFUSED.") + with monkeypatch.context() as patch: + patch.setattr( + inbox.os, "replace", lambda *a, **kw: (_ for _ in ()).throw(OSError("interrupted")) + ) + with pytest.raises(OSError, match="interrupted"): + append(tmp_path, new) + append(tmp_path, new) + append(tmp_path, new) + assert len(pending(tmp_path, 0)) == 2 + assert render_inbox(pending(tmp_path, 0)[:1], budgets="budget") == rendered + assert path.read_text() == original diff --git a/tests/test_orchestrator.py b/tests/test_orchestrator.py index e161fe81..8a154bfe 100644 --- a/tests/test_orchestrator.py +++ b/tests/test_orchestrator.py @@ -635,10 +635,10 @@ def snapshot() -> str: assert result.outcome == "eval-error" and "snapshot" in result.note assert seals["n"] == 2 - harness = _SeqHarness(["the claim", "again"], submit_on=(1, 2)) + harness = _SeqHarness(["the claim", "again", "still dirty"], submit_on=(1, 2, 3)) evaluator = FakeEvaluator(values=[13.9, 13.9]) measurer, snapshot2 = _wire(evaluator, tmp_path) - paths = iter([["src/pilot/solvers/tsp.py"], ["src/pilot/solvers/tsp.py", "docs/roadmap.md"]]) + paths = iter([["src/pilot/solvers/tsp.py"], ["docs/roadmap.md"], ["docs/roadmap.md"]]) result = attempt_once( CONFIG, CONTRACT, @@ -745,7 +745,7 @@ def test_author_sleep_refuses_an_out_of_scope_tree_before_launching(tmp_path: Pa launcher=_fake_launcher(launched), changed=["docs/ruler-tamper.md"], ) - assert result.outcome == "scope-violation" and launched == [] + assert result.outcome == "no-improvement" and launched == [] def test_malformed_syscall_request_is_a_loud_error(tmp_path: Path) -> None: @@ -2305,7 +2305,7 @@ def test_stale_checkpoint_checks_scope_before_sealing(tmp_path): launcher=lambda *a: pytest.fail("launch"), submit_preflight=lambda: SubmitPreflight("stale", "tip", "main"), ) - assert result.outcome == "scope-violation" and not evaluator.calls + assert result.outcome == "no-improvement" and not evaluator.calls def test_withdraw_consumed_after_session_without_compute(tmp_path): @@ -2434,3 +2434,187 @@ def run(self, brief_text, workspace, resume_session_id=None): assert park.capacity_wait and park.phase == "author-sleep" assert park.session and park.session.session_id == "s1" assert park.candidate_sha and park.sleeps_used == 0 and park.launches_used == 0 + + +@pytest.mark.parametrize("kind", ["launch", "submit", "stale", "outdated-pin"]) +@pytest.mark.parametrize("retry", ["end", "dirty", "clean", "outage"]) +def test_scope_refusal_runs_nothing_and_resumes(tmp_path, kind, retry): + from outerloop.endpoints import EndpointUnavailable + from outerloop.inbox import pending + from outerloop.orchestrator import RunParked, SubmitPreflight + + # Nonzero starting meters catch both accidental charges and resets. + meters: list[tuple[int, int, float]] = [] + seals: list[tuple[str, ...]] = [] + launches: list = [] + paths = [".github/workflows/ci.yml", *[f"report-{i:02}.md" for i in range(12)]] + request: dict[str, object] = {"launches": [{"name": "probe", "command": "x", "minutes": 1}]} + if kind != "launch": + request.update(submit=True, report="H: candidate") + evaluator = FakeEvaluator(values=[13.9, 13.9]) + measurer, _ = _wire(evaluator, tmp_path) + directory = tmp_path / "run" + + class Author: + calls = 0 + + def run(self, brief_text, workspace, resume_session_id=None): + self.calls += 1 + if self.calls == 1: + _write_syscall(workspace, request) + elif self.calls == 2: + assert resume_session_id == "s1" + assert "REFUSED" in brief_text + assert "Nothing was snapshotted, launched, measured or charged" in brief_text + assert "once the tree only changes paths the contract allows" in brief_text + assert ".github/workflows/ci.yml" in brief_text + assert "report-08.md" in brief_text and "report-09.md" not in brief_text + assert not seals and not launches and not evaluator.calls + assert meters and all(m == (1, 2, 0.25) for m in meters) + if retry == "outage": + raise EndpointUnavailable("unavailable") + if retry == "clean": + paths[:] = ["src/pilot/solvers/tsp.py"] + if retry in ("clean", "dirty"): + _write_syscall(workspace, request) + else: + assert retry == "clean" and kind == "submit" + return ok_session() + + def snapshot(): + seals.append(tuple(paths)) + return "candidate" + + contract = DEEP_CONTRACT.replace( + " direction: min\n", " direction: min\n gpus: 1\n eval_minutes: 10\n" + ) + author = Author() + + def launcher(sha, request): + launches.append((sha, request)) + return "afterany:1" + + def attempt(): + return attempt_once( + CONFIG, + contract, + tmp_path, + author, + measurer, + "base", + snapshot, + inbox_dir=directory, + ruler="r", + changed_paths=lambda: paths, + launcher=launcher, + submit_preflight=lambda: SubmitPreflight( + kind if kind in ("stale", "outdated-pin") else "ready", "tip", "main" + ), + launches_used=1, + sleeps_used=2, + gpu_hours_used=0.25, + on_meter=lambda *m: meters.append(m), + ) + + if retry == "clean" and kind != "submit": + with pytest.raises(RunParked) as caught: + attempt() + park = caught.value + assert park.sleeps_used == 3 + assert park.launches_used == (2 if kind == "launch" else 1) + assert park.gpu_hours_used == pytest.approx(0.25 + (1 / 60 if kind == "launch" else 0)) + assert len(seals) == 1 + assert len(launches) == (1 if kind == "launch" else 0) + assert not evaluator.calls + else: + result = attempt() + expected = {"dirty": "scope-violation", "outage": "session-outage"} + assert result.outcome == expected.get(retry, "no-improvement") + if retry == "clean": + assert len(seals) == 1 and len(evaluator.calls) == 2 and not launches + assert meters[-1] == pytest.approx((1, 3, 0.25 + 20 / 60)) + else: + assert not seals and not launches and not evaluator.calls + assert all(m == (1, 2, 0.25) for m in meters) + assert author.calls == 2 + if retry == "dirty": + assert ".github/workflows/ci.yml" in result.note + receipts = [m for m in pending(directory, 0) if m.key.startswith("refusal:")] + assert len(receipts) == 1 + assert receipts[0].source == "kernel" and receipts[0].kind == "note" + assert ".github/workflows/ci.yml" in receipts[0].payload["quoted_text"] + assert ".github/workflows/ci.yml" in (tmp_path / ".outerloop/messages.json").read_text() + + +def test_scope_refusal_bound_is_independent_and_resets_after_acceptance(tmp_path): + from outerloop.inbox import pending + + paths = ["report.md"] + meters: list[tuple[int, int, float]] = [] + + class Author: + calls = 0 + + def run(self, brief_text, workspace, resume_session_id=None): + self.calls += 1 + if self.calls == 1: + # Spend the existing refusal allowance before any scope refusal. + _write_syscall( + workspace, {"launches": [{"name": str(i), "command": "x"} for i in range(4)]} + ) + elif self.calls in (2, 3, 4, 5): + if self.calls == 3: + assert "out-of-scope paths" in brief_text + paths[:] = ["src/pilot/solvers/tsp.py"] + elif self.calls == 4: + paths[:] = ["report.md"] + elif self.calls == 5: + assert "out-of-scope paths" in brief_text + _write_syscall(workspace, {"submit": True, "report": "H: candidate"}) + else: + pytest.fail("unbounded scope refusal") + return ok_session() + + harness = Author() + result, _, evaluator = run_climb( + tmp_path, + [13.9, 13.9], + harness=harness, + contract=DEEP_CONTRACT, + launcher=lambda *a: pytest.fail("launch"), + changed=paths, + on_meter=lambda *m: meters.append(m), + ) + assert result.outcome == "scope-violation" and harness.calls == 5 + assert len(evaluator.calls) == 2 + assert meters[-1] == (0, 1, 0.0) + receipts = [ + m + for m in pending(tmp_path.parent / (tmp_path.name + "-run"), 0) + if m.key.startswith("refusal:") + ] + assert len(receipts) == 3 + + +def test_measure_and_decide_scope_backstop_is_terminal(): + from outerloop.contract import load_contract + from outerloop.orchestrator import AttemptResult, measure_and_decide + + class NeverMeasured: + def results(self, measures): + pytest.fail("out-of-scope tree measured") + + contract = load_contract(CONTRACT, CONFIG.target) + result = measure_and_decide( + contract, + contract.benchmarks[0], + base_sha="base", + candidate_sha="candidate", + seed=1, + suite_seed=2, + measured_paths=[".github/workflows/ci.yml"], + measurer=NeverMeasured(), + min_relative_improvement=0.005, + ) + assert isinstance(result, AttemptResult) + assert result.outcome == "scope-violation" and ".github/workflows/ci.yml" in result.note From 2630a39dc8118ab6def906ff1defe2e102b139b4 Mon Sep 17 00:00:00 2001 From: Mengye Ren Date: Tue, 29 Sep 2026 21:31:28 -0400 Subject: [PATCH 2/2] Scope refusals repeat and never end the run; plain message with the allowed scope; rejected trees never sealed Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 18 ++- docs/design/lifecycle.md | 28 ++-- src/outerloop/attempt.py | 4 +- src/outerloop/orchestrator.py | 189 +++++++++++++++--------- tests/test_attempt.py | 162 ++++++++++++++++++++ tests/test_attempt_review.py | 34 +++-- tests/test_orchestrator.py | 268 ++++++++++++++++++++++++++++++---- 7 files changed, 578 insertions(+), 125 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 110dd4fc..a2f9322e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,15 +6,21 @@ Versions follow [SemVer](https://semver.org). ## [Unreleased] -- Launch, submit, and stale-submit checkpoint scope violations now refuse once - and resume the author with the offending paths in the kernel inbox. Refusal - seals nothing, runs no jobs or measurements, and spends no request budget. - A second scope violation without an accepted request remains terminal, as - does the authoritative measurement scope check. +- Launch, submit, and stale-submit checkpoint scope violations now refuse every + request and resume the author with the offending paths and bounded allowed + scope in the kernel inbox. Later refusals say “Refused again:”. Refusals + repeat and never end the run; session walltime and contract sleep, launch, + and GPU-hour budgets bound the loop. Refusal seals nothing, runs no jobs or + measurements, and spends no request budget. The authoritative measurement + scope check remains terminal. Abandoning a refused tree ends normally + without measurement, a line snapshot, or a push; outage and budget endings + also preserve the rejection. Scope admission precedes malformed request + and budget refusals. - Upgrading: no action or backfill needed. Scope refusals use the existing kernel note payload and refusal keys; old inboxes, ended runs, and in-flight runs remain readable without rewriting delivered messages. The next request - uses the new admission behavior. No run-record fields change; rollback is + uses the new admission behavior. The rejection flag is in-memory only; + no run-record fields change; rollback is safe and restores terminal scope admission checks. - Deployment author overrides select a backend/model per target or agent slot, diff --git a/docs/design/lifecycle.md b/docs/design/lifecycle.md index 25857e82..02f9a9d6 100644 --- a/docs/design/lifecycle.md +++ b/docs/design/lifecycle.md @@ -228,7 +228,7 @@ comment never does. | Kept | Why it is the kernel's | | --- | --- | | the paired measurement, the private seed, the floor, the suite, the cached baseline rule, the zero-change rule (an unchanged tree cannot be credited), the verdict bound to its sealed tree, base and contract | the number must be nobody's claim | -| scope on the diff before anything is sealed, launched or measured; one uncharged refusal before a consecutive violation ends the run | the out-of-scope edit could be to the ruler | +| scope on the diff before anything is sealed, launched or measured; uncharged refusals repeat and never end the run | the out-of-scope edit could be to the ruler | | containment, the lane from the contract, `--nice` on launches, always queue, cancel on end | the session cannot hold GPUs or credentials | | launch, sleep and GPU-hour counts; refusal on exhaustion with the numbers | the meter is the only bound on spend | | the publish: open or fast-forward the PR head to the sealed tree, the ledger row rule, disarm before a head moves, refuse when a human pushed or the contract moved; under `merge: auto`, record the blessed head and let the sweep merge a clean, quiet PR at that head only, never arm GitHub auto-merge; otherwise humans merge | credit, merge authority, and nobody's work overwritten | @@ -236,19 +236,19 @@ comment never does. | standing: which comments are messages, the bot's own markers, the task label; the issue claim and its release; one delivery per message | authorization and liveness | | leases, the sweep, deadline floors, the stuck cap, the outage latch, the tamper guard, the report on every ending, the line seal at every terminal | liveness and audit | -Scope admission for launch, submit, and stale-submit checkpoints refuses the -first out-of-scope tree with a kernel inbox note listing up to ten paths. -Nothing is sealed, launched, measured, or charged for the refused request; -the author can retry once the tree only changes contract-allowed paths. -Scope has its own in-memory refusal bound, independent of malformed-request -and budget refusals, so an unrelated refusal does not remove this recovery -opportunity. An accepted request resets the bound; a second scope violation -without acceptance ends as `scope-violation`. A session that cannot resume -also ends as `scope-violation`; endpoint loss during refusal delivery ends as -`session-outage` without sealing the rejected tree. The note remains in the -inbox for operators and later readers. The authoritative scope re-check in -`measure_and_decide`, including wake re-entry, remains terminal. Judges still -receive the run record. +Scope admission for launch, submit, and stale-submit checkpoints refuses every +out-of-scope request with a kernel inbox note listing up to ten offending paths +and the contract's allowed scope (up to ten entries and 1,000 characters). +Later refusals in the same run start with “Refused again:”. Nothing is sealed, +launched, measured, or charged for a refused request. Refusals repeat and never +end the run, including requests with malformed content or budget problems. +The session walltime and contract sleep, launch, and GPU-hour budgets bound +the loop. A session that cannot resume ends as `session-error`; endpoint loss +during refusal delivery ends as `session-outage` without sealing the rejected +tree. Abandoning a rejected tree ends without measuring, sealing, or pushing it. +Every refusal remains in the inbox for operators and later readers. The +authoritative scope re-check in `measure_and_decide`, including wake re-entry, +remains terminal. Judges still receive the run record. ## What becomes the author's, and what is deleted diff --git a/src/outerloop/attempt.py b/src/outerloop/attempt.py index 0231ed39..3ed957be 100644 --- a/src/outerloop/attempt.py +++ b/src/outerloop/attempt.py @@ -3084,7 +3084,7 @@ def _wake_author( result = dc_replace(result, submit_report=str(stage.get("report") or "no report was given")) report_path = run_dir / "report.md" report_path.write_text(result.report(config, redact_secrets=secrets)) - if not snapshot_attempted: + if not snapshot_attempted and not result.tree_rejected: _push_line_snapshot( ws, _line_ref_for(bench, config.agent_id), @@ -3557,7 +3557,7 @@ def _finish_attempt( ) report_path = run_dir / "report.md" report_path.write_text(result.report(config, redact_secrets=secrets)) - if not snapshot_attempted: + if not snapshot_attempted and not result.tree_rejected: _push_line_snapshot( ws, line_ref, run_id, result.outcome, secrets, bot_login=config.bot_login ) diff --git a/src/outerloop/orchestrator.py b/src/outerloop/orchestrator.py index 9d977b3e..af15a24d 100644 --- a/src/outerloop/orchestrator.py +++ b/src/outerloop/orchestrator.py @@ -520,6 +520,8 @@ class AttemptResult: panel_rounds: int = 0 panel_blocking_open: bool = False panel_degraded: bool = False + # In-memory ending guard: rejected trees must never reach line snapshots. + tree_rejected: bool = False def report(self, config: RunConfig, redact_secrets: tuple[str, ...] = ()) -> str: lines = [ @@ -862,6 +864,7 @@ def measure_and_decide( if violations: return AttemptResult( outcome="scope-violation", + tree_rejected=True, note=f"out-of-scope paths: {', '.join(sorted(violations)[:10])}", run_seed=seed, ) @@ -1488,8 +1491,18 @@ def _ack(messages: list[Message]) -> None: baseline_note = "" measured: tuple[str, ...] = () refused_once = False - # Independent of malformed/budget refusals; only acceptance resets this bound. - scope_refused_once = False + # Rejection protects stop/error paths; it is never a refusal limit. + tree_rejected = False + scope_refused = any( + m.source == "kernel" + and m.kind == "note" + and m.key.startswith("refusal:") + and ( + str(m.payload.get("text", "")).startswith(("Refused:", "Refused again:")) + or str(m.payload.get("quoted_text", "")).startswith("out-of-scope paths") + ) + for m in pending_messages(inbox_dir, 0) + ) # Reuse a negative only for the same measurement base and candidate tree. failed_gate: tuple[str, str, AttemptResult] | None = judged tree = tree_of or (lambda sha: sha) @@ -1520,10 +1533,11 @@ def _resume(message: Message, *, allow_checkpoint: bool = True) -> AttemptResult spec, harness, prompt, workspace, resume_session_id=session.session_id ) except EndpointUnavailable as exc: - if not allow_checkpoint: + if not allow_checkpoint or scope_validator(list(changed_paths()), contract): # A rejected tree cannot be sealed even to preserve an outage park. return AttemptResult( outcome="session-outage", + tree_rejected=True, baseline=baseline, session=session, note=f"scope refusal delivery unavailable: {exc}", @@ -1545,6 +1559,18 @@ def _resume(message: Message, *, allow_checkpoint: bool = True) -> AttemptResult judged=failed_gate, capacity_wait=True, ) from None + except Exception as exc: + if not tree_rejected: + raise + # Preserve rejection even when the harness fails without a session result. + return AttemptResult( + outcome="session-error", + tree_rejected=True, + baseline=baseline, + session=session, + note=f"scope refusal delivery failed: {type(exc).__name__}: {exc}", + run_seed=run_seed, + ) session = wake_result.session if wake_result.ok: _ack(messages) @@ -1561,6 +1587,7 @@ def _resume(message: Message, *, allow_checkpoint: bool = True) -> AttemptResult baseline=baseline, candidate=candidate, session=session, + tree_rejected=tree_rejected, note=wake_result.error or session.error_detail or session.stop_reason, run_seed=run_seed, panel_transcript="\n\n".join(panel_sections), @@ -1586,18 +1613,95 @@ def _not_run_note(request: SyscallRequest | None) -> str: evals_charge = 0.0 # GPU-hours this pass took for gate evals presealed = "" # a seal taken early to compare against the judged tree while launcher is not None or on_replies is not None: + read_error = "" try: request = _consume_request() except SyscallError as exc: - # loud, never silent: the author meant something by the file + # Scope admission also covers unreadable requests. + read_error = f"unhonorable syscall request: {exc}" + request = SyscallRequest(launches=(), problem=read_error) + if request is None: + break + # Refresh ancestry before deriving the paths admitted below. + no_backend = ( + "sleep is not available here: this run has no compute backend for " + "launches; end your leg instead" + if launcher is None and not (on_stop and request.submit) + else "" + ) + preflight = ( + submit_preflight() + if request.sleep + and not request.problem + and request.submit + and not no_backend + and submit_preflight is not None + else SubmitPreflight("ready") + ) + stale_submit = preflight.status in ("stale", "outdated-pin") + # A valid end abandons the tree; it does not request admission. + # Otherwise scope takes precedence over syscall/budget problems. + if not (request.end and not request.problem): + # Admission precedes every seal, dispatch and budget charge, including + # a stale submit's jobless checkpoint. Keep the measurement backstop. + measured = tuple(changed_paths()) + violations = scope_validator(list(measured), contract) + if violations: + where = ( + "checkpoint" if stale_submit else "submit" if request.submit else "launch" + ) + # Preserve scope entries as written, bounded by count and length. + allowed = ", ".join(contract.scope.allowed[:10]) + if len(contract.scope.allowed) > 10: + allowed += ", …" + if len(allowed) > 1000: + allowed = allowed[:997] + "…" + prefix = "Refused again:" if scope_refused else "Refused:" + problem = ( + f"{prefix} this request at {where} changes paths the contract " + "does not allow authors to change: " + f"{', '.join(sorted(violations)[:10])}. " + "Nothing ran and nothing was charged. " + f"Authors may change: {allowed}." + ) + scope_refused = True + tree_rejected = True + presealed = "" + message = Message( + 0, + "note", + "kernel", + inbox_thread, + time.time(), + f"refusal:{session.session_id}:{inbox_seq}", + {"text": problem}, + origin=inbox_dir.name, + ) + if not _can_resume(): + append(inbox_dir, message) + _write_messages() + return AttemptResult( + outcome="session-error", + tree_rejected=True, + baseline=baseline, + session=session, + note="Session cannot resume to receive the scope refusal.", + run_seed=run_seed, + panel_transcript="\n\n".join(panel_sections), + panel_rounds=panel_reads, + ) + failed = _resume(message, allow_checkpoint=False) + if failed is not None: + return failed + continue + if read_error: return AttemptResult( outcome="session-error", baseline=baseline, session=session, - note=f"unhonorable syscall request: {exc}", + note=read_error, + tree_rejected=tree_rejected, ) - if request is None: - break if request.withdraw and not request.problem: problem = ( on_withdraw(request.withdraw) @@ -1638,9 +1742,10 @@ def _not_run_note(request: SyscallRequest | None) -> str: if request.report and on_replies is not None: on_replies(({"to": "thread", "text": request.report, "reply_to": None},)) session = dc_replace(session, final_text=request.report) - return on_stop(session) + return dc_replace(on_stop(session), tree_rejected=tree_rejected) return AttemptResult( outcome="no-improvement", + tree_rejected=tree_rejected, session=session, note=(failed_gate[2].note or failed_gate[2].outcome) if failed_gate @@ -1648,18 +1753,6 @@ def _not_run_note(request: SyscallRequest | None) -> str: ) if not request.sleep: break - no_backend = ( - "sleep is not available here: this run has no compute backend for " - "launches; end your leg instead" - if launcher is None and not (on_stop and request.submit) - else "" - ) - preflight = ( - submit_preflight() - if request.submit and not no_backend and submit_preflight is not None - else SubmitPreflight("ready") - ) - stale_submit = preflight.status in ("stale", "outdated-pin") if stale_submit: # Budget only the effective checkpoint, never the rejected compute. request = SyscallRequest(launches=()) @@ -1704,50 +1797,6 @@ def _not_run_note(request: SyscallRequest | None) -> str: request, gpus=bench.gpus, max_concurrent_gpus=contract.budgets.max_concurrent_gpus ) if not problem: - # Admission precedes every seal, dispatch and budget charge, including - # a stale submit's jobless checkpoint. Keep the measurement backstop. - measured = tuple(changed_paths()) - violations = scope_validator(list(measured), contract) - if violations: - where = ( - " at checkpoint" if stale_submit else "" if request.submit else " at launch" - ) - problem = f"out-of-scope paths{where}: {', '.join(sorted(violations)[:10])}" - if scope_refused_once or not _can_resume(): - return AttemptResult( - outcome="scope-violation", - baseline=baseline, - session=session, - note=problem, - run_seed=run_seed, - panel_transcript="\n\n".join(panel_sections), - panel_rounds=panel_reads, - ) - scope_refused_once = True - presealed = "" - failed = _resume( - Message( - 0, - "note", - "kernel", - inbox_thread, - time.time(), - f"refusal:{session.session_id}:{inbox_seq}", - { - "text": ( - "Your syscall request was REFUSED. Nothing was snapshotted, " - "launched, measured or charged. The request can be made again " - "once the tree only changes paths the contract allows." - ), - "quoted_text": problem, - }, - origin=inbox_dir.name, - ), - allow_checkpoint=False, - ) - if failed is not None: - return failed - continue if stale_submit: sha = snapshot() sleeps_used += 1 @@ -1821,7 +1870,7 @@ def _not_run_note(request: SyscallRequest | None) -> str: # the walltime THIS submit declares (else the contract's): # a resubmit without a declaration reverts to the default, # never inheriting a prior park's. - scope_refused_once = False + tree_rejected = False submitted = request sleeps_used += 1 evals_charge = evals_gpu_hours( @@ -1896,7 +1945,14 @@ def _not_run_note(request: SyscallRequest | None) -> str: on_meter(launches_used, sleeps_used, gpu_hours_used) if submitted is None: if on_stop is not None: - return on_stop(session) + return dc_replace(on_stop(session), tree_rejected=tree_rejected) + if tree_rejected: + return AttemptResult( + outcome="no-improvement", + session=session, + tree_rejected=True, + note="ended without a submit", + ) if launcher is not None and getattr(harness, "supports_resume", True): return AttemptResult( outcome="no-improvement", session=session, note="ended without a submit" @@ -1912,6 +1968,7 @@ def _not_run_note(request: SyscallRequest | None) -> str: if violations: return AttemptResult( outcome="scope-violation", + tree_rejected=True, baseline=baseline, session=session, note=f"out-of-scope paths: {', '.join(sorted(violations)[:10])}", diff --git a/tests/test_attempt.py b/tests/test_attempt.py index 6db48cfc..89ff676e 100644 --- a/tests/test_attempt.py +++ b/tests/test_attempt.py @@ -7459,3 +7459,165 @@ def recovered(self, brief, *args, **kwargs): if not wake: assert "try a new move" in seen_briefs[0] assert load_record(root, "tsp-1").ending != "aborted" + + +@pytest.mark.parametrize("entry", ["fresh", "wake"]) +@pytest.mark.parametrize( + "ending", ["end", "explicit-end", "outage", "timeout", "budget", "error", "repeat", "crash"] +) +def test_scope_rejected_tree_never_reaches_active_line( + tmp_path, target_repo_lines, monkeypatch, entry, ending +): + import json + from dataclasses import replace + + from outerloop.endpoints import EndpointUnavailable + from outerloop.roles import author_spec + + state = tmp_path / "state" + run_id = "scope-line" + github = FakeGitHub() + dispatch = _fake_dispatch() + if entry == "wake": + with _queued_local([]): + parked = live_attempt( + config=RunConfig(target="org/pilot", benchmark="tsp"), + run_root=state, + run_id=run_id, + harness=ScriptedHarness(edits={".outerloop/syscall.json": '{"type":"sleep"}'}), + github=github, # type: ignore[arg-type] + bot_auth=NoAuth(), + now=1_000_000.0, + created="2026-08-06T00:00:00Z", + dispatch=dispatch, + ) + assert parked.outcome == "parked" + + tips = [] + + class Author: + calls = 0 + + def run(self, brief_text, workspace, resume_session_id=None): + self.calls += 1 + if self.calls == 1: + tips.append(_git(target_repo_lines, "rev-parse", "agents/agent-01").strip()) + (workspace / "rejected.txt").write_text("must never be sealed") + else: + assert self.calls <= (4 if ending == "repeat" else 2) + assert ( + "Refused:" in brief_text if self.calls == 2 else "Refused again:" in brief_text + ) + if ending == "crash": + raise RuntimeError("harness failed") + if ending == "outage": + raise EndpointUnavailable("endpoint unavailable") + if ending in ("timeout", "budget", "error"): + return replace( + ScriptedHarness(edits={}).run(brief_text, workspace), + is_error=True, + stop_reason="timeout" if ending == "timeout" else "tool_use", + error_detail="error_max_turns: Reached maximum number of turns (120)" + if ending == "budget" + else "session failed", + ) + if self.calls == 1 or (ending == "repeat" and self.calls < 4): + (workspace / ".outerloop/syscall.json").write_text( + json.dumps({"type": "sleep", "submit": True, "report": "candidate"}) + ) + elif ending == "explicit-end": + (workspace / ".outerloop/syscall.json").write_text('{"type":"end"}') + return ScriptedHarness(edits={}).run(brief_text, workspace) + + def no_snapshot(*args, **kwargs): + pytest.fail("rejected tree reached a snapshot") + + monkeypatch.setattr(climb_mod, "_push_line_snapshot", no_snapshot) + monkeypatch.setattr(climb_mod, "snapshot_tree", no_snapshot) + author = Author() + with _queued_local([]): + if entry == "fresh": + outcome = live_attempt( + config=RunConfig(target="org/pilot", benchmark="tsp"), + run_root=state, + run_id=run_id, + harness=author, + github=github, # type: ignore[arg-type] + bot_auth=NoAuth(), + now=1_000_000.0, + created="2026-08-06T00:00:00Z", + dispatch=dispatch, + ) + else: + outcome = resume_run( + state, + run_id, + dispatch=dispatch, + github=github, # type: ignore[arg-type] + bot_auth=NoAuth(), + now=1_000_100.0, + harness=author, + spec=author_spec(), + ) + assert author.calls == (4 if ending == "repeat" else 2) + assert outcome.outcome == { + "outage": "session-outage", + "timeout": "session-budget", + "budget": "session-budget", + "error": "session-error", + "crash": "session-error", + }.get(ending, "no-improvement") + assert not github.prs + assert _git(target_repo_lines, "rev-parse", "agents/agent-01").strip() == tips[0] + record = load_record(state, run_id) + assert record.state == "ended" + assert not record.stage.get("launches_used", 0) + assert not record.stage.get("gpu_hours_used", 0) + + +@pytest.mark.parametrize("submitted", [False, True]) +def test_candidate_wake_scope_backstop_never_snapshots_active_line( + tmp_path, monkeypatch, submitted +): + from dataclasses import replace + + from outerloop.dispatch import snapshot_tree + from outerloop.github import Workspace + + # A legacy park may predate admission checks. Its authoritative scope + # verdict must suppress both submitted and ordinary terminal snapshots. + state, run_id = _write_parked_candidate( + tmp_path, monkeypatch, contract=CONTRACT_LINES, agent_id="agent-01" + ) + record = load_record(state, run_id) + root = state / "runs" / run_id / "ws" + (root / "rejected.txt").write_text("legacy rejected candidate") + snap = snapshot_tree(Workspace(root=root), str(record.stage["base_sha"])) + save_record( + state, + replace( + record, + stage={ + **record.stage, + "candidate_sha": snap.commit, + "candidate_ref": snap.ref, + "submitted": submitted, + }, + ), + 1_000_050.0, + ) + + def no_snapshot(*args, **kwargs): + pytest.fail("scope backstop verdict reached a line snapshot") + + monkeypatch.setattr(climb_mod, "_push_line_snapshot", no_snapshot) + outcome = resume_run( + state, + run_id, + dispatch=_fake_dispatch(), + github=FakeGitHub(), # type: ignore[arg-type] + bot_auth=NoAuth(), + now=1_000_100.0, + ) + assert outcome.outcome == ("negative-result" if submitted else "scope-violation") + assert load_record(state, run_id).state == "ended" diff --git a/tests/test_attempt_review.py b/tests/test_attempt_review.py index 4596a90d..2bd7dd51 100644 --- a/tests/test_attempt_review.py +++ b/tests/test_attempt_review.py @@ -457,7 +457,13 @@ def run(self, brief_text, workspace, resume_session_id=None): workspace, "-c", "user.name=t", "-c", "user.email=t@t", "commit", "-qm", "edit" ) else: - assert "REFUSED" in brief_text and "out-of-scope paths" in brief_text + assert ( + "Refused:" in brief_text + if len(self.calls) == 1 + else "Refused again:" in brief_text + ) + if len(self.calls) == 3: + return super().run(brief_text, workspace, resume_session_id) assert ( main(["launch", "--name", "probe", "--minutes", "1", "--", "true"], root=workspace) == 0 @@ -475,12 +481,14 @@ def run(self, brief_text, workspace, resume_session_id=None): now=NOW, dispatch=DispatchSettings(compute=LocalCompute(), image="", account="", partition=""), ) - assert len(author.calls) == 2 - assert out.action == "scope-violation" - assert ( - "out-of-scope paths at launch: docs/roadmap.md" - in (run_dir(root, "tsp-r1") / "report.md").read_text() - ) + assert len(author.calls) == 4 + assert out.action == "replied" + from outerloop.inbox import pending + + notes = [m for m in pending(run_dir(root, "tsp-r1"), 0) if m.key.startswith("refusal:")] + assert len(notes) == 3 + assert all("docs/roadmap.md" in m.payload["text"] for m in notes) + assert all(m.payload["text"].startswith("Refused again:") for m in notes[1:]) def test_publish_review_addendum_failure_keeps_the_record(review_run, monkeypatch, caplog): @@ -1624,7 +1632,13 @@ def run(self, brief_text, workspace, resume_session_id=None): if edit_ledger: (workspace / "BENCHMARKS.md").write_text("author's ledger\n") else: - assert "REFUSED" in brief_text and "out-of-scope paths" in brief_text + assert ( + "Refused:" in brief_text + if len(self.calls) == 1 + else "Refused again:" in brief_text + ) + if len(self.calls) == 3: + return super().run(brief_text, workspace, resume_session_id) assert ( main(["launch", "--name", "probe", "--minutes", "1", "--", "true"], root=workspace) == 0 @@ -1636,8 +1650,8 @@ def run(self, brief_text, workspace, resume_session_id=None): author = FoldingHarness() outcome = wake_review(root, "tsp-r1", author, github) refused = bool(edit_ledger or rollback_to) - assert len(author.calls) == (2 if refused else 1) - assert outcome.action == ("scope-violation" if refused else "parked") + assert len(author.calls) == (4 if refused else 1) + assert outcome.action == ("replied" if refused else "parked") assert bool(launched) is not refused if rollback_to: assert any("docs/roadmap.md" in paths for paths in seen) diff --git a/tests/test_orchestrator.py b/tests/test_orchestrator.py index 8a154bfe..ed265023 100644 --- a/tests/test_orchestrator.py +++ b/tests/test_orchestrator.py @@ -635,7 +635,7 @@ def snapshot() -> str: assert result.outcome == "eval-error" and "snapshot" in result.note assert seals["n"] == 2 - harness = _SeqHarness(["the claim", "again", "still dirty"], submit_on=(1, 2, 3)) + harness = _SeqHarness(["the claim", "again", "still dirty", "abandoned"], submit_on=(1, 2, 3)) evaluator = FakeEvaluator(values=[13.9, 13.9]) measurer, snapshot2 = _wire(evaluator, tmp_path) paths = iter([["src/pilot/solvers/tsp.py"], ["docs/roadmap.md"], ["docs/roadmap.md"]]) @@ -654,7 +654,8 @@ def snapshot() -> str: launcher=lambda sha, req: "", tree_of=lambda sha: "same-tree", ) - assert result.outcome == "scope-violation" and "docs/roadmap.md" in result.note + assert result.outcome == "no-improvement" and result.tree_rejected + assert len(harness.prompts) == 4 def test_identical_resubmit_past_the_sleep_budget_ends_on_the_verdict(tmp_path: Path) -> None: @@ -2437,12 +2438,18 @@ def run(self, brief_text, workspace, resume_session_id=None): @pytest.mark.parametrize("kind", ["launch", "submit", "stale", "outdated-pin"]) -@pytest.mark.parametrize("retry", ["end", "dirty", "clean", "outage"]) -def test_scope_refusal_runs_nothing_and_resumes(tmp_path, kind, retry): +@pytest.mark.parametrize( + "retry", ["end", "explicit-end", "clean", "outage", "timeout", "budget", "error"] +) +@pytest.mark.parametrize("entry", ["fresh", "wake", "replies-only"]) +def test_scope_refusal_runs_nothing_and_resumes(tmp_path, kind, retry, entry): from outerloop.endpoints import EndpointUnavailable from outerloop.inbox import pending from outerloop.orchestrator import RunParked, SubmitPreflight + if entry == "replies-only" and retry == "clean": + pytest.skip("clean compute requests require a launcher") + # Nonzero starting meters catch both accidental charges and resets. meters: list[tuple[int, int, float]] = [] seals: list[tuple[str, ...]] = [] @@ -2462,20 +2469,33 @@ def run(self, brief_text, workspace, resume_session_id=None): self.calls += 1 if self.calls == 1: _write_syscall(workspace, request) - elif self.calls == 2: + elif self.calls == 2 or (retry == "clean" and self.calls <= 4): assert resume_session_id == "s1" - assert "REFUSED" in brief_text - assert "Nothing was snapshotted, launched, measured or charged" in brief_text - assert "once the tree only changes paths the contract allows" in brief_text + assert ( + "Refused:" in brief_text if self.calls == 2 else "Refused again:" in brief_text + ) + assert "Nothing ran and nothing was charged." in brief_text + assert "Authors may change: src/pilot/solvers/." in brief_text assert ".github/workflows/ci.yml" in brief_text assert "report-08.md" in brief_text and "report-09.md" not in brief_text assert not seals and not launches and not evaluator.calls assert meters and all(m == (1, 2, 0.25) for m in meters) if retry == "outage": raise EndpointUnavailable("unavailable") + if retry in ("timeout", "budget", "error"): + return replace( + ok_session(), + is_error=True, + stop_reason="timeout" if retry == "timeout" else "tool_use", + error_detail="error_max_turns: Reached maximum number of turns (120)" + if retry == "budget" + else "session failed", + ) + if retry == "explicit-end": + _write_syscall(workspace, {"type": "end", "report": "abandoned"}) if retry == "clean": - paths[:] = ["src/pilot/solvers/tsp.py"] - if retry in ("clean", "dirty"): + if self.calls == 4: + paths[:] = ["src/pilot/solvers/tsp.py"] _write_syscall(workspace, request) else: assert retry == "clean" and kind == "submit" @@ -2506,7 +2526,9 @@ def attempt(): inbox_dir=directory, ruler="r", changed_paths=lambda: paths, - launcher=launcher, + launcher=launcher if entry != "replies-only" else None, + on_replies=lambda replies: None, + resume_session_id="s1" if entry == "wake" else "", submit_preflight=lambda: SubmitPreflight( kind if kind in ("stale", "outdated-pin") else "ready", "tip", "main" ), @@ -2528,8 +2550,14 @@ def attempt(): assert not evaluator.calls else: result = attempt() - expected = {"dirty": "scope-violation", "outage": "session-outage"} + expected = { + "outage": "session-outage", + "timeout": "session-budget", + "budget": "session-budget", + "error": "session-error", + } assert result.outcome == expected.get(retry, "no-improvement") + assert result.tree_rejected == (retry != "clean") if retry == "clean": assert len(seals) == 1 and len(evaluator.calls) == 2 and not launches assert meters[-1] == pytest.approx((1, 3, 0.25 + 20 / 60)) @@ -2537,19 +2565,18 @@ def attempt(): assert not seals and not launches and not evaluator.calls assert all(m == (1, 2, 0.25) for m in meters) assert author.calls == 2 - if retry == "dirty": - assert ".github/workflows/ci.yml" in result.note receipts = [m for m in pending(directory, 0) if m.key.startswith("refusal:")] - assert len(receipts) == 1 + assert len(receipts) == (3 if retry == "clean" else 1) + assert all(m.payload["text"].startswith("Refused again:") for m in receipts[1:]) assert receipts[0].source == "kernel" and receipts[0].kind == "note" - assert ".github/workflows/ci.yml" in receipts[0].payload["quoted_text"] + assert ".github/workflows/ci.yml" in receipts[0].payload["text"] assert ".github/workflows/ci.yml" in (tmp_path / ".outerloop/messages.json").read_text() -def test_scope_refusal_bound_is_independent_and_resets_after_acceptance(tmp_path): +def test_repeated_scope_refusals_then_clean_request(tmp_path): from outerloop.inbox import pending - paths = ["report.md"] + paths = ["src/pilot/solvers/tsp.py"] meters: list[tuple[int, int, float]] = [] class Author: @@ -2563,16 +2590,22 @@ def run(self, brief_text, workspace, resume_session_id=None): workspace, {"launches": [{"name": str(i), "command": "x"} for i in range(4)]} ) elif self.calls in (2, 3, 4, 5): - if self.calls == 3: - assert "out-of-scope paths" in brief_text - paths[:] = ["src/pilot/solvers/tsp.py"] - elif self.calls == 4: + if self.calls == 2: paths[:] = ["report.md"] - elif self.calls == 5: - assert "out-of-scope paths" in brief_text + else: + assert ( + "Refused:" in brief_text + if self.calls == 3 + else "Refused again:" in brief_text + ) + assert "report.md. Nothing ran and nothing was charged." in brief_text + assert "Authors may change: src/pilot/solvers/." in brief_text + assert meters and all(m == (0, 0, 0.0) for m in meters) + if self.calls == 5: + paths[:] = ["src/pilot/solvers/tsp.py"] _write_syscall(workspace, {"submit": True, "report": "H: candidate"}) else: - pytest.fail("unbounded scope refusal") + assert self.calls == 6 return ok_session() harness = Author() @@ -2585,7 +2618,8 @@ def run(self, brief_text, workspace, resume_session_id=None): changed=paths, on_meter=lambda *m: meters.append(m), ) - assert result.outcome == "scope-violation" and harness.calls == 5 + assert result.outcome == "no-improvement" and harness.calls == 6 + assert not result.tree_rejected assert len(evaluator.calls) == 2 assert meters[-1] == (0, 1, 0.0) receipts = [ @@ -2593,7 +2627,89 @@ def run(self, brief_text, workspace, resume_session_id=None): for m in pending(tmp_path.parent / (tmp_path.name + "-run"), 0) if m.key.startswith("refusal:") ] - assert len(receipts) == 3 + assert len(receipts) == 4 + + +def test_scope_refusal_after_accepted_submit_still_says_again(tmp_path): + from outerloop.inbox import pending + + paths = iter([["report.md"], ["src/pilot/solvers/tsp.py"], ["report.md"]]) + harness = _SeqHarness(["dirty", "clean", "dirty again", "abandoned"], submit_on=(1, 2, 3)) + evaluator = FakeEvaluator(values=[13.9, 13.9]) + measurer, snapshot = _wire(evaluator, tmp_path) + result = attempt_once( + CONFIG, + DEEP_CONTRACT, + tmp_path, + harness, + measurer, + "base", + snapshot, + inbox_dir=tmp_path.parent / (tmp_path.name + "-run"), + ruler="r", + changed_paths=lambda: next(paths), + launcher=lambda *a: pytest.fail("launch"), + ) + assert result.outcome == "no-improvement" and result.tree_rejected + assert len(harness.prompts) == 4 and len(evaluator.calls) == 2 + notes = [ + m + for m in pending(tmp_path.parent / (tmp_path.name + "-run"), 0) + if m.key.startswith("refusal:") + ] + assert len(notes) == 2 + assert notes[0].payload["text"].startswith("Refused:") + assert notes[1].payload["text"].startswith("Refused again:") + + +@pytest.mark.parametrize("history", ["none", "legacy", "current"]) +@pytest.mark.parametrize("large_scope", ["count", "length"]) +def test_scope_refusal_bounded_scope_and_prior_run_notes(tmp_path, history, large_scope): + from outerloop.inbox import Message, append, pending + + directory = tmp_path.parent / (tmp_path.name + "-run") + if history != "none": + # Delivered notes from earlier kernels remain readable and mark a repeat. + payload = ( + { + "text": "Your syscall request was REFUSED.", + "quoted_text": "out-of-scope paths: old.md", + } + if history == "legacy" + else { + "text": ( + "Refused: this request changes paths the contract does not allow " + "authors to change: old.md." + ) + } + ) + append(directory, Message(0, "note", "kernel", "", 0, "refusal:old:0", payload)) + allowed = ( + [f"src/allowed-{i:02}/" for i in range(12)] + if large_scope == "count" + else ["src/" + "long/" * 250] + ) + contract = DEEP_CONTRACT.replace("[src/pilot/solvers/]", "[" + ", ".join(allowed) + "]") + harness = _SeqHarness(["candidate", "abandoned"], submit_on=(1,)) + result, _, evaluator = run_climb( + tmp_path, + [], + harness=harness, + contract=contract, + changed=["report.md"], + launcher=lambda *a: pytest.fail("launch"), + inbox_seq=1 if history != "none" else 0, + ) + assert result.outcome == "no-improvement" and result.tree_rejected + assert not evaluator.calls + note = pending(directory, 0)[-1].payload["text"] + assert note.startswith("Refused:" if history == "none" else "Refused again:") + scope = note.split("Authors may change: ", 1)[1] + assert len(scope) <= 1001 + assert scope.endswith("….") + if large_scope == "count": + assert ", ".join(allowed[:10]) in scope + assert allowed[10] not in scope def test_measure_and_decide_scope_backstop_is_terminal(): @@ -2618,3 +2734,101 @@ def results(self, measures): ) assert isinstance(result, AttemptResult) assert result.outcome == "scope-violation" and ".github/workflows/ci.yml" in result.note + assert result.tree_rejected + + +@pytest.mark.parametrize("problem", ["budget", "malformed"]) +def test_scope_refusals_repeat_for_requests_with_other_problems(tmp_path, problem): + from outerloop.inbox import pending + + class Author: + calls = 0 + + def run(self, brief_text, workspace, resume_session_id=None): + self.calls += 1 + assert self.calls <= 4 + if self.calls > 1: + assert ( + "Refused:" in brief_text if self.calls == 2 else "Refused again:" in brief_text + ) + if self.calls == 4: + return ok_session() + request: dict[str, object] = { + "launches": [{"name": str(i), "command": "x"} for i in range(4)] + } + if problem == "malformed": + request = {"submit": "not-a-bool"} + _write_syscall(workspace, request) + return ok_session() + + result, author, evaluator = run_climb( + tmp_path, + [], + harness=Author(), + contract=DEEP_CONTRACT, + changed=["report.md"], + launcher=lambda *a: pytest.fail("launch"), + ) + assert result.outcome == "no-improvement" and result.tree_rejected + assert author.calls == 4 and not evaluator.calls + receipts = [ + m + for m in pending(tmp_path.parent / (tmp_path.name + "-run"), 0) + if m.key.startswith("refusal:") + ] + assert len(receipts) == 3 + + +@pytest.mark.parametrize("explicit", [False, True]) +def test_scope_refusal_preserves_rejection_through_stop_callback(tmp_path, explicit): + from outerloop.orchestrator import AttemptResult + + class Author: + calls = 0 + + def run(self, brief_text, workspace, resume_session_id=None): + self.calls += 1 + if self.calls == 1: + _write_syscall(workspace, {"submit": True, "report": "candidate"}) + elif explicit: + _write_syscall(workspace, {"type": "end"}) + return ok_session() + + result, author, evaluator = run_climb( + tmp_path, + [], + harness=Author(), + changed=["report.md"], + on_replies=lambda replies: None, + on_stop=lambda session: AttemptResult(outcome="review", session=session), + ) + assert result.outcome == "review" and result.tree_rejected + assert author.calls == 2 and not evaluator.calls + + +def test_scope_admission_uses_refreshed_submit_ancestry(tmp_path): + from outerloop.orchestrator import RunParked, SubmitPreflight + + paths = ["docs/upstream.md", "src/pilot/solvers/tsp.py"] + + class Author: + def run(self, brief_text, workspace, resume_session_id=None): + _write_syscall(workspace, {"submit": True, "report": "folded upstream"}) + return ok_session() + + def preflight(): + # Fetching the new base removes upstream-owned changes from the diff. + paths[:] = ["src/pilot/solvers/tsp.py"] + return SubmitPreflight("outdated-pin", "tip", "main") + + with pytest.raises(RunParked) as caught: + run_climb( + tmp_path, + [], + harness=Author(), + changed=paths, + launcher=lambda *a: pytest.fail("launch"), + submit_preflight=preflight, + ) + assert caught.value.base_sha == "tip" + assert caught.value.sleeps_used == 1 and caught.value.launches_used == 0