Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 15 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,21 @@ Versions follow [SemVer](https://semver.org).

## [Unreleased]

- Launch ledger submissions require dispatched launch job IDs, preventing stale
checkpoints from attributing discarded launches to a commit. Gate capacity
waits still record any dispatched sibling launches. Existing ledger rows and
run records remain readable and unchanged; no schema change or backfill.

- PR measurement tables identify the base and candidate commits and the shared
eval command from the base tree; experiment tables identify launch commits.
Verify and review briefs caution against comparing numbers across commits
without checking history.
- Upgrading: no action or backfill needed; the first tick tolerates launch ledger
rows without `commit` (shown as unknown), including ended runs and in-flight
PRs. New launch records include the sealed commit; existing records and PR
bodies are not rewritten. Rollback is safe: older readers ignore the added
field.

- Hermes author resumes that exceed the replay budget stay parked with a
configuration-blocked status, retaining their session and snapshot without
consuming wake retries. Author/judge separation checks effective key paths
Expand Down
34 changes: 20 additions & 14 deletions src/outerloop/attempt.py
Original file line number Diff line number Diff line change
Expand Up @@ -723,20 +723,25 @@ def _park_run(
launch_ids = list(job_ids)
else:
launch_ids = []
ledger_launches = tuple(
dc_replace(launch, why=redact(launch.why, secrets))
for launch in parked.syscall.launches
)
_best_effort(
"launch ledger",
lambda: append_submitted(
run_dir_of(run_root, record.run_id),
sleep=parked.sleeps_used,
launches=ledger_launches,
job_ids=launch_ids,
at=now,
),
)
# Capacity waits defer gate evals, not author launches: the launcher
# returns job ids or refuses the batch. Checkpoints may carry
# discarded descriptors; gate ids cannot stand in for launch ids.
if launch_ids:
ledger_launches = tuple(
dc_replace(launch, why=redact(launch.why, secrets))
for launch in parked.syscall.launches
)
_best_effort(
"launch ledger",
lambda: append_submitted(
run_dir_of(run_root, record.run_id),
sleep=parked.sleeps_used,
launches=ledger_launches,
job_ids=launch_ids,
at=now,
commit=parked.candidate_sha,
),
)
# (the session id the wake resumes is the record's own
# resume_session_id, set below for every park — no stage duplicate)
stage["launches_used"] = parked.launches_used
Expand Down Expand Up @@ -4051,6 +4056,7 @@ def refuse(
redact_secrets=secrets,
display_digits=bench.display_digits,
experiments=experiments_rows(run_dir),
base_sha=base_sha,
)
body += f"\n\n{progress_link(config.target)}\n"
if issue_number:
Expand Down
11 changes: 10 additions & 1 deletion src/outerloop/launchlog.py
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,13 @@ def _append(run_dir: Path, rows: list[dict[str, Any]]) -> None:


def append_submitted(
run_dir: Path, *, sleep: int, launches: tuple[Launch, ...], job_ids: list[str], at: float
run_dir: Path,
*,
sleep: int,
launches: tuple[Launch, ...],
job_ids: list[str],
at: float,
commit: str = "",
) -> None:
"""One record per launch of a sleep, with the job ids it fanned out to. The
ids are positional over `launch_jobs` order, exactly as the park recorded
Expand All @@ -66,6 +72,7 @@ def append_submitted(
rows.append(
{
"event": "submitted",
"commit": commit,
"sleep": sleep,
"name": launch.name,
"why": launch.why,
Expand Down Expand Up @@ -168,6 +175,7 @@ def history(run_dir: Path) -> list[dict[str, Any]]:
"concurrency": row.get("concurrency", 0),
"job_ids": list(row.get("job_ids") or []),
"submitted_at": row.get("at"),
"commit": str(row.get("commit") or ""),
"jobs": [],
}
for row in rows:
Expand Down Expand Up @@ -207,6 +215,7 @@ def experiments_rows(run_dir: Path) -> list[dict[str, Any]]:
{
"sleep": entry.get("sleep"),
"launch": name,
"commit": entry["commit"],
"why": str(entry.get("why") or ""),
"array": array,
"concurrency": int(entry.get("concurrency") or 0),
Expand Down
14 changes: 9 additions & 5 deletions src/outerloop/orchestrator.py
Original file line number Diff line number Diff line change
Expand Up @@ -2194,8 +2194,8 @@ def _experiments_section(rows: list[dict[str, Any]]) -> list[str]:
f"{len(rows)} job(s) launched by the author this run, from the kernel's ledger; "
"the result column is the last line each job printed.",
"",
"| sleep | launch | why | job | ended | result |",
"| --- | --- | --- | --- | --- | --- |",
"| sleep | launch | commit | why | job | ended | result |",
"| --- | --- | --- | --- | --- | --- | --- |",
]
for row in rows[:MAX_EXPERIMENT_ROWS]:
pace = ""
Expand All @@ -2204,14 +2204,15 @@ def _experiments_section(rows: list[dict[str, Any]]) -> list[str]:
pace = f" (x{row['array']}, {k} at a time)"
lines.append(
f"| {row.get('sleep', '')} | {_cell(row.get('launch', ''), 48)}{pace} | "
f"{_cell(row.get('commit') or 'unknown', 7)} | "
f"{_cell(row.get('why', ''), 120)} | {_cell(row.get('job', ''), 48)} | "
f"{_ended(row)} | {_cell(row.get('result', ''), 160)} |"
)
rest = rows[MAX_EXPERIMENT_ROWS:]
if rest:
ok = sum(1 for r in rest if r.get("back") and r.get("exit_code") == 0)
lines.append(
f"| | … {len(rest)} more job(s): {ok} exit 0, {len(rest) - ok} otherwise | | | | |"
f"| | … {len(rest)} more job(s): {ok} exit 0, {len(rest) - ok} otherwise | | | | | |"
)
return lines

Expand All @@ -2222,6 +2223,8 @@ def pr_body(
redact_secrets: tuple[str, ...],
display_digits: int | None = None,
experiments: list[dict[str, Any]] | None = None,
*,
base_sha: str = "",
) -> str:
"""The PR body for an improved run: the author's report, the experiments
the run actually ran (from the launch ledger), the measured table, and
Expand Down Expand Up @@ -2312,8 +2315,9 @@ def pr_body(
f"| candidate | {fmt_metric(result.candidate, display_digits)} |",
*suite_lines,
"",
"Both numbers were measured by the orchestrator re-running the "
"contract's eval command — not taken from the session. CI "
f"Base `{base_sha[:7] or 'unknown'}` and candidate "
f"`{result.candidate_sha[:7] or 'unknown'}` were both measured by the orchestrator "
"using the same eval command read from the base tree — not taken from the session. CI "
"re-verifies independently.",
*panel_section,
]
Expand Down
8 changes: 7 additions & 1 deletion src/outerloop/review.py
Original file line number Diff line number Diff line change
Expand Up @@ -361,6 +361,12 @@ def build_agent_brief(
)


MEASUREMENT_PROVENANCE_CAUTION = (
"Numbers measured on different commits are not directly comparable; check the commits "
"in the checkout history before attributing a gap between them."
)


def build_prompt(pr: PullRequest, today: str | None = None) -> str:
diff = pr.diff
truncated = ""
Expand All @@ -374,7 +380,7 @@ def build_prompt(pr: PullRequest, today: str | None = None) -> str:
header += f"Repository: {pr.repo} — PR #{pr.number} by {pr.author}\n\n"
diff_fence = _fence(diff)
return (
header + f"Pull request: {pr.title}\n\n"
header + MEASUREMENT_PROVENANCE_CAUTION + "\n\n" + f"Pull request: {pr.title}\n\n"
f"Description:\n{pr.body or '(none)'}\n\n"
f"Diff:\n{diff_fence}diff\n{diff}\n{diff_fence}{truncated}"
)
Expand Down
2 changes: 2 additions & 0 deletions src/outerloop/verifier.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@
MAX_DIFF_CHARS,
MAX_FINDINGS,
MAX_SUMMARY_CHARS,
MEASUREMENT_PROVENANCE_CAUTION,
OPT_OUT_LABEL,
PLAIN_STYLE,
Finding,
Expand Down Expand Up @@ -289,6 +290,7 @@ def build_verify_prompt(
for author, body in thread[-MAX_THREAD_COMMENTS:]:
safe_author = " ".join(str(author).split()).replace("`", "")[:100]
parts.append(f"### {safe_author}\n{_fenced(body[:MAX_THREAD_COMMENT_CHARS])}")
parts.append(MEASUREMENT_PROVENANCE_CAUTION)
parts.append("## The change (diff)\n" + _fenced(pr.diff[:MAX_DIFF_CHARS]))
return "\n\n".join(parts)

Expand Down
2 changes: 2 additions & 0 deletions tests/fixtures/launches-before-provenance.jsonl
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
{"event":"submitted","sleep":1,"name":"probe","why":"test idea","minutes":5,"array":1,"concurrency":0,"job_ids":["1"],"at":10.0}
{"event":"ended","sleep":1,"name":"probe","exit_code":0,"state":"","elapsed":60,"last_line":"4096","at":20.0}
51 changes: 51 additions & 0 deletions tests/test_attempt.py
Original file line number Diff line number Diff line change
Expand Up @@ -90,6 +90,7 @@ def test_park_run_appends_the_launch_ledger(tmp_path) -> None:
ledger_dir = tmp_path / "runs" / "tsp-7"
entries = history(ledger_dir)
assert [(e["name"], e["job_ids"]) for e in entries] == [("a", ["201"]), ("sw", ["202", "203"])]
assert [e["commit"] for e in entries] == ["c" * 40, "c" * 40]
assert entries[0]["why"] == "probe a" and entries[0]["sleep"] == 1 and entries[0]["jobs"] == []
assert why_by_job(ledger_dir)["203"] == {
"name": "sw",
Expand Down Expand Up @@ -117,6 +118,53 @@ def test_park_run_appends_the_launch_ledger(tmp_path) -> None:
]


@pytest.mark.parametrize(
("phase", "afterany", "launch_afterany", "capacity_wait", "expected_ids"),
[
pytest.param("author-sleep", "", "", False, [], id="stale-checkpoint"),
pytest.param("author-sleep", "afterany:501", "", False, ["501"], id="sleep"),
pytest.param(
"candidate", "afterany:101:501", "afterany:501", False, ["501"], id="submit-gate"
),
pytest.param("candidate", "", "", True, [], id="capacity-no-launch"),
pytest.param(
"candidate", "afterany:501", "afterany:501", True, ["501"], id="capacity-with-launch"
),
pytest.param("candidate", "afterany:101", "", False, [], id="gate-only"),
],
)
def test_park_run_records_only_dispatched_launches(
tmp_path, phase, afterany, launch_afterany, capacity_wait, expected_ids
):
from outerloop.launchlog import read_ledger
from outerloop.syscall import Launch, SyscallRequest

record = RunRecord(run_id="test", target="org/pilot", task_title="t", state="running")
# Retain a descriptor at the stale checkpoint to exercise the ledger
# boundary independently of the kernel's discarded-request normalization.
parked = RunParked(
phase=phase,
afterany=afterany,
launch_afterany=launch_afterany,
capacity_wait=capacity_wait,
base_sha="base",
candidate_sha="sealed",
seed=1,
suite_seed=0,
syscall=SyscallRequest(launches=(Launch(name="probe", command="x", minutes=5),)),
sleeps_used=1,
)
_park_run(tmp_path, record, parked, "ref", None, 1000)
rows = read_ledger(tmp_path / "runs" / "test")
if expected_ids:
assert len(rows) == 1
assert rows[0]["event"] == "submitted"
assert rows[0]["job_ids"] == expected_ids
assert rows[0]["commit"] == "sealed"
else:
assert rows == []


def test_park_run_keeps_the_submits_report_for_the_wake(tmp_path) -> None:
"""A submitted park's stage report is the author's report at submit, not the
session's last words: the wake's panel and the PR read it."""
Expand Down Expand Up @@ -908,6 +956,9 @@ def test_improvement_produces_branch_commit_and_pr(tmp_path, target_repo) -> Non
assert pr["head"] == "feat/auto/agent-01/tsp-1"
assert pr["title"] == "[agent] tsp: 13.88 -> 13.1" # 4 sig figs, not full floats
assert "measured by the orchestrator" in pr["body"]
base_commit = _git(target_repo, "rev-parse", "main").strip()
candidate_commit = _git(target_repo, "rev-parse", str(pr["head"])).strip()
assert f"Base `{base_commit[:7]}` and candidate `{candidate_commit[:7]}`" in pr["body"]
# run record went parked with the PR url
record = load_record(tmp_path / "state", "tsp-1")
assert record.state == "parked"
Expand Down
25 changes: 25 additions & 0 deletions tests/test_launchlog.py
Original file line number Diff line number Diff line change
Expand Up @@ -152,3 +152,28 @@ def test_experiments_rows_come_from_the_ledger_including_each_jobs_last_line(
]
assert rows[1]["why"] == "" and rows[1]["array"] == 2 and rows[1]["concurrency"] == 1
assert len(history(tmp_path)[0]["jobs"]) == 1 # the duplicate ended row was skipped


def test_legacy_commit_tolerance_and_retry(tmp_path: Path) -> None:
from outerloop.launchlog import experiments_rows
from outerloop.orchestrator import _experiments_section

legacy = Path(__file__).parent / "fixtures" / "launches-before-provenance.jsonl"
(tmp_path / LEDGER).write_bytes(legacy.read_bytes())
launch = (Launch(name="probe", command="x", minutes=5),)
for _ in range(2):
rows = experiments_rows(tmp_path)
assert rows[0]["commit"] == ""
assert "| unknown |" in "\n".join(_experiments_section(rows))
# Re-parking a pre-upgrade launch must not invent provenance.
append_submitted(tmp_path, sleep=1, launches=launch, job_ids=["1"], at=30, commit="a" * 40)
assert (tmp_path / LEDGER).read_bytes() == legacy.read_bytes()
# An interrupted append is skipped; retry writes the new launch once.
with (tmp_path / LEDGER).open("a") as fh:
fh.write('{"event":"submitted"')
for _ in range(2):
append_submitted(tmp_path, sleep=2, launches=launch, job_ids=["2"], at=40, commit="b" * 40)
rows = experiments_rows(tmp_path)
assert [r["commit"] for r in rows] == ["", "b" * 40]
rendered = "\n".join(_experiments_section(rows))
assert "| unknown |" in rendered and "| bbbbbbb |" in rendered
Loading
Loading