From 20670590466ef38eadfe2eae5a6a25c6ff48842e Mon Sep 17 00:00:00 2001 From: Mengye Ren Date: Mon, 28 Sep 2026 12:47:57 -0400 Subject: [PATCH 1/2] Hermes as an author backend, with a bounded resume replay Hermes was limited to judging because it could not resume a session; resume arrived later (saved-transcript rehydration) but the author allowlists were never updated. Hermes is now an author backend alongside Claude Code and Codex: init, attempt validation, tick preflight, author key selection, fresh and wake wiring, and the syscall path treat it as a peer. It always runs contained. Each wake starts a fresh Hermes invocation and replays the saved transcript, so a long-running author's prompt grew without bound. The replay now keeps the original brief and the latest results, fills the rest with the most recent turns within OUTERLOOP_HERMES_RESUME_MAX_CHARS (default 120000), and states how many turns were omitted; the saved transcript stays complete. If the brief and latest results alone exceed the budget, the run parks as blocked on configuration without consuming wake retries, and resumes once the budget is raised. Author and judge credentials are compared by resolved path and value, and init and preflight validate a Hermes author's model or endpoint, image and runtime. Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 21 +++ docs/design/research-loop-buildout.md | 5 +- docs/install.md | 59 ++++++- scripts/tick_deploy.sh | 1 + src/outerloop/attempt.py | 164 ++++++++++++------- src/outerloop/cli.py | 9 +- src/outerloop/climbboard.py | 9 +- src/outerloop/harness.py | 69 +++++++- src/outerloop/init.py | 59 ++++++- src/outerloop/role_runner.py | 10 +- src/outerloop/tick.py | 53 ++++-- tests/fixtures/hermes_resume_legacy.json | 1 + tests/test_attempt.py | 112 +++++++++++++ tests/test_default_claude_model.py | 14 +- tests/test_hermes_author.py | 195 +++++++++++++++++++++++ tests/test_hermes_harness.py | 101 ++++++++++++ tests/test_init.py | 66 +++++++- tests/test_install_harness.py | 35 +++- tests/test_tick.py | 16 +- 19 files changed, 893 insertions(+), 106 deletions(-) create mode 100644 tests/fixtures/hermes_resume_legacy.json create mode 100644 tests/test_hermes_author.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 81edb2de..c4d7a84d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,27 @@ Versions follow [SemVer](https://semver.org). ## [Unreleased] +- Hermes author resumes that exceed the replay budget stay parked with a + configuration-blocked status, retaining their session and snapshot without + consuming wake retries. Author/judge separation checks effective key paths + and credential values before constructing sessions; init rejects incomplete + Hermes configuration. +- Upgrading: no action or backfill needed; legacy records without + `stage.hermes_resume_required_chars` are unblocked. The first oversized wake + records the required budget; raising `OUTERLOOP_HERMES_RESUME_MAX_CHARS` lets + the next tick or wake resume. Before rollback, resolve blocked runs: older + kernels ignore this optional field and may consume retries or abort them. + +- Hermes is an author peer: init, native provider/endpoint validation, contained + fresh and resumed sessions, absolute syscall commands, and separate author keys. + Resume replay preserves the original brief and latest results within + `OUTERLOOP_HERMES_RESUME_MAX_CHARS` (default 120000), with explicit omission counts. +- Upgrading: no backfill; existing records and full saved transcripts remain + readable. The first Hermes wake applies the replay bound. Finish Hermes author + runs before rollback: older kernels reject unsupported Hermes author wakes + (the endpoint-profile predecessor accepts endpoint routes only). Ended records + remain readable. See [Hermes setup and compatibility](docs/install.md). + - Endpoint profiles declare compatible APIs and use `model[endpoint=profile]` selectors, preserving native vendor model IDs. Judge credentials enforce file separation; verdicts redact the session key before posting or aggregation. diff --git a/docs/design/research-loop-buildout.md b/docs/design/research-loop-buildout.md index b08bf6a7..14868acc 100644 --- a/docs/design/research-loop-buildout.md +++ b/docs/design/research-loop-buildout.md @@ -33,7 +33,8 @@ kernel prescribes none of it. the *trigger* to the author; the plumbing underneath is this. - **Session resume** — `climb_once`'s resume-entry (`resume_session_id` + the inbox, the #129 primitives) is the wake-the-same-session - mechanism; `supports_resume` gates backends that cannot (hermes). + mechanism; `supports_resume` gates backends that cannot. Hermes resumes + through bounded saved-transcript replay. - **The composition seam** (#128) — the decide-next policy extraction. Retained as internal structure; no further orchestrator decision policies get built on it (the author decides next moves now). @@ -95,7 +96,7 @@ Build notes, not spec (spec lands with the PR): delivers (job output into the resumed session, not a gate decision). - **The harness seam.** The wake resumes via `resume_session_id` with the results (data-fenced) as the continuation. Per-backend: claude and codex - both resume; hermes (`supports_resume=False`) cannot host this. + both resume; Hermes also resumes via bounded saved-transcript replay. - **Containment unchanged.** Launched jobs run agent-directed code with eval-grade containment (the `SubprocessEvaluator` posture). diff --git a/docs/install.md b/docs/install.md index 745279be..11009867 100644 --- a/docs/install.md +++ b/docs/install.md @@ -303,7 +303,7 @@ outerloop checkout, run `bash scripts/install_claude.sh [target_path]` or The default target is `$OUTERLOOP__BIN`, else `~/.local/bin/`. Claude 2.1.272 is pinned for Linux x64 (glibc/musl) and ARM64; other platforms are refused. Installation needs `curl`, `sha256sum`, and a writable target -directory. Hermes remains a review backend, provisioned with +directory. Hermes runs authors and reviewers, provisioned with `bash scripts/install_hermes.sh [target_dir]`. **Host prerequisites for model backends.** From the outerloop checkout: @@ -312,7 +312,7 @@ directory. Hermes remains a review backend, provisioned with `bash scripts/install_claude.sh`. - Codex, as author or reviewer: the pinned Codex CLI; install with `bash scripts/install_codex.sh`. -- Hermes, as reviewer only (not an author backend): the pinned hermes-agent +- Hermes, as author or reviewer: the pinned hermes-agent source checkout and runtime; install with `bash scripts/install_hermes.sh`. `init` records the absolute Claude @@ -323,10 +323,63 @@ launch before any job runs. After installing or moving it, run `outerloop init --force` to record its path again. `--dry-run` prints the launch command without checking the CLI. For Hermes, set `REVIEW_HERMES_REPO` to the installed checkout (the installer defaults to `~/hermes-agent`). Full `init` installs a missing Hermes runtime when -`OUTERLOOP_PANEL` includes a Hermes lens or `REVIEW_BACKEND=hermes`, reading the +`OUTERLOOP_AUTHOR_BACKEND=hermes`, `OUTERLOOP_PANEL` includes a Hermes lens, or `REVIEW_BACKEND=hermes`, reading the shell or existing `.env`, and records `REVIEW_HERMES_REPO`. `--no-install-harness` skips this installation too. +To use Hermes as the author, set `OUTERLOOP_AUTHOR_BACKEND=hermes`, +`OUTERLOOP_AUTHOR_MODEL` to your provider's model ID, `REVIEW_HERMES_REPO` +to the installed pinned checkout, and `OUTERLOOP_IMAGE` to the agent container. +Set `REVIEW_HERMES_PROVIDER=openai` or `openrouter`, or select an +[endpoint profile](endpoints.md) with `OUTERLOOP_AUTHOR_ENDPOINT`. +Native provider credentials come from `OUTERLOOP_HERMES_KEY_FILE` (default +`~/.config/outerloop/hermes_key`); endpoint credentials come from the profile. +Author and judge keys must be separate. The author gets file and terminal +tools; the kernel owns branches, commits, sleep/wake, and submission as for +other backends. + +Hermes resumes from the saved transcript in the per-run home. Keep that home +until the run ends. `OUTERLOOP_HERMES_RESUME_MAX_CHARS` (default `120000`, a +positive character count) bounds the entire replay brief, including new results. +The kernel preserves the original brief and the latest results verbatim, keeps +a contiguous tail of recent messages that fits, and reports the number of +omitted turns (individual user/assistant messages). If the original brief and +latest results alone exceed the budget, resume fails explicitly; increase the +setting before retrying. The saved transcript remains complete and unchanged +in format; omission affects only the prompt sent on that wake. + +The pinned Hermes version has native compression (`compression.enabled=true`, +threshold `0.50`, floored at `0.75` below 512K context; +`cli-config.yaml.example:631` and `:663`, implemented +in `agent/context_compressor.py`; defaults parsed in `agent/agent_init.py:1478`). +It remains enabled for each invocation. +However, `run_agent.py:1558` starts a fresh `run_conversation(user_query)`; +`--save_sample` exports a trajectory, not a resumable compressed session. +Our replay is text read from a brief file, so native compression cannot bound +what the kernel replays across wakes. The kernel limit above handles that. +Hermes sample output provides assistant turn counts but no dollar usage; +`SessionResult.cost_usd` remains zero, so use provider-side spend limits for +Hermes billing. Kernel execution, turn and walltime limits still apply. + +Upgrade compatibility: no record or transcript migration is needed. Existing +Claude/Codex records and Hermes judge transcripts remain readable, including +records with absent legacy author fields. The first wake applies the replay +limit without rewriting old turns. New runs may record `author_backend=hermes`. +Kernels predating Hermes author support reject those author wakes; the +endpoint-profile predecessor supports endpoint Hermes wakes but rejects native +provider Hermes authors. Finish Hermes author runs before rollback. Ended +records remain readable; no backfill is required. + +Hermes resume configuration blocks reuse the existing `parked` run state. +The optional `stage.hermes_resume_required_chars` field records the minimum replay +budget after an oversized wake; the session ID, snapshot reference, and pending +inbox stay intact. Tick logs and live run status show `configuration-blocked` until +`OUTERLOOP_HERMES_RESUME_MAX_CHARS` reaches that value, then the next tick or manual +wake retries. Legacy records (including ended records) lacking the field require +no backfill. Existing full transcripts remain readable. Resolve blocked runs +before rolling back: older kernels ignore the field and can abort or exhaust +wake retries on oversized resumes. + The Hermes installer needs `git` and `uv`. After verifying the pinned source it installs a uv-managed Python under `.runtime//python` and runs `uv sync --frozen --no-install-project` into the sibling runtime's `venv`. diff --git a/scripts/tick_deploy.sh b/scripts/tick_deploy.sh index 3cfb28a1..c7ed1b2c 100755 --- a/scripts/tick_deploy.sh +++ b/scripts/tick_deploy.sh @@ -147,6 +147,7 @@ if [ -n "$ENV_TRUSTED" ]; then OUTERLOOP_AUTHOR_ENDPOINT REVIEW_ENDPOINT REVIEW_MODEL \ OUTERLOOP_AUTHOR_BACKEND OUTERLOOP_AUTHOR_MODEL OUTERLOOP_CLAUDE_MODEL \ OUTERLOOP_CLAUDE_BIN OUTERLOOP_CODEX_BIN OUTERLOOP_CODEX_KEY_FILE \ + OUTERLOOP_HERMES_KEY_FILE OUTERLOOP_HERMES_RESUME_MAX_CHARS \ OUTERLOOP_CLAUDE_KEY_FILE OUTERLOOP_STEWARD_KEY_FILE \ OUTERLOOP_VERTEX_PROJECT OUTERLOOP_VERTEX_REGION \ OUTERLOOP_VERTEX_ADC OUTERLOOP_VERTEX_SMALL_MODEL \ diff --git a/src/outerloop/attempt.py b/src/outerloop/attempt.py index 4dc0e378..5377c362 100644 --- a/src/outerloop/attempt.py +++ b/src/outerloop/attempt.py @@ -20,7 +20,7 @@ import shutil import time import traceback -from collections.abc import Callable, Iterable +from collections.abc import Callable, Iterable, Mapping from dataclasses import asdict, dataclass from dataclasses import replace as dc_replace from functools import partial @@ -59,6 +59,7 @@ from outerloop.harness import ( ClaudeModelUnset, Harness, + ResumeContextBlocked, SessionResult, default_binary, default_claude_model, @@ -163,41 +164,57 @@ def resolve_author_key_file(backend: str, explicit: str = "") -> str: always ~-expanded, so every caller gets a real path (an env value like "~/.config/..." must not reach the token provider verbatim).""" if not explicit: - if backend == "codex": - explicit = os.environ.get("OUTERLOOP_CODEX_KEY_FILE") or CODEX_KEY_DEFAULT - else: - explicit = os.environ.get("OUTERLOOP_CLAUDE_KEY_FILE") or CLAUDE_KEY_DEFAULT + defaults = {"claude": CLAUDE_KEY_DEFAULT, "codex": CODEX_KEY_DEFAULT} + explicit = os.environ.get(f"OUTERLOOP_{backend.upper()}_KEY_FILE") or defaults.get( + backend, str(CONFIG_DIR / f"{backend}_key") + ) return os.path.expanduser(explicit) -def codex_author_config_error(backend: str, model: str, image: str) -> str: - """Why a codex author would die at startup ("" when it won't). Validates the - EFFECTIVE (backend, model) — the fresh climb passes args; a wake - passes the PARKED RUN's persisted pair — so backend and model are checked as - a unit and never a fleet backend against a run's model. codex writes+executes, - so it must be contained (--image) and needs a non-claude model.""" +def author_config_error( + backend: str, + model: str, + image: str, + *, + environ: Mapping[str, str] | None = None, +) -> str: + """Validate the effective author configuration before claiming or waking a run.""" + env = os.environ if environ is None else environ + if backend not in ("claude", "codex", "hermes"): + return f"unknown author backend {backend!r} (expected 'claude', 'codex' or 'hermes')" + if backend == "hermes": + from outerloop.harness import hermes_resume_max_chars + from outerloop.hermes_install import hermes_ready + + repo = env.get("REVIEW_HERMES_REPO", "") + if not repo or not hermes_ready(Path(repo).expanduser()): + return "hermes author needs REVIEW_HERMES_REPO with pinned source and runtime" + if not image: + return "author-backend hermes requires --image (it runs contained)" + try: + hermes_resume_max_chars(env) + except ValueError as exc: + return str(exc) try: - _, profile = resolve_endpoint(model, backend) + _, profile = resolve_endpoint(model, backend, environ=env) except ValueError as exc: return str(exc) if profile: if backend != "claude" and not image: return f"author-backend {backend} requires --image (it runs contained)" - if backend == "hermes": - from outerloop.hermes_install import hermes_ready - - repo = os.environ.get("REVIEW_HERMES_REPO", "") - if not repo or not hermes_ready(Path(repo).expanduser()): - return "hermes author needs REVIEW_HERMES_REPO with pinned source and runtime" return "" if backend == "hermes": - return "hermes author requires OUTERLOOP_AUTHOR_ENDPOINT" - if backend not in ("claude", "codex"): - # a typo'd OUTERLOOP_AUTHOR_BACKEND passes the env DEFAULT silently - # (argparse validates the flag, not its default) and the climb rejects it - # at build_harness — catch it on the tick host so a claimed intake - # issue never strands on it - return f"unknown author backend {backend!r} (expected 'claude' or 'codex')" + from outerloop.role_runner import HERMES_PROVIDERS + + if not model: + return "hermes author needs OUTERLOOP_AUTHOR_MODEL" + provider = env.get("REVIEW_HERMES_PROVIDER", "") + if provider not in HERMES_PROVIDERS: + return ( + "hermes author needs REVIEW_HERMES_PROVIDER (openai or openrouter) " + "or an endpoint profile" + ) + return "" if backend == "claude": # symmetric to the codex check: a claude harness 404s on a non-claude # model (e.g. OUTERLOOP_AUTHOR_MODEL left on a codex id while the @@ -215,11 +232,15 @@ def codex_author_config_error(backend: str, model: str, image: str) -> str: return "" +# Compatibility for callers of the former backend-specific validator. +codex_author_config_error = author_config_error + + def fleet_author_model(backend: str) -> str: """The fleet author's model from the environment: OUTERLOOP_AUTHOR_MODEL, else the deployment's Claude model for a claude author (ClaudeModelUnset when that is missing too). Another backend without OUTERLOOP_AUTHOR_MODEL gets "", and - codex_author_config_error names the fix.""" + author_config_error names the fix.""" model = author_model_setting(backend, os.environ.get("OUTERLOOP_AUTHOR_MODEL", "")) if not model and backend == "claude": return default_claude_model() @@ -1686,6 +1707,20 @@ def changed_paths() -> list[str]: if sleep_ref: drop_snapshot(ws, Snapshot(commit="", tree="", ref=sleep_ref)) return AttemptOutcome(run_id=run_id, outcome="parked") + except ResumeContextBlocked as exc: + latest = load_record(run_root, run_id) + save_record( + run_root, + dc_replace( + latest, + state=PARKED, + stage={**latest.stage, "hermes_resume_required_chars": exc.required_chars}, + wake_attempts=max(0, latest.wake_attempts - 1), + ), + now, + ) + log.warning("run %s: %s", run_id, exc) + return AttemptOutcome(run_id=run_id, outcome="parked", pr_url=latest.pr_url) except ScopeHistoryError as exc: return _end_refused_wake(run_root, record, exc, now, secrets, ws.auth) finally: @@ -3132,18 +3167,19 @@ def _judge_lens_key( "(role separation: the judge's own key, never the author's)" ) path = Path(raw).expanduser() - author_path = Path(resolve_author_key_file(author_backend)).expanduser() - if path.resolve() == author_path.resolve(): - raise ValueError( - f"{backend} panel key file {path} is the {author_backend} author key " - "(role separation: the judge needs its own key)" - ) + for author in dict.fromkeys((author_backend, "claude", "codex", "hermes")): + author_path = Path(resolve_author_key_file(author)).expanduser() + if path.resolve() == author_path.resolve(): + raise ValueError( + f"{backend} panel key file {path} is the {author} author key " + "(role separation: the judge needs its own key)" + ) if path.resolve() == claude_panel_path.resolve(): raise ValueError( f"{backend} panel key file {path} is the claude panel key file " "(an anthropic key must never reach another provider's login)" ) - return role_key(raw, author_backend) + return role_key(raw, backend) def _panel_lenses_from_args( @@ -3170,6 +3206,13 @@ def _panel_lenses_from_args( if author_model is None: author_model = getattr(args, "model", "") or "" parsed = resolve_lenses(args.panel, author_backend, author_model) + _, author_profile = resolve_endpoint(author_model, author_backend) + author_path = ( + author_profile.key_file + if author_profile + else Path(resolve_author_key_file(author_backend, getattr(args, "key_file", ""))) + ) + prepared = [] # the anthropic panel key is read only when a claude lens will use it — # a codex-only panel must not demand an unrelated credential panel_key = ( @@ -3179,8 +3222,8 @@ def _panel_lenses_from_args( ) lenses = [] secrets: list[str] = [panel_key] if panel_key else [] + hermes_repo_env = os.environ.get("REVIEW_HERMES_REPO", "").strip() for kind, backend, model in parsed: - hermes_repo_env = os.environ.get("REVIEW_HERMES_REPO", "").strip() # per-backend judge keys coexist — a codex lens is never handed the # anthropic panel key, and role separation forbids defaulting to the # AUTHOR's codex key: the judge key is its own, named explicitly @@ -3191,26 +3234,14 @@ def _panel_lenses_from_args( raise ValueError(f"a {backend} endpoint panel lens requires --image") from outerloop.endpoints import validate_judge_key_file - _, author_profile = resolve_endpoint(author_model, author_backend) - validate_judge_key_file( - endpoint, - author_profile.key_file - if author_profile - else (getattr(args, "key_file", "") or resolve_author_key_file(author_backend)), - claude_panel_path, - ) + validate_judge_key_file(endpoint, author_path, claude_panel_path) lens_key = endpoint.key() - author_key_file = getattr(args, "key_file", "") or resolve_author_key_file( - author_backend - ) - if lens_key == model_key(author_key_file, author_backend, author_model): - raise ValueError("a panel judge key is the author key (role separation)") secrets.append(lens_key) elif backend == "codex": lens_key = _judge_lens_key( backend="codex", key_file_env="OUTERLOOP_PANEL_CODEX_KEY_FILE", - author_backend="codex", + author_backend=author_backend, claude_panel_path=claude_panel_path, image=args.image, ) @@ -3219,12 +3250,11 @@ def _panel_lenses_from_args( elif backend == "hermes": # hermes reads its key from its provider's env var, but the FILE # is resolved and separated exactly like codex's (the key still - # lands next to the session). The author's OpenAI key coexists, so - # separate against the codex author key. + # lands next to the session). Separate against the active author. lens_key = _judge_lens_key( backend="hermes", key_file_env="OUTERLOOP_PANEL_HERMES_KEY_FILE", - author_backend="codex", + author_backend=author_backend, claude_panel_path=claude_panel_path, image=args.image, ) @@ -3232,6 +3262,22 @@ def _panel_lenses_from_args( secrets.append(lens_key) else: lens_key = panel_key + lens_path = ( + endpoint.key_file + if endpoint + else Path(os.environ[f"OUTERLOOP_PANEL_{backend.upper()}_KEY_FILE"]).expanduser() + if backend in ("codex", "hermes") + else claude_panel_path + ) + if lens_path.resolve() == author_path.expanduser().resolve() or ( + lens_key and lens_key == model_key(author_path, author_backend, author_model) + ): + raise ValueError( + "a panel judge key is the effective author key " + "(role separation: the judge needs its own key)" + ) + prepared.append((kind, backend, model, lens_key)) + for kind, backend, model, lens_key in prepared: try: if not args.image: log.warning( @@ -3256,9 +3302,6 @@ def _panel_lenses_from_args( except (ValueError, ClaudeModelUnset) as exc: raise ValueError(f"panel entry {kind}:{backend}: {exc}") from exc lenses.append(PanelLens(kind=kind, harness=judge)) - _, author_endpoint = resolve_endpoint(author_model, author_backend) - if author_endpoint and author_endpoint.key() in secrets: - raise ValueError("a panel judge key is the author key (role separation)") return tuple(lenses), tuple(dict.fromkeys(secrets)) @@ -4868,7 +4911,7 @@ def _run_id(value: str) -> str: choices=("claude", "codex", "hermes"), default=os.environ.get("OUTERLOOP_AUTHOR_BACKEND") or "claude", help="agent backend for the author/editor role (config-driven: default " - "from OUTERLOOP_AUTHOR_BACKEND). codex runs contained (apptainer + " + "from OUTERLOOP_AUTHOR_BACKEND). codex and hermes run contained (apptainer + " "--sandbox danger-full-access) and REQUIRES --image and a codex/openai " "--model (e.g. gpt-5.6-terra).", ) @@ -4914,7 +4957,7 @@ def _run_id(value: str) -> str: "--key-file", default="", help="author key file; default resolves per backend (config-driven): " - "OUTERLOOP_CLAUDE_KEY_FILE for claude, OUTERLOOP_CODEX_KEY_FILE for codex", + "OUTERLOOP__KEY_FILE for native backends", ) parser.add_argument("--issue", type=int, default=0) parser.add_argument( @@ -4997,13 +5040,14 @@ def _run_id(value: str) -> str: # an explicit --key-file still overrides (a manual re-run pinning a key) if args.key_file: wake_key_file = os.path.expanduser(args.key_file) - _err = codex_author_config_error(wake_backend, wake_model, args.image) + _err = author_config_error(wake_backend, wake_model, args.image) if _err: # this wake job HOLDS the run's lease (transferred on dispatch); release # it before exiting so a misconfig doesn't strand the run until the TTL # reap (the resume_run finally below only runs once we reach it) _release_own_lease(args.run_root, args.resume) parser.error(f"parked run {args.resume}: {_err}") + args.key_file = wake_key_file # the wake runs the SAME verification panel as a fresh climb, so a # dispatched improvement is not published unverified. try: @@ -5054,6 +5098,7 @@ def _run_id(value: str) -> str: model=wake_model, container_image=args.image, codex_extra_args=codex_extra, + hermes_provider=os.environ.get("REVIEW_HERMES_PROVIDER", ""), hermes_repo=Path(os.environ["REVIEW_HERMES_REPO"]) if wake_backend == "hermes" else None, @@ -5109,10 +5154,10 @@ def _run_id(value: str) -> str: args.model = author_model_setting(args.author_backend, args.model) except ValueError as exc: parser.error(str(exc)) - _err = codex_author_config_error(args.author_backend, args.model, args.image) + _err = author_config_error(args.author_backend, args.model, args.image) if _err: parser.error(_err) - # config-driven: the author key defaults per backend (claude vs codex) so the + # The author key defaults per backend so the # tick never threads it — see resolve_author_key_file (result is ~-expanded). _, author_endpoint = resolve_endpoint(args.model, args.author_backend) args.key_file = ( @@ -5192,6 +5237,7 @@ def _run_id(value: str) -> str: model=args.model, container_image=args.image, codex_extra_args=codex_extra, + hermes_provider=os.environ.get("REVIEW_HERMES_PROVIDER", ""), hermes_repo=Path(os.environ["REVIEW_HERMES_REPO"]) if args.author_backend == "hermes" else None, diff --git a/src/outerloop/cli.py b/src/outerloop/cli.py index 04c2671d..9199f244 100644 --- a/src/outerloop/cli.py +++ b/src/outerloop/cli.py @@ -69,6 +69,8 @@ "OUTERLOOP_CLAUDE_BIN", "OUTERLOOP_CODEX_BIN", "OUTERLOOP_CODEX_KEY_FILE", + "OUTERLOOP_HERMES_KEY_FILE", + "OUTERLOOP_HERMES_RESUME_MAX_CHARS", "OUTERLOOP_CLAUDE_KEY_FILE", "OUTERLOOP_STEWARD_KEY_FILE", "OUTERLOOP_VERTEX_PROJECT", @@ -415,12 +417,7 @@ def missing_harness_binary(values: Mapping[str, str], environ: Mapping[str, str] else "hermes author needs REVIEW_HERMES_REPO with pinned source and runtime" ) if key not in HARNESS_BIN_KEYS: - hint = ( - f" Hermes is a review backend; install its source with `{HARNESS_INSTALL['hermes']}`." - if backend == "hermes" - else "" - ) - return f"unsupported author backend {backend!r}; choose claude or codex.{hint}" + return f"unsupported author backend {backend!r}; choose claude, codex or hermes." env = {**values, **environ} recorded = env.get(key, "") binary = default_binary(backend, env) diff --git a/src/outerloop/climbboard.py b/src/outerloop/climbboard.py index 9a687d9b..75fe094a 100644 --- a/src/outerloop/climbboard.py +++ b/src/outerloop/climbboard.py @@ -1088,7 +1088,10 @@ def collect_status( if record.target != target or record.state not in _LIVE_STATES: continue stage = record.stage or {} - note = str(stage.get("report") or "") + from outerloop.harness import resume_config_block + + blocked = resume_config_block(stage) + note = blocked or str(stage.get("report") or "") hyp = report_hypothesis(note) or str(stage.get("hypothesis") or "")[:MAX_HYPOTHESIS_CHARS] exp_done, exp_total, exp_minutes = _experiment_progress(root, record) depth_k, sleep_k, bench_minutes = budgets.get(record.benchmark, (None, None, 0)) @@ -1104,9 +1107,9 @@ def collect_status( "agent": record.agent_id, "benchmark": record.benchmark, "state": record.state, - "phase": stage.get("phase", ""), + "phase": "configuration-blocked" if blocked else stage.get("phase", ""), # the agent's own headline: what it says it is working on - "direction": _phrase(hyp or note.replace("\n", " ")), + "direction": blocked or _phrase(hyp or note.replace("\n", " ")), "hypothesis": hyp, "since": record.updated or record.created, # a run that never launched HAS used zero — absent keys must diff --git a/src/outerloop/harness.py b/src/outerloop/harness.py index 640b979b..97a702f6 100644 --- a/src/outerloop/harness.py +++ b/src/outerloop/harness.py @@ -495,6 +495,65 @@ def _save_resume_transcript( ) +def hermes_resume_max_chars(environ: Mapping[str, str] | None = None) -> int: + """Maximum rehydrated brief size, including the new results message.""" + env = os.environ if environ is None else environ + raw = env.get("OUTERLOOP_HERMES_RESUME_MAX_CHARS", "120000") + try: + value = int(raw) + except ValueError: + value = 0 + if value <= 0: + raise ValueError("OUTERLOOP_HERMES_RESUME_MAX_CHARS must be a positive integer") + return value + + +class ResumeContextBlocked(ValueError): + """Required context cannot fit; preserve the parked session for an operator.""" + + def __init__(self, required_chars: int): + self.required_chars = required_chars + super().__init__( + "configuration-blocked: original brief and latest results exceed " + "OUTERLOOP_HERMES_RESUME_MAX_CHARS; set it to at least " + f"{required_chars} and tick or wake again" + ) + + +def resume_config_block(stage: dict[str, object]) -> str: + required = int(str(stage.get("hermes_resume_required_chars", 0))) + if not required: + return "" + try: + budget = hermes_resume_max_chars() + except ValueError as exc: + return f"configuration-blocked: {exc}" + return str(ResumeContextBlocked(required)) if budget < required else "" + + +def _bounded_resume_brief(turns: list[dict[str, str]], latest: str, budget: int) -> str: + """Keep the original brief and a contiguous recent tail; never truncate results.""" + tail = len(turns) + + def render(start: int) -> str: + omitted = start - 1 + note = ( + f"[Resume context: {omitted} earlier turns omitted to fit the replay budget.]\n\n" + if omitted + else "" + ) + return f"{note}{_render_resume_transcript([turns[0], *turns[start:]])}\n\n{latest}" + + complete = render(1) + if len(complete) <= budget: + return complete + if len(render(tail)) > budget: + raise ResumeContextBlocked(min(len(complete), len(render(tail)))) + while tail > 1 and len(render(tail - 1)) <= budget: + tail -= 1 + return render(tail) + + def _render_resume_transcript(turns: list[dict[str, str]]) -> str: """Render prior turns as a readable prefix for the resume brief.""" blocks = ["=== Earlier in this session (your prior context) ==="] @@ -1392,6 +1451,8 @@ class HermesHarness: container_image: str = "" apptainer_binary: str = field(default_factory=apptainer_from_env) + resume_max_chars: int | None = None + def run( self, brief_text: str, workspace: Path, resume_session_id: str | None = None ) -> SessionResult: @@ -1428,7 +1489,13 @@ def run( detail="no saved transcript to restore this session's context", ) prior_turns = loaded - brief_to_send = f"{_render_resume_transcript(prior_turns)}\n\n{brief_text}" + brief_to_send = _bounded_resume_brief( + prior_turns, + brief_text, + self.resume_max_chars + if self.resume_max_chars is not None + else hermes_resume_max_chars(), + ) if self.provider: # minimal headless config: provider + default model, nothing else hermes_dir = session_home / ".hermes" diff --git a/src/outerloop/init.py b/src/outerloop/init.py index 0882d4f8..f3a24611 100644 --- a/src/outerloop/init.py +++ b/src/outerloop/init.py @@ -43,7 +43,7 @@ DEFAULT_PAT_FILE = CONFIG_DIR / "bot_pat" API = "https://api.github.com" # The climbing author's harnesses (attempt.py's --author-backend choices). -AUTHOR_BACKENDS = ("claude", "codex") +AUTHOR_BACKENDS = ("claude", "codex", "hermes") @dataclass @@ -59,7 +59,7 @@ class InitAnswers: author_key_file: str = "" # the author's model key file, when known image: str = "" # the agent image (OUTERLOOP_IMAGE) uncontained: bool = False # --no-image: write OUTERLOOP_IMAGE= so no image is picked up - author_bin: str = "" # the author harness binary (claude/codex), absolute, when found + author_bin: str = "" # absolute binary or Hermes source path, when ready # Existing deployment values untouched by a focused GitHub App update. preserved_env: dict[str, str] = field(default_factory=dict) @@ -143,9 +143,12 @@ def _existing_env() -> str: def author_bin_env(backend: str) -> str: - """The `.env` key naming `backend`'s harness binary (`OUTERLOOP_CLAUDE_BIN`, - `OUTERLOOP_CODEX_BIN`), the same names `attempt` reads.""" - return f"OUTERLOOP_{(backend or AUTHOR_BACKENDS[0]).upper()}_BIN" + """The `.env` key naming the backend's binary or pinned source directory.""" + return ( + "REVIEW_HERMES_REPO" + if backend == "hermes" + else f"OUTERLOOP_{(backend or AUTHOR_BACKENDS[0]).upper()}_BIN" + ) def locate_harness(backend: str) -> str: @@ -154,6 +157,13 @@ def locate_harness(backend: str) -> str: with no login PATH, would not find it); "" when absent. Recorded in .env so every job spawns the same binary the operator installed.""" name = backend or AUTHOR_BACKENDS[0] + if name == "hermes": + repo = ( + Path(os.environ.get("REVIEW_HERMES_REPO") or Path.home() / "hermes-agent") + .expanduser() + .resolve() + ) + return str(repo) if hermes_ready(repo) else "" recorded = os.environ.get(author_bin_env(name), "") if recorded: # an operator's explicit path is kept when it works and reported when it does not @@ -525,7 +535,7 @@ def _collect(args: argparse.Namespace, interactive: bool) -> tuple[InitAnswers, # (that one is about auth). When asked, offer the fixed set, not a blank. ask_author = interactive and not args.github_app backend = args.author_backend or ( - _ask("Author backend (claude or codex)", "claude") if ask_author else "" + _ask("Author backend (claude, codex or hermes)", "claude") if ask_author else "" ) model = args.author_model or ( _ask("Author model (blank = the backend's default)") if ask_author else "" @@ -840,6 +850,14 @@ def main(argv: list[str] | None = None) -> int: args = parser.parse_args(sys.argv[2:] if argv is None else argv) interactive = not args.yes + from outerloop.harness import hermes_resume_max_chars + + try: + hermes_resume_max_chars() + except ValueError as exc: + print(f"outerloop init: {exc}", file=sys.stderr) + return 2 + try: answers, pat_file = _collect(args, interactive) except StartError as exc: @@ -959,6 +977,35 @@ def main(argv: list[str] | None = None) -> int: # compute nodes, which the login node cannot speak for answers.image = ensure_image(interactive=interactive, probe=(answers.compute == "local")) + if answers.author_backend == "hermes": + from outerloop.attempt import author_config_error + from outerloop.endpoints import author_model_setting + + effective = dict(os.environ) + effective.update( + (key, value) + for key, value in (line.split("=", 1) for line in render_env(answers, "").splitlines()) + ) + image = effective.get("OUTERLOOP_IMAGE", "") + if image and not Path(image).expanduser().is_file(): + print(f"outerloop init: Hermes container image {image} is not a file", file=sys.stderr) + return 2 + try: + model = author_model_setting( + "hermes", effective.get("OUTERLOOP_AUTHOR_MODEL", ""), effective + ) + error = author_config_error( + "hermes", + model, + effective.get("OUTERLOOP_IMAGE", ""), + environ=effective, + ) + except ValueError as exc: + error = str(exc) + if error: + print(f"outerloop init: {error}; configure Hermes and rerun init", file=sys.stderr) + return 2 + # The App is the recommended credential (scoped, revocable, no plaintext # token); the PAT is the fallback. Offer it first when interactive. if not args.github_app and not pat_file and interactive: diff --git a/src/outerloop/role_runner.py b/src/outerloop/role_runner.py index a12a2540..b12d5078 100644 --- a/src/outerloop/role_runner.py +++ b/src/outerloop/role_runner.py @@ -62,7 +62,7 @@ # hermes resolves credentials per provider (a registry); "openai" maps to its # canonical `openai-api` provider id (api-key auth against api.openai.com — # plain "openai" is a provider GROUP there, not an id). -_HERMES_PROVIDERS = { +HERMES_PROVIDERS = { "openrouter": ("openrouter", "OPENROUTER_API_KEY"), "openai": ("openai-api", "OPENAI_API_KEY"), } @@ -137,12 +137,12 @@ def build_harness( if backend == "hermes": if hermes_repo is None: raise ValueError("hermes backend needs hermes_repo (the pinned clone)") - if not profile and (hermes_provider or "openrouter") not in _HERMES_PROVIDERS: + if not profile and (hermes_provider or "openrouter") not in HERMES_PROVIDERS: raise ValueError(f"unknown hermes provider: {hermes_provider!r}") seed, key_env = ( ("outerloop_endpoint", "OUTERLOOP_SESSION_KEY") if profile - else _HERMES_PROVIDERS[hermes_provider or "openrouter"] + else HERMES_PROVIDERS[hermes_provider or "openrouter"] ) # `terminal` (the shell) is keyed on the SAME signal claude uses — the # spec granting the Bash tool — not on can_execute, so every backend @@ -242,6 +242,10 @@ def run_role( from outerloop.syscall import install_tool install_tool(workspace) + if not is_judge: + from outerloop.syscall import tool_command + + brief_text = brief_text.replace("python .outerloop/syscall", tool_command(workspace)) session = harness.run(brief_text, workspace, resume_session_id) if session.is_error: return RoleResult( diff --git a/src/outerloop/tick.py b/src/outerloop/tick.py index 94c7d54c..eea7c6fe 100644 --- a/src/outerloop/tick.py +++ b/src/outerloop/tick.py @@ -1671,6 +1671,14 @@ def _sweep_one( ): return + from outerloop.harness import resume_config_block + + blocked = resume_config_block(record.stage) + if blocked: + log.warning("run %s: %s", record.run_id, blocked) + deferred.append(record.run_id) + return + # Layer 5: too many failed attempts is a terminal, reported state. if record.wake_attempts >= MAX_WAKE_ATTEMPTS: if not dry_run: @@ -1688,6 +1696,10 @@ def _sweep_one( stuck.append(record.run_id) return + if record.stage.get("hermes_resume_required_chars"): + wake(record, "resume configuration unblocked", "configuration") + return + job_ids = _poll_targets(record) if not job_ids: if messages: @@ -2407,14 +2419,14 @@ def _author_config_error(spec: ServiceSpec) -> str: (e.g. OUTERLOOP_AUTHOR_BACKEND=codex with no non-claude model) never strands a claimed intake issue. Reads the fleet author config from env — the same source the climb defaults from — and the image the tick already knows.""" - from outerloop.attempt import codex_author_config_error, fleet_author_model + from outerloop.attempt import author_config_error, fleet_author_model backend = os.environ.get("OUTERLOOP_AUTHOR_BACKEND") or "claude" try: model = fleet_author_model(backend) except (ClaudeModelUnset, ValueError) as exc: return str(exc) - return codex_author_config_error(backend, model, spec.image) + return author_config_error(backend, model, spec.image) def _panel_preflight_error(spec: ServiceSpec) -> str: @@ -2449,7 +2461,10 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: return str(exc) from outerloop.attempt import fleet_author_model from outerloop.endpoints import resolve_endpoint + from outerloop.harness import hermes_resume_max_chars + if any(backend == "hermes" for _, backend, _ in lenses): + hermes_resume_max_chars() author_backend = os.environ.get("OUTERLOOP_AUTHOR_BACKEND") or "claude" _, author_endpoint = resolve_endpoint( author_model_setting(author_backend, os.environ.get("OUTERLOOP_AUTHOR_MODEL", "")), @@ -2514,12 +2529,14 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: return ( f"{lens_backend} panel key path {key_path} is relative; only absolute paths fly" ) - author = Path(resolve_author_key_file("codex")).expanduser() - if key_path.resolve() == author.resolve(): - return ( - f"{lens_backend} panel key file {key_path} is the codex author " - "key (role separation: the judge needs its own key)" - ) + for author_backend_name in dict.fromkeys((author_backend, "claude", "codex", "hermes")): + author = Path(resolve_author_key_file(author_backend_name)).expanduser() + if key_path.resolve() == author.resolve(): + return ( + f"{lens_backend} panel key file {key_path} is the " + f"{author_backend_name} author " + "key (role separation: the judge needs its own key)" + ) claude_panel = Path(spec.panel_key_file or PANEL_KEY_DEFAULT).expanduser() if key_path.resolve() == claude_panel.resolve(): return ( @@ -2528,7 +2545,13 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: "provider's login)" ) key = FileTokenProvider(key_path).token() - if author_endpoint and key == author_endpoint.key(): + from outerloop.endpoints import model_key + + if key and key == model_key( + resolve_author_key_file(author_backend), + author_backend, + fleet_author_model(author_backend), + ): return "a panel judge key is the author key (role separation)" if lens_backend == "hermes": repo = os.environ.get("REVIEW_HERMES_REPO", "").strip() @@ -2540,13 +2563,13 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: "run bash scripts/install_hermes.sh " f"{repo or '$REVIEW_HERMES_REPO'}" ) - from outerloop.role_runner import _HERMES_PROVIDERS + from outerloop.role_runner import HERMES_PROVIDERS provider = os.environ.get("REVIEW_HERMES_PROVIDER", "").lower() or "openrouter" - if provider not in _HERMES_PROVIDERS: + if provider not in HERMES_PROVIDERS: return ( f"unknown REVIEW_HERMES_PROVIDER {provider!r} " - f"(have: {sorted(_HERMES_PROVIDERS)})" + f"(have: {sorted(HERMES_PROVIDERS)})" ) if not any(backend == "claude" for _, backend, _ in lenses): return "" # codex-only panel: the claude key checks below don't apply @@ -2583,7 +2606,9 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: from outerloop.role_runner import role_key key = role_key(path) - if author_endpoint and key == author_endpoint.key(): + from outerloop.endpoints import model_key + + if key and key == model_key(author, fleet_backend, fleet_author_model(fleet_backend)): return "a panel judge key is the author key (role separation)" return "" except Exception as exc: @@ -3394,7 +3419,7 @@ def _service_spec_from_env(root: Path) -> tuple[Any, ServiceSpec | None]: log.warning( "local mode: no container image at %s; sessions run under the harness " "sandbox and evaluations run bare on this machine; the panel is %s; a " - "codex author needs the image (docs/install.md, local mode)", + "codex or hermes author needs the image (docs/install.md, local mode)", image, "on by OUTERLOOP_PANEL_UNCONTAINED=1" if os.environ.get("OUTERLOOP_PANEL_UNCONTAINED") == "1" diff --git a/tests/fixtures/hermes_resume_legacy.json b/tests/fixtures/hermes_resume_legacy.json new file mode 100644 index 00000000..709c3098 --- /dev/null +++ b/tests/fixtures/hermes_resume_legacy.json @@ -0,0 +1 @@ +{"turns": [{"role": "user", "text": "original"}, {"role": "assistant", "text": "old reply"}]} diff --git a/tests/test_attempt.py b/tests/test_attempt.py index 157e59d8..464d0921 100644 --- a/tests/test_attempt.py +++ b/tests/test_attempt.py @@ -3845,6 +3845,10 @@ def test_codex_panel_lens_requires_the_judges_own_key(monkeypatch) -> None: def test_codex_only_panel_never_reads_the_claude_key(monkeypatch, tmp_path) -> None: + author = tmp_path / "author" + author.write_text("sk-author") + author.chmod(0o600) + monkeypatch.setenv("OUTERLOOP_CLAUDE_KEY_FILE", str(author)) # a codex-only panel must not demand the (unused) anthropic panel key import argparse @@ -3940,6 +3944,10 @@ def test_codex_panel_lens_refuses_to_run_uncontained(monkeypatch, tmp_path) -> N def test_hermes_panel_lens_shares_the_judge_key_rules(monkeypatch, tmp_path) -> None: + author = tmp_path / "author" + author.write_text("sk-author") + author.chmod(0o600) + monkeypatch.setenv("OUTERLOOP_CLAUDE_KEY_FILE", str(author)) # hermes joins the shelled-judge rules via the SAME helper as codex: # image required, own key named, never the author's or the claude panel's import argparse @@ -7178,3 +7186,107 @@ def head_contains(self, target, base, head): ws = Workspace(root=tmp_path) assert _contains_tip(ws, "tip", "measured", cast(Any, GitHub()), "o/r") is bool(answer) assert not _contains_tip(ws, "tip", "measured") + + +@pytest.mark.parametrize("pr", [False, True]) +def test_hermes_resume_configuration_block_preserves_park(tmp_path, monkeypatch, caplog, pr): + from dataclasses import replace + + from outerloop.climbboard import collect_status + from outerloop.harness import HermesHarness, _save_resume_transcript + from outerloop.roles import author_spec + from outerloop.tick import _sweep_one + + state, run_id, wsroot, _ = _write_parked_author_sleep(tmp_path, monkeypatch) + record = load_record(state, run_id) + record = replace( + record, + author_backend="hermes", + author_model="gpt-native", + wake_attempts=1, + pr_url="https://github.com/org/pilot/pull/1" if pr else "", + deadline=1_000_001, + ) + save_record(state, record, 1_000_000) + home = wsroot.parent / f"{wsroot.name}-home" + home.mkdir() + _save_resume_transcript( + home, + "s1", + [ + {"role": "user", "text": "original brief"}, + {"role": "assistant", "text": "previous reply"}, + ], + ) + monkeypatch.setattr("outerloop.harness.hermes_ready", lambda repo: True) + monkeypatch.setenv("OUTERLOOP_HERMES_RESUME_MAX_CHARS", "10") + kwargs = dict( + dispatch=_fake_dispatch(), + github=CommentingGitHub(), + bot_auth=NoAuth(), + now=1_000_100.0, + spec=author_spec(), + ) + for _ in range(2): + outcome = resume_run( + state, + run_id, + harness=HermesHarness(api_key="key", repo_dir=tmp_path / "hermes"), + **kwargs, + ) + assert outcome.outcome == "parked" + saved = load_record(state, run_id) + assert saved.state == "parked" and saved.wake_attempts == 0 + assert saved.resume_session_id == "s1" and saved.pr_url == record.pr_url + assert saved.stage["candidate_ref"] == record.stage["candidate_ref"] + assert _git(wsroot, "rev-parse", str(record.stage["candidate_ref"])).strip() + assert int(str(saved.stage["hermes_resume_required_chars"])) > 10 + status = collect_status(state, record.target, 1_000_100, records=[saved])["runs"][0] + assert status["phase"] == "configuration-blocked" + assert "OUTERLOOP_HERMES_RESUME_MAX_CHARS" in status["direction"] + assert "configuration-blocked" in caplog.text + deferred: list[str] = [] + woke: list[tuple] = [] + + def sweep(): + _sweep_one( + state, + _fake_dispatch().compute, + None, # type: ignore[arg-type] + 1_000_200, + 0, + 60, + False, + load_record(state, run_id), + "test", + lambda *a: woke.append(a), + deferred, + [], + [], + ) + + sweep() + assert deferred == [run_id] and not woke + monkeypatch.setenv("OUTERLOOP_HERMES_RESUME_MAX_CHARS", "1000000") + sweep() + assert woke + # The operator's retry reaches the author again using the same saved session. + resumed = [] + + def successful_resume(self, brief, workspace, resume_session_id=None): + resumed.append(resume_session_id) + return SessionResult( + session_id="s1", + final_text="done", + cost_usd=0, + num_turns=1, + stop_reason="end_turn", + is_error=False, + transcript_path="", + ) + + monkeypatch.setattr(HermesHarness, "run", successful_resume) + resume_run( + state, run_id, harness=HermesHarness(api_key="key", repo_dir=tmp_path / "hermes"), **kwargs + ) + assert resumed == ["s1"] diff --git a/tests/test_default_claude_model.py b/tests/test_default_claude_model.py index ebad70b2..cac7baae 100644 --- a/tests/test_default_claude_model.py +++ b/tests/test_default_claude_model.py @@ -22,6 +22,15 @@ _PARSE_ARGS = argparse.ArgumentParser.parse_args +@pytest.fixture(autouse=True) +def author_credential(tmp_path, monkeypatch): + for backend in ("claude", "codex"): + key = tmp_path / f"{backend}-author" + key.write_text("distinct-author-credential") + key.chmod(0o600) + monkeypatch.setenv(f"OUTERLOOP_{backend.upper()}_KEY_FILE", str(key)) + + @pytest.mark.parametrize("value", [None, "", " \t"]) def test_default_claude_model_requires_the_setting(monkeypatch, value): if value is None: @@ -391,9 +400,9 @@ def capture_author(backend, model, image): seen.append((backend, model)) return "" - monkeypatch.setattr(attempt, "codex_author_config_error", capture_author) + monkeypatch.setattr(attempt, "author_config_error", capture_author) monkeypatch.setattr(attempt, "role_key", lambda *a: "panel-key") - monkeypatch.setattr("outerloop.role_runner.role_key", lambda *a: "panel-key") + monkeypatch.setattr("outerloop.role_runner.role_key", lambda *a: "author-key") real_panel = attempt._panel_lenses_from_args def capture_panel(args, **kwargs): @@ -493,6 +502,7 @@ def capture(*args, **kwargs): # `review:claude` names no model and is not on the author's backend: refused with pytest.raises(ValueError, match="review:claude:"): attempt._panel_lenses_from_args(args) + monkeypatch.setenv("OUTERLOOP_PANEL_CODEX_KEY_FILE", str(tmp_path / "codex-judge-key")) args.panel = "verify,review:codex:gpt-pinned,review:claude:claude-x" with contextlib.suppress(Exception): # harness construction is not under test attempt._panel_lenses_from_args(args) diff --git a/tests/test_hermes_author.py b/tests/test_hermes_author.py new file mode 100644 index 00000000..b176fe95 --- /dev/null +++ b/tests/test_hermes_author.py @@ -0,0 +1,195 @@ +"""Author selection, wake routing and legacy state for Hermes.""" + +import json +from pathlib import Path +from types import SimpleNamespace + +import pytest + +from outerloop import attempt +from outerloop.attempt import author_config_error, resolve_author_key_file, resume_author +from outerloop.runstate import RECORD_NAME, RunRecord, load_record, run_dir, save_record +from outerloop.tick import ServiceSpec, _author_config_error + + +@pytest.fixture +def hermes_config(monkeypatch, tmp_path): + monkeypatch.setattr("outerloop.hermes_install.hermes_ready", lambda repo: True) + monkeypatch.setenv("REVIEW_HERMES_REPO", str(tmp_path / "hermes")) + monkeypatch.setenv("REVIEW_HERMES_PROVIDER", "openai") + monkeypatch.setenv("OUTERLOOP_AUTHOR_BACKEND", "hermes") + monkeypatch.setenv("OUTERLOOP_AUTHOR_MODEL", "gpt-native") + monkeypatch.delenv("OUTERLOOP_AUTHOR_ENDPOINT", raising=False) + + +def test_native_config_and_preflight(hermes_config, monkeypatch, tmp_path): + spec = ServiceSpec( + target="o/r", account="", partition="", run_root=tmp_path, image="image.sif", home=tmp_path + ) + assert _author_config_error(spec) == "" + assert "requires --image" in author_config_error("hermes", "gpt-native", "") + assert "AUTHOR_MODEL" in author_config_error("hermes", "", "image.sif") + monkeypatch.setenv("REVIEW_HERMES_PROVIDER", "unknown") + assert "PROVIDER" in _author_config_error(spec) + monkeypatch.setenv("REVIEW_HERMES_PROVIDER", "openrouter") + assert author_config_error("hermes", "org/model", "image.sif") == "" + monkeypatch.setenv("OUTERLOOP_HERMES_RESUME_MAX_CHARS", "0") + assert "positive integer" in _author_config_error(spec) + monkeypatch.delenv("OUTERLOOP_HERMES_RESUME_MAX_CHARS") + monkeypatch.setattr("outerloop.hermes_install.hermes_ready", lambda repo: False) + assert "pinned source and runtime" in _author_config_error(spec) + + +def test_hermes_key_and_parked_route(hermes_config, monkeypatch, tmp_path): + key = tmp_path / "hermes_key" + monkeypatch.setenv("OUTERLOOP_HERMES_KEY_FILE", str(key)) + monkeypatch.setenv("OUTERLOOP_CLAUDE_KEY_FILE", "/unrelated") + assert resolve_author_key_file("hermes") == str(key) + record = RunRecord( + "hermes-run", "o/r", "task", "parked", author_backend="hermes", author_model="gpt-native" + ) + for now in (1, 2, 3): + save_record(tmp_path, record, now=now) + record = load_record(tmp_path, record.run_id) + assert resume_author(record, "claude-fleet", "claude") == ("hermes", "gpt-native", str(key)) + + +@pytest.mark.parametrize("wake", [False, True]) +def test_attempt_builds_effective_hermes_author(hermes_config, monkeypatch, tmp_path, wake): + image = tmp_path / "image.sif" + image.touch() + argv = ["climb", "--run-root", str(tmp_path), "--image", str(image), "--panel-skip", "test"] + if wake: + save_record( + tmp_path, + RunRecord( + "h", + "o/r", + "task", + "parked", + author_backend="hermes", + author_model="gpt-native", + stage={"phase": "author-sleep"}, + ), + now=1, + ) + argv += ["--resume", "h"] + monkeypatch.setenv("OUTERLOOP_AUTHOR_BACKEND", "codex") + monkeypatch.setenv("OUTERLOOP_AUTHOR_MODEL", "other-model") + else: + argv += ["--author-backend", "hermes", "--target", "o/r", "--benchmark", "b"] + monkeypatch.setattr("sys.argv", argv) + monkeypatch.setattr( + attempt, "resolve_bot_auth", lambda *a: SimpleNamespace(token=lambda: "bot") + ) + monkeypatch.setattr(attempt, "model_key", lambda *a: "author-key") + monkeypatch.setattr(attempt, "_lease_held_by_another_job", lambda *a: "") + monkeypatch.setattr("outerloop.tick.dispatch_wake_armed", lambda *a: True) + monkeypatch.setattr( + "outerloop.disk.check_mount", lambda *a, **kw: SimpleNamespace(ok=lambda: True) + ) + monkeypatch.setattr(attempt, "arm_self_deadline", lambda *a: 0) + seen = {} + + class Built(Exception): + pass + + def build(*args, **kwargs): + seen.update(kwargs) + raise Built + + monkeypatch.setattr(attempt, "build_harness", build) + with pytest.raises(Built): + attempt.main() + assert seen["backend"] == "hermes" + assert seen["model"] == "gpt-native" + assert seen["hermes_provider"] == "openai" + assert seen["hermes_repo"] == tmp_path / "hermes" + assert seen["container_image"] == str(image) + + +@pytest.mark.parametrize("fixture", ["author_route_legacy.json", "author_route_missing_route.json"]) +def test_legacy_author_records_remain_readable(fixture, tmp_path): + data = json.loads((Path(__file__).parent / "fixtures" / fixture).read_text()) + directory = run_dir(tmp_path, data["run_id"]) + directory.mkdir(parents=True) + path = directory / RECORD_NAME + original = json.dumps(data) + path.write_text(original) + first = load_record(tmp_path, data["run_id"]) + from outerloop.harness import resume_config_block + + assert resume_config_block(first.stage) == "" # legacy missing field is unblocked + for _ in range(2): + assert load_record(tmp_path, data["run_id"]) == first + # Retry after an interrupted writer left an uninstalled temporary file. + (directory / "state.json.tmp").write_text("{") + assert load_record(tmp_path, data["run_id"]) == first + assert path.read_text() == original + + +@pytest.mark.parametrize("judge", ["codex", "hermes"]) +def test_panel_rejects_hermes_author_key(hermes_config, monkeypatch, tmp_path, judge): + from outerloop.attempt import _judge_lens_key + from outerloop.tick import _panel_preflight_error + + key = tmp_path / "author_key" + key.write_text("secret") + key.chmod(0o600) + image = tmp_path / "image.sif" + image.touch() + env = f"OUTERLOOP_PANEL_{judge.upper()}_KEY_FILE" + monkeypatch.setenv("OUTERLOOP_HERMES_KEY_FILE", str(key)) + monkeypatch.setenv(env, str(key)) + with pytest.raises(ValueError, match="hermes author key"): + _judge_lens_key( + backend=judge, + key_file_env=env, + author_backend="hermes", + claude_panel_path=tmp_path / "panel", + image=str(image), + ) + spec = ServiceSpec( + target="o/r", + account="", + partition="", + run_root=tmp_path, + image=str(image), + home=tmp_path, + panel=f"review:{judge}:gpt-native", + ) + assert "hermes author key" in _panel_preflight_error(spec) + + +@pytest.mark.parametrize("judge", ["claude", "codex", "hermes"]) +@pytest.mark.parametrize("duplicate", [False, True]) +def test_effective_explicit_author_key_separation( + hermes_config, monkeypatch, tmp_path, judge, duplicate +): + author = tmp_path / "explicit-author" + author.write_text("same-credential") + author.chmod(0o600) + key = tmp_path / "judge" if duplicate else author + key.write_text("same-credential") + key.chmod(0o600) + monkeypatch.setenv(f"OUTERLOOP_PANEL_{judge.upper()}_KEY_FILE", str(key)) + monkeypatch.setattr(attempt, "build_harness", lambda *a, **k: pytest.fail("built session")) + args = SimpleNamespace( + panel=f"review:{judge}:{'claude-test' if judge == 'claude' else 'gpt-native'}", + panel_key_file=str(key) if judge == "claude" else str(tmp_path / "unused"), + author_backend="hermes", + model="gpt-native", + key_file=str(author), + image="image.sif", + claude_bin="claude", + codex_bin="codex", + ) + with pytest.raises(ValueError, match="role separation"): + attempt._panel_lenses_from_args(args) + + +def test_invalid_resume_budget_does_not_break_judge_construction(monkeypatch): + from outerloop.harness import HermesHarness + + monkeypatch.setenv("OUTERLOOP_HERMES_RESUME_MAX_CHARS", "invalid") + assert HermesHarness(api_key="judge", repo_dir=Path("/opt/hermes")).resume_max_chars is None diff --git a/tests/test_hermes_harness.py b/tests/test_hermes_harness.py index 609314c5..f066aa91 100644 --- a/tests/test_hermes_harness.py +++ b/tests/test_hermes_harness.py @@ -387,3 +387,104 @@ def fake_popen(command, cwd, env, **kw): h2 = HermesHarness(api_key="sk-h", repo_dir=repo, provider="openai-api") h2.run("brief", ws) assert captured["command"][0] == str(runtime / "venv/bin/python") + + +def test_author_fresh_and_resume_commands(monkeypatch, tmp_path): + from outerloop.role_runner import build_harness, run_role + from outerloop.roles import author_spec + from outerloop.syscall import install_tool, tool_command + + workspace = tmp_path / "author" + workspace.mkdir() + install_tool(workspace) + home = tmp_path / "author-home" + spec = author_spec(max_turns=12, walltime_s=90) + harness = build_harness( + "author-secret", + spec, + backend="hermes", + hermes_repo=tmp_path / "hermes", + hermes_provider="openai", + model="gpt-native", + container_image="image.sif", + ) + captured = [] + + def popen(command, **kwargs): + seen = {"command": command, **kwargs} + captured.append(seen) + return _make_hermes_popen(home, "reply", seen)(command) + + monkeypatch.setattr(harness_mod.subprocess, "Popen", popen) + first = run_role(spec, harness, "ORIGINAL python .outerloop/syscall sleep", workspace) + assert first.ok and first.session.num_turns == 1 + second = run_role( + spec, + harness, + "LATEST_RESULTS python .outerloop/syscall submit", + workspace, + first.session.session_id, + ) + assert second.ok and second.session.session_id == first.session.session_id + for seen in captured: + command = seen["command"] + assert any(arg.startswith("--enabled_toolsets=file,terminal") for arg in command) + assert "--max_turns=12" in command + assert f"{workspace}:{workspace}" in command + assert f"{home}:{home}" in command + assert f"{tmp_path / 'hermes'}:{tmp_path / 'hermes'}:ro" in command + assert seen["cwd"] == home + assert seen["env"]["APPTAINERENV_OPENAI_API_KEY"] == "author-secret" + assert seen["env"]["APPTAINERENV_TERMINAL_CWD"] == str(workspace) + assert tool_command(workspace) in seen["brief_text"] + assert "python .outerloop/syscall" not in seen["brief_text"] + assert "ORIGINAL" in captured[1]["brief_text"] + assert "LATEST_RESULTS" in captured[1]["brief_text"] + + +def test_bounded_replay_keeps_brief_tail_and_latest(): + from outerloop.harness import _bounded_resume_brief + + turns = [{"role": "user", "text": "ORIGINAL_BRIEF"}] + turns += [{"role": "assistant", "text": f"OLD_{i}_" + "x" * 200} for i in range(100)] + turns += [ + {"role": "user", "text": "RECENT_INSTRUCTIONS"}, + {"role": "assistant", "text": "RECENT_REPLY"}, + ] + replay = _bounded_resume_brief(turns, "LATEST_RESULTS", 400) + assert len(replay) <= 400 + for text in ("ORIGINAL_BRIEF", "RECENT_INSTRUCTIONS", "RECENT_REPLY", "LATEST_RESULTS"): + assert text in replay + assert "100 earlier turns omitted" in replay + assert "OLD_" not in replay + short = [{"role": "user", "text": "brief"}, {"role": "assistant", "text": "ok"}] + complete = _bounded_resume_brief(short, "results", 1000) + assert _bounded_resume_brief(short, "results", len(complete)) == complete + assert "omitted" not in complete + with pytest.raises(ValueError, match="original brief and latest results"): + _bounded_resume_brief(turns, "LATEST_RESULTS" * 100, 600) + + +def test_legacy_resume_replay_retry_does_not_rewrite_history(monkeypatch, tmp_path): + from outerloop.harness import _resume_transcript_path + + workspace = tmp_path / "legacy" + workspace.mkdir() + home = tmp_path / "legacy-home" + home.mkdir() + legacy = json.loads((Path(__file__).parent / "fixtures/hermes_resume_legacy.json").read_text()) + path = _resume_transcript_path(home, "legacy-id") + path.write_text(json.dumps(legacy)) + original = path.read_bytes() + harness = HermesHarness(api_key="key", repo_dir=tmp_path / "hermes", resume_max_chars=10) + for _ in range(2): + with pytest.raises(harness_mod.ResumeContextBlocked, match="configuration-blocked"): + harness.run("latest results", workspace, "legacy-id") + assert path.read_bytes() == original + harness.resume_max_chars = 1000 + seen: dict[str, Any] = {} + monkeypatch.setattr(harness_mod.subprocess, "Popen", _make_hermes_popen(home, "reply", seen)) + assert not harness.run("latest results", workspace, "legacy-id").is_error + saved = json.loads(path.read_text()) + assert saved["turns"][:2] == legacy["turns"] + assert "original" in seen["brief_text"] and "latest results" in seen["brief_text"] diff --git a/tests/test_init.py b/tests/test_init.py index 2588c2ec..45daacd7 100644 --- a/tests/test_init.py +++ b/tests/test_init.py @@ -374,7 +374,9 @@ def test_github_app_warns_when_it_lands_under_a_different_account( def test_main_yes_rejects_an_unknown_author_backend(tmp_path: Path, monkeypatch, capsys) -> None: monkeypatch.setattr(init, "CONFIG_DIR", tmp_path) - rc = init.main(["--yes", "--compute", "local", "--target", "o/r", "--author-backend", "hermes"]) + rc = init.main( + ["--yes", "--compute", "local", "--target", "o/r", "--author-backend", "unknown"] + ) assert rc == 2 assert "author backend must be one of claude, codex" in capsys.readouterr().err @@ -1094,3 +1096,65 @@ def test_render_env_keeps_comments_and_order_of_a_hand_edited_file() -> None: ) # App auth replaces the PAT assert "OUTERLOOP_GITHUB_APP_FILE=/app.json" in lines # appended assert text.endswith("\n") and "\n\n\n" not in text + + +def test_init_hermes_author(tmp_path, monkeypatch): + monkeypatch.setattr("outerloop.hermes_install.hermes_ready", lambda repo: True) + monkeypatch.setenv("REVIEW_HERMES_PROVIDER", "openrouter") + image = tmp_path / "image.sif" + image.touch() + monkeypatch.setattr(init, "ensure_image", lambda **kw: str(image)) + monkeypatch.setattr(init, "CONFIG_DIR", tmp_path) + monkeypatch.setattr(init, "hermes_ready", lambda repo: True) + monkeypatch.setattr(init, "locate_harness", lambda backend: "/opt/hermes") + assert ( + init.main( + [ + "--yes", + "--compute", + "local", + "--target", + "o/r", + "--author-backend", + "hermes", + "--author-model", + "org/model", + "--no-install-harness", + ] + ) + == 0 + ) + env = (tmp_path / ".env").read_text() + assert "OUTERLOOP_AUTHOR_BACKEND=hermes" in env + assert "REVIEW_HERMES_REPO=/opt/hermes" in env + assert "OUTERLOOP_HERMES_BIN" not in env + + +@pytest.mark.parametrize("missing", ["model", "provider", "image", "runtime", "budget"]) +def test_init_refuses_incomplete_hermes(tmp_path, monkeypatch, capsys, missing): + monkeypatch.setattr(init, "CONFIG_DIR", tmp_path) + monkeypatch.setattr(init, "locate_harness", lambda backend: "/opt/hermes") + monkeypatch.setattr("outerloop.hermes_install.hermes_ready", lambda repo: missing != "runtime") + monkeypatch.setenv("REVIEW_HERMES_PROVIDER", "openai" if missing != "provider" else "") + monkeypatch.setenv( + "OUTERLOOP_HERMES_RESUME_MAX_CHARS", "bad" if missing == "budget" else "120000" + ) + monkeypatch.delenv("OUTERLOOP_AUTHOR_ENDPOINT", raising=False) + image = tmp_path / "image.sif" + image.touch() + monkeypatch.setattr(init, "ensure_image", lambda **kw: "" if missing == "image" else str(image)) + args = [ + "--yes", + "--compute", + "local", + "--target", + "o/r", + "--author-backend", + "hermes", + "--no-install-harness", + ] + if missing != "model": + args += ["--author-model", "gpt-native"] + assert init.main(args) == 2 + assert not (tmp_path / ".env").exists() + assert "outerloop init:" in capsys.readouterr().err diff --git a/tests/test_install_harness.py b/tests/test_install_harness.py index 3a977546..1c874844 100644 --- a/tests/test_install_harness.py +++ b/tests/test_install_harness.py @@ -42,23 +42,50 @@ def run(argv, *, check): if mode == "failure": raise subprocess.CalledProcessError(1, argv) if mode != "no_output": - target.parent.mkdir() - target.write_text("#!/bin/sh\n") - target.chmod(0o755) + if backend == "hermes": + from outerloop.hermes_install import HERMES_SHA, hermes_runtime + + target.mkdir(parents=True) + (target / "run_agent.py").touch() + runtime = hermes_runtime(target) + (runtime / "venv/bin").mkdir(parents=True) + python = runtime / "venv/bin/python" + python.write_text("#!/bin/sh\n") + python.chmod(0o755) + (runtime / ".complete").write_text(HERMES_SHA) + else: + target.parent.mkdir() + target.write_text("#!/bin/sh\n") + target.chmod(0o755) monkeypatch.setattr(init.subprocess, "run", run) args = ["--yes", "--compute", "local", "--target", "o/r", "--author-backend", backend] if mode == "skip": args.append("--no-install-harness") + if backend == "hermes": + image = tmp_path / "image.sif" + image.touch() + monkeypatch.setattr(init, "ensure_image", lambda **kw: str(image)) + monkeypatch.setenv("REVIEW_HERMES_PROVIDER", "openai") + args += ["--author-model", "gpt-native"] + if mode == "present": + # Provision the same pinned runtime an earlier installer left. + run(["bash", str(ROOT / "scripts/install_hermes.sh"), str(target)], check=True) + calls.clear() rc = init.main(args) if mode in ("failure", "no_output", "permission"): assert rc == 1 assert "Run manually: bash " in capsys.readouterr().err assert not (tmp_path / "config" / ".env").exists() + elif backend == "hermes" and mode == "skip": + assert rc == 2 # skipping a missing runtime cannot report successful setup + assert not (tmp_path / "config" / ".env").exists() else: assert rc == 0 env = (tmp_path / "config" / ".env").read_text() - assert (f"{init.author_bin_env(backend)}={target}" in env) == (mode != "skip") + assert (f"{init.author_bin_env(backend)}={target}" in env) == ( + mode != "skip" or backend == "hermes" + ) assert len(calls) == (mode not in ("present", "skip")) if mode == "missing": monkeypatch.setattr(init, "locate_harness", lambda _: str(target)) diff --git a/tests/test_tick.py b/tests/test_tick.py index f5ec4b52..ef75ebf0 100644 --- a/tests/test_tick.py +++ b/tests/test_tick.py @@ -38,6 +38,16 @@ TTL = 4500.0 +@pytest.fixture(autouse=True) +def author_credential(tmp_path, monkeypatch): + """Panel preflight compares credentials with the effective author key.""" + key = tmp_path / "author-key" + key.write_text("distinct-author-credential") + key.chmod(0o600) + monkeypatch.setenv("OUTERLOOP_CLAUDE_KEY_FILE", str(key)) + return key + + @dataclass class FakeSlurm: """status() by job id; '!' prefix means the query itself fails.""" @@ -1865,7 +1875,9 @@ def runner(argv, timeout_s): assert _author_config_error(make()) == "" -def test_panel_key_preflight_blocks_claim_and_launch(tmp_path: Path, monkeypatch: Any) -> None: +def test_panel_key_preflight_blocks_claim_and_launch( + tmp_path: Path, monkeypatch: Any, author_credential: Path +) -> None: """Panel on + a key the climb would reject (missing, group-readable, empty): the intake lane claims nothing and the self-initiated lane submits nothing — the strand is caught before any side effect. The @@ -1946,7 +1958,7 @@ def make(**kw: Any) -> ServiceSpec: monkeypatch.setenv("OUTERLOOP_CLAUDE_KEY_FILE", "keys/author") rel_err = _panel_preflight_error(make(panel_key_file=str(good))) assert "author key path" in rel_err and "relative" in rel_err - monkeypatch.delenv("OUTERLOOP_CLAUDE_KEY_FILE") + monkeypatch.setenv("OUTERLOOP_CLAUDE_KEY_FILE", str(author_credential)) # both lanes consult it BEFORE side effects: nothing claimed or submitted bad = make(panel_key_file=str(tmp_path / "nope")) From d0a5c315de7466005bd947029985e2a851076ac1 Mon Sep 17 00:00:00 2001 From: Mengye Ren Date: Mon, 28 Sep 2026 13:06:36 -0400 Subject: [PATCH 2/2] Review fixes: clear the replay-block marker; one effective author credential - The configuration-blocked marker is cleared when a resume succeeds or the run ends, so a later normal sleep is not re-woken. - Attempt and tick preflight share one helper for the author's effective credential, including endpoint profiles. - The Hermes replay budget is validated only when Hermes is the author. Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 9 ++++--- src/outerloop/attempt.py | 52 ++++++++++++++++++++++++++++----------- src/outerloop/init.py | 16 ++++++------ src/outerloop/runstate.py | 4 +++ src/outerloop/tick.py | 44 ++++++++++++++------------------- tests/test_attempt.py | 18 +++++++++++++- tests/test_endpoints.py | 38 ++++++++++++++++++++++++++++ tests/test_init.py | 25 +++++++++++++++++++ tests/test_runstate.py | 13 ++++++++++ 9 files changed, 167 insertions(+), 52 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index c4d7a84d..b814a6a5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,12 +10,15 @@ Versions follow [SemVer](https://semver.org). configuration-blocked status, retaining their session and snapshot without consuming wake retries. Author/judge separation checks effective key paths and credential values before constructing sessions; init rejects incomplete - Hermes configuration. + Hermes configuration only for Hermes authors. Endpoint author credentials are + shared by attempt and tick preflight, including native panel key comparisons. - Upgrading: no action or backfill needed; legacy records without `stage.hermes_resume_required_chars` are unblocked. The first oversized wake records the required budget; raising `OUTERLOOP_HERMES_RESUME_MAX_CHARS` lets - the next tick or wake resume. Before rollback, resolve blocked runs: older - kernels ignore this optional field and may consume retries or abort them. + the next tick or wake resume. Successful resumes and endings clear the marker, + so later normal sleeps are not configuration wakes. Before rollback, resolve + blocked runs: older kernels ignore this optional field and may consume retries + or abort them. - Hermes is an author peer: init, native provider/endpoint validation, contained fresh and resumed sessions, absolute syscall commands, and separate author keys. diff --git a/src/outerloop/attempt.py b/src/outerloop/attempt.py index 5377c362..a185b5dd 100644 --- a/src/outerloop/attempt.py +++ b/src/outerloop/attempt.py @@ -171,6 +171,32 @@ def resolve_author_key_file(backend: str, explicit: str = "") -> str: return os.path.expanduser(explicit) +@dataclass(frozen=True) +class AuthorCredential: + key_file: Path + backend: str + model: str + + def key(self) -> str: + """Read after structural panel checks, preserving their diagnostics.""" + return model_key(self.key_file, self.backend, self.model) + + +def effective_author_credential(backend: str, model: str, explicit: str = "") -> AuthorCredential: + """Resolve the actual author credential, including endpoint profiles.""" + _, profile = resolve_endpoint(model, backend) + path = profile.key_file if profile else Path(resolve_author_key_file(backend, explicit)) + return AuthorCredential(path, backend, model) + + +def _clear_resume_block(run_root: Path, run_id: str, now: float) -> None: + record = load_record(run_root, run_id) + if "hermes_resume_required_chars" in record.stage: + stage = dict(record.stage) + stage.pop("hermes_resume_required_chars") + save_record(run_root, dc_replace(record, stage=stage), now) + + def author_config_error( backend: str, model: str, @@ -1682,7 +1708,9 @@ def changed_paths() -> list[str]: else None, judged=judged or _stage_judged(record), ) + _clear_resume_block(run_root, run_id, now) except RunParked as p: + _clear_resume_block(run_root, run_id, now) # slept again, or the gate dispatched its measures (a candidate park the # existing wake path decides). Keep the NEW park's snapshot ref; the OLD # sleep ref is superseded once the new park persists. @@ -3206,12 +3234,10 @@ def _panel_lenses_from_args( if author_model is None: author_model = getattr(args, "model", "") or "" parsed = resolve_lenses(args.panel, author_backend, author_model) - _, author_profile = resolve_endpoint(author_model, author_backend) - author_path = ( - author_profile.key_file - if author_profile - else Path(resolve_author_key_file(author_backend, getattr(args, "key_file", ""))) + author_credential = effective_author_credential( + author_backend, author_model, getattr(args, "key_file", "") ) + author_path = author_credential.key_file prepared = [] # the anthropic panel key is read only when a claude lens will use it — # a codex-only panel must not demand an unrelated credential @@ -3270,7 +3296,7 @@ def _panel_lenses_from_args( else claude_panel_path ) if lens_path.resolve() == author_path.expanduser().resolve() or ( - lens_key and lens_key == model_key(author_path, author_backend, author_model) + lens_key and lens_key == author_credential.key() ): raise ValueError( "a panel judge key is the effective author key " @@ -5078,7 +5104,9 @@ def _run_id(value: str) -> str: or _wake_stage.get("submitted") or getattr(_wake_record, "pr_url", "") ): - wake_api_key = model_key(wake_key_file, wake_backend, wake_model) + wake_api_key = effective_author_credential( + wake_backend, wake_model, wake_key_file + ).key() if wake_api_key and wake_api_key in wake_panel_secrets: args.panel_skip = "a panel judge key is this run's author key (role separation)" wake_lenses = () @@ -5159,15 +5187,11 @@ def _run_id(value: str) -> str: parser.error(_err) # The author key defaults per backend so the # tick never threads it — see resolve_author_key_file (result is ~-expanded). - _, author_endpoint = resolve_endpoint(args.model, args.author_backend) - args.key_file = ( - str(author_endpoint.key_file) - if author_endpoint - else resolve_author_key_file(args.author_backend, args.key_file) - ) + author_credential = effective_author_credential(args.author_backend, args.model, args.key_file) + args.key_file = str(author_credential.key_file) # same 0600 discipline as the PAT: this key spends real money. A missing # file is tolerated only when Vertex (ADC) covers the claude backend. - api_key = model_key(args.key_file, args.author_backend, args.model) + api_key = author_credential.key() stamp = datetime.now(UTC).strftime("%Y%m%d-%H%M%S") # the agent id keeps concurrent same-benchmark slots (the width dial's # portfolio case) from minting one run directory in the same second diff --git a/src/outerloop/init.py b/src/outerloop/init.py index f3a24611..e49afb56 100644 --- a/src/outerloop/init.py +++ b/src/outerloop/init.py @@ -850,19 +850,19 @@ def main(argv: list[str] | None = None) -> int: args = parser.parse_args(sys.argv[2:] if argv is None else argv) interactive = not args.yes - from outerloop.harness import hermes_resume_max_chars - - try: - hermes_resume_max_chars() - except ValueError as exc: - print(f"outerloop init: {exc}", file=sys.stderr) - return 2 - try: answers, pat_file = _collect(args, interactive) except StartError as exc: print(f"outerloop init: {exc}", file=sys.stderr) return 2 + if answers.author_backend == "hermes": + from outerloop.harness import hermes_resume_max_chars + + try: + hermes_resume_max_chars() + except ValueError as exc: + print(f"outerloop init: {exc}", file=sys.stderr) + return 2 if not answers.target: print("outerloop init: a target repo is required (--target owner/repo)", file=sys.stderr) return 2 diff --git a/src/outerloop/runstate.py b/src/outerloop/runstate.py index 56481fd5..7e5203ad 100644 --- a/src/outerloop/runstate.py +++ b/src/outerloop/runstate.py @@ -255,6 +255,10 @@ def _save_record(root: Path, record: RunRecord, now: float) -> None: raise ValueError("waiting run with an experiment needs a deadline") directory = run_dir(root, record.run_id) directory.mkdir(parents=True, exist_ok=True) + if record.state == ENDED: + stage = dict(record.stage) + stage.pop("hermes_resume_required_chars", None) + record = replace(record, stage=stage) stamped = replace(record, updated=now, created=record.created or now) # unique tmp name: two concurrent writers must not interleave into the # same tmp file before the atomic replace diff --git a/src/outerloop/tick.py b/src/outerloop/tick.py index eea7c6fe..f6e7f581 100644 --- a/src/outerloop/tick.py +++ b/src/outerloop/tick.py @@ -1696,7 +1696,9 @@ def _sweep_one( stuck.append(record.run_id) return - if record.stage.get("hermes_resume_required_chars"): + # Only a positive pending requirement represents a configuration wake. + # A successful resume removes it before any subsequent normal sleep. + if int(str(record.stage.get("hermes_resume_required_chars", 0))) > 0: wake(record, "resume configuration unblocked", "configuration") return @@ -2443,7 +2445,11 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: if not spec.panel.strip(): return "" try: - from outerloop.attempt import PANEL_KEY_DEFAULT, resolve_author_key_file + from outerloop.attempt import ( + PANEL_KEY_DEFAULT, + effective_author_credential, + resolve_author_key_file, + ) from outerloop.endpoints import author_model_setting from outerloop.github import FileTokenProvider from outerloop.panel import resolve_lenses @@ -2466,10 +2472,10 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: if any(backend == "hermes" for _, backend, _ in lenses): hermes_resume_max_chars() author_backend = os.environ.get("OUTERLOOP_AUTHOR_BACKEND") or "claude" - _, author_endpoint = resolve_endpoint( - author_model_setting(author_backend, os.environ.get("OUTERLOOP_AUTHOR_MODEL", "")), - author_backend, + author_credential = effective_author_credential( + author_backend, fleet_author_model(author_backend) ) + author_path = author_credential.key_file traditional = [] for kind, backend, model in lenses: _, profile = resolve_endpoint(model, backend) @@ -2479,21 +2485,14 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: if not spec.image or not Path(spec.image).is_file(): return f"a {backend} endpoint panel lens requires a real container image" author_backend = os.environ.get("OUTERLOOP_AUTHOR_BACKEND") or "claude" - from outerloop.endpoints import model_key, validate_judge_key_file + from outerloop.endpoints import validate_judge_key_file validate_judge_key_file( profile, - author_endpoint.key_file - if author_endpoint - else resolve_author_key_file(author_backend), + author_path, spec.panel_key_file or PANEL_KEY_DEFAULT, ) - author_key = model_key( - resolve_author_key_file(author_backend), - author_backend, - fleet_author_model(author_backend), - ) - if profile.key() == author_key: + if profile.key() == author_credential.key(): return "a panel judge key is the author key (role separation)" if backend == "hermes": from outerloop.hermes_install import hermes_ready @@ -2545,12 +2544,8 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: "provider's login)" ) key = FileTokenProvider(key_path).token() - from outerloop.endpoints import model_key - - if key and key == model_key( - resolve_author_key_file(author_backend), - author_backend, - fleet_author_model(author_backend), + if key_path.resolve() == author_path.resolve() or ( + key and key == author_credential.key() ): return "a panel judge key is the author key (role separation)" if lens_backend == "hermes": @@ -2588,8 +2583,7 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: # (claude vs codex keys coexist), config-driven like the climb itself — so # the role-separation check compares the panel key against the RIGHT author # key, and a codex run is never judged by a stray Claude key. - fleet_backend = os.environ.get("OUTERLOOP_AUTHOR_BACKEND") or "claude" - author = Path(resolve_author_key_file(fleet_backend)) + author = author_path if not author.is_absolute(): # same rule as the panel key: the climb resolves paths from a # flight directory, so a relative author path both misconfigures @@ -2606,9 +2600,7 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: from outerloop.role_runner import role_key key = role_key(path) - from outerloop.endpoints import model_key - - if key and key == model_key(author, fleet_backend, fleet_author_model(fleet_backend)): + if key and key == author_credential.key(): return "a panel judge key is the author key (role separation)" return "" except Exception as exc: diff --git a/tests/test_attempt.py b/tests/test_attempt.py index 464d0921..ced547eb 100644 --- a/tests/test_attempt.py +++ b/tests/test_attempt.py @@ -7189,7 +7189,10 @@ def head_contains(self, target, base, head): @pytest.mark.parametrize("pr", [False, True]) -def test_hermes_resume_configuration_block_preserves_park(tmp_path, monkeypatch, caplog, pr): +@pytest.mark.parametrize("sleep_again", [False, True]) +def test_hermes_resume_configuration_block_preserves_park( + tmp_path, monkeypatch, caplog, pr, sleep_again +): from dataclasses import replace from outerloop.climbboard import collect_status @@ -7275,6 +7278,10 @@ def sweep(): def successful_resume(self, brief, workspace, resume_session_id=None): resumed.append(resume_session_id) + if sleep_again: + from outerloop.syscall_cli import main + + assert main(["sleep"], root=workspace) == 0 return SessionResult( session_id="s1", final_text="done", @@ -7290,3 +7297,12 @@ def successful_resume(self, brief, workspace, resume_session_id=None): state, run_id, harness=HermesHarness(api_key="key", repo_dir=tmp_path / "hermes"), **kwargs ) assert resumed == ["s1"] + saved = load_record(state, run_id) + assert "hermes_resume_required_chars" not in saved.stage + if sleep_again: + assert saved.state == "parked" + assert saved.stage["phase"] == "author-sleep" + woke.clear() + for _ in range(2): + sweep() + assert not woke diff --git a/tests/test_endpoints.py b/tests/test_endpoints.py index a0624fed..9c15b77d 100644 --- a/tests/test_endpoints.py +++ b/tests/test_endpoints.py @@ -528,3 +528,41 @@ def resume(*args, **kwargs): with pytest.raises(Resumed): attempt.main() assert seen["backend"] == "claude" and seen["model"] == "claude-native" + + +@pytest.mark.parametrize("backend", ["claude", "codex", "hermes"]) +@pytest.mark.parametrize("same_path", [False, True]) +def test_endpoint_author_native_panel_separation( + profile, tmp_path, monkeypatch, backend, same_path +): + from outerloop.attempt import _panel_lenses_from_args + from outerloop.tick import ServiceSpec, _panel_preflight_error + + image = tmp_path / "image.sif" + image.touch() + monkeypatch.setenv("OUTERLOOP_AUTHOR_BACKEND", "codex") + monkeypatch.setenv("OUTERLOOP_AUTHOR_MODEL", "open-model[endpoint=local]") + judge = Path(profile["OUTERLOOP_ENDPOINT_LOCAL_KEY_FILE"]) if same_path else tmp_path / "judge" + judge.write_text("endpoint-secret") + judge.chmod(0o600) + monkeypatch.setenv(f"OUTERLOOP_PANEL_{backend.upper()}_KEY_FILE", str(judge)) + panel = f"verify:{backend}:judge-model" + spec = ServiceSpec( + account="", + partition="", + run_root=tmp_path, + home=tmp_path, + panel=panel, + image=str(image), + panel_key_file=str(judge) if backend == "claude" else "", + ) + assert "role separation" in _panel_preflight_error(spec) + args = SimpleNamespace( + panel=panel, + author_backend="codex", + model="open-model[endpoint=local]", + image=str(image), + panel_key_file=spec.panel_key_file, + ) + with pytest.raises(ValueError, match="role separation"): + _panel_lenses_from_args(args) diff --git a/tests/test_init.py b/tests/test_init.py index 45daacd7..19f7b83e 100644 --- a/tests/test_init.py +++ b/tests/test_init.py @@ -1158,3 +1158,28 @@ def test_init_refuses_incomplete_hermes(tmp_path, monkeypatch, capsys, missing): assert init.main(args) == 2 assert not (tmp_path / ".env").exists() assert "outerloop init:" in capsys.readouterr().err + + +@pytest.mark.parametrize("backend", ["claude", "codex"]) +def test_init_ignores_hermes_budget_for_other_authors(tmp_path, monkeypatch, backend): + monkeypatch.setattr(init, "CONFIG_DIR", tmp_path) + monkeypatch.setenv("OUTERLOOP_HERMES_RESUME_MAX_CHARS", "invalid") + assert ( + init.main( + [ + "--yes", + "--compute", + "local", + "--target", + "o/r", + "--author-backend", + backend, + "--author-model", + "native-model", + "--claude-model", + "claude-model", + "--no-install-harness", + ] + ) + == 0 + ) diff --git a/tests/test_runstate.py b/tests/test_runstate.py index aad047e3..9748ccf3 100644 --- a/tests/test_runstate.py +++ b/tests/test_runstate.py @@ -483,3 +483,16 @@ def other_node_writes(seconds): monkeypatch.setattr(runstate.time, "sleep", lambda seconds: None) lease = runstate.acquire_tick_lease(tmp_path, "alpha2:2", 101, 300, settle_s=0.25) runstate.release_tick_lease(lease, "alpha2:2") + + +def test_ended_record_drops_resume_block(tmp_path): + save_record( + tmp_path, + make_record( + state=ENDED, + ending=STUCK, + stage={"hermes_resume_required_chars": 200000, "sleeps_used": 2}, + ), + now=1.0, + ) + assert load_record(tmp_path, "r1").stage == {"sleeps_used": 2}