From c16eaeab002f7362c26e64bd9108a083cd8a7b97 Mon Sep 17 00:00:00 2001 From: Mengye Ren Date: Mon, 28 Sep 2026 12:15:28 -0400 Subject: [PATCH 1/2] Self-hosted model endpoints for any role; Hermes v2026.9.24 with reasoning echo Operators can run authors and judges on a model they serve themselves. Deployment settings define named endpoint profiles (OUTERLOOP_ENDPOINT__URL, _KEY_FILE, _MODEL, _API); a role selects one with `model[endpoint=name]`, or `[endpoint=name]` for the profile's model, in the author setting, panel lens specs and REVIEW_ENDPOINT. Claude Code uses the Anthropic API (base URL without /v1, auth token from the key file, vendor and cloud paths off), Codex a custom Responses provider, Hermes a named custom provider with reasoning_echo so each turn keeps its earlier reasoning. Keys travel only through the environment, are redacted from verdicts, and keep the judge/author key separation. Old run records keep their native routing. Hermes moves to v2026.9.24: its flags are argparse now, reasoning_echo exists, and newer instruction-file names are sanitized for judges. Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 21 + docs/endpoints.md | 162 ++++++ scripts/tick_deploy.sh | 7 + src/outerloop/attempt.py | 87 ++- src/outerloop/cli.py | 46 +- src/outerloop/compute.py | 14 +- src/outerloop/endpoints.py | 137 +++++ src/outerloop/harness.py | 103 +++- src/outerloop/harnesses.toml | 4 +- src/outerloop/panel.py | 16 +- src/outerloop/review_agent.py | 29 +- src/outerloop/review_agent_cli.py | 19 + src/outerloop/role_runner.py | 40 +- src/outerloop/runstate.py | 4 +- src/outerloop/tick.py | 56 +- tests/fixtures/README.md | 18 + tests/fixtures/author_route_legacy.json | 31 + .../fixtures/author_route_missing_model.json | 30 + .../fixtures/author_route_missing_route.json | 28 + tests/fixtures/hermes_sample_20260924.json | 28 + tests/test_attempt.py | 2 +- tests/test_default_claude_model.py | 1 + tests/test_endpoints.py | 530 ++++++++++++++++++ tests/test_hermes_harness.py | 12 +- tests/test_install_harness.py | 2 +- tests/test_local_compute.py | 23 + tests/test_review_agent.py | 69 +++ tests/test_start.py | 5 +- 28 files changed, 1456 insertions(+), 68 deletions(-) create mode 100644 docs/endpoints.md create mode 100644 src/outerloop/endpoints.py create mode 100644 tests/fixtures/README.md create mode 100644 tests/fixtures/author_route_legacy.json create mode 100644 tests/fixtures/author_route_missing_model.json create mode 100644 tests/fixtures/author_route_missing_route.json create mode 100644 tests/fixtures/hermes_sample_20260924.json create mode 100644 tests/test_endpoints.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 3d0960ee..81edb2de 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,27 @@ Versions follow [SemVer](https://semver.org). ## [Unreleased] +- Endpoint profiles declare compatible APIs and use `model[endpoint=profile]` + selectors, preserving native vendor model IDs. Judge credentials enforce file + separation; verdicts redact the session key before posting or aggregation. +- Upgrading: legacy records missing model/route fields retain native routing; + missing native model configuration fails closed instead of adopting fleet endpoints. + + +- Named, file-authenticated endpoint profiles work for authors, panel lenses, + and standalone reviewers on Claude Code, Codex, and Hermes. Sessions use each + backend's native API configuration; keys reach contained sessions through env, + never argv. See [endpoint settings and validation](docs/endpoints.md). +- Hermes is pinned to v2026.9.24 (`f97608f178d1ffeca59860195ab7da295f7c8e5f`). + Remove Fire quoting for its new argparse entrypoint, enable + `model.reasoning_echo` for endpoint profiles, and sanitize its instruction-file + aliases in judge checkouts. +- Upgrading: existing runs need no backfill; the first tick validates endpoint + selections without changing legacy routes. New endpoint authors save + `model[endpoint=profile]` selectors; keep those profiles until runs finish, and finish + endpoint runs before rolling back. Hermes source/runtime must be upgraded + with the harness. See [compatibility and rollback](docs/endpoints.md#upgrade-and-rollback). + ### Fixed - Manual harness upgrades honor `--root`, environment, and `.env` state roots. Retry records are replaced atomically; unreadable or invalid records are logged and ignored. Deploy loads the configured cache root before selecting the uv cache. diff --git a/docs/endpoints.md b/docs/endpoints.md new file mode 100644 index 00000000..225fef05 --- /dev/null +++ b/docs/endpoints.md @@ -0,0 +1,162 @@ +# Self-hosted model endpoints + +Authors, panel judges, and the standalone reviewer use the same named endpoint +profile. Put coordinates and a **key-file path**, never a bearer key, in the +operator `.env`: + +```dotenv +OUTERLOOP_ENDPOINT_AUTHOR_URL=https://llm.example.internal/v1 +OUTERLOOP_ENDPOINT_AUTHOR_KEY_FILE=/keys/model-author +OUTERLOOP_ENDPOINT_AUTHOR_MODEL=open-model +OUTERLOOP_ENDPOINT_AUTHOR_API=chat +OUTERLOOP_ENDPOINT_JUDGE_URL=https://llm.example.internal/v1 +OUTERLOOP_ENDPOINT_JUDGE_KEY_FILE=/keys/model-judge +OUTERLOOP_ENDPOINT_JUDGE_MODEL=open-model +OUTERLOOP_ENDPOINT_JUDGE_API=chat,responses + +OUTERLOOP_AUTHOR_BACKEND=hermes +OUTERLOOP_AUTHOR_ENDPOINT=author +OUTERLOOP_PANEL=verify:hermes:[endpoint=judge],review:codex:open-model[endpoint=judge] +REVIEW_HERMES_REPO=/opt/hermes-agent +OUTERLOOP_IMAGE=/opt/agent.sif +``` + +The files must be readable, nonempty, and private (`chmod 600`). Paths must be +absolute (`~` is expanded). Profile names start with a letter and contain only +letters, digits, and underscores; references are case-insensitive and their env +keys are uppercase. Only `_URL`, `_KEY_FILE`, `_MODEL`, and `_API` are forwarded by the +profile allowlist. URLs cannot contain credentials, a query, or a fragment. + +A profile owns its model. `OUTERLOOP_AUTHOR_MODEL` may be omitted; if set, it must +match the profile's served model. Panel syntax is +`kind[:backend[:model[endpoint=profile]]]`, where kind is `verify` or `review`. Omitting the +model before `[endpoint=...]` takes the profile model. Native model IDs containing `@`, `/`, and `:` retain their +meaning. An endpoint author does not lend its profile or credential to an +implicit judge: select a judge profile or an explicit conventional model. Author +and judge credentials must differ, including when two files contain the same key. +Judges can share a judge profile. + +For the standalone reviewer/summarizer contract, set `REVIEW_BACKEND`, +`REVIEW_ENDPOINT`, and optionally `REVIEW_MODEL`. The same profile keys apply; +`REVIEW_HERMES_REPO` is still required for Hermes. Endpoint selection replaces the +conventional reviewer key-variable and Hermes provider settings. Library callers +can pass `endpoint="judge"` to `build_harness` for any role specification. + +## Backend wiring + +The required `_API` declares a comma-separated list of supported APIs: `anthropic`, +`responses`, and/or `chat`. Preflight requires `anthropic` for Claude, `responses` +for Codex, and `chat` for Hermes. This declaration validates configuration; it does +not probe server support. +Use the base URL accepted by that backend's client. A server may expose all three +APIs, but if its API prefixes differ, define profiles with the appropriate URLs. +For example, an Anthropic-compatible client may need +`https://llm.example.internal` while OpenAI-compatible clients need +`https://llm.example.internal/v1`. + +- **Claude Code:** `ANTHROPIC_BASE_URL` and `ANTHROPIC_AUTH_TOKEN`; the profile + model also sets the default Opus, Sonnet, Haiku, and small-fast model variables. + Nonessential traffic is disabled. Vertex, Bedrock, and Foundry are disabled; + API keys, ADC, and ambient authentication are excluded from the session env. +- **Codex:** the per-run `.codex/config.toml` selects a named provider with + `base_url`, `env_key`, `wire_api = "responses"`, and + `requires_openai_auth = false`. Custom-provider sessions skip OpenAI login. + This follows the [official configuration reference](https://developers.openai.com/codex/config-reference). + Configuration is tested here; live self-hosted Responses interoperability is + not verified by this change. +- **Hermes:** the per-run `.hermes/config.yaml` selects a named + `custom_providers` list entry with `base_url`, `key_env`, and + `api_mode: chat_completions`. **`reasoning_echo: true` belongs under `model`, + not inside the custom-provider entry.** No `--base_url` override is passed. + `TERMINAL_CWD` selects the workspace while sample files stay in the private + session home. + +Keys are read by the orchestrator and placed only in the session environment. +Contained sessions receive them through `APPTAINERENV_*`; keys are never put in +argv or generated provider config files. Existing role budgets and containment +requirements still apply. Hermes authors require an endpoint profile and the +pinned source/runtime installation. Configure Hermes author settings directly in +`.env`; the interactive init wizard still provisions native Claude/Codex authors. + +Start/tick/climb and reviewer resolution reject unknown profiles, invalid or +missing key files, unsupported backends, and model/profile mismatches before a +model session. Tick also checks judge credential separation and Hermes runtime +readiness before claiming work. These are local configuration checks, not probes +of a server's authentication, API support, or model availability. + +## Hermes v2026.9.24 interface audit + +Read both tags from a clone outside the worktree. The new tag resolves to +`f97608f178d1ffeca59860195ab7da295f7c8e5f`; the previous v2026.8.13 pin was +`f80f453ae0679347e38abc917c7f94f717bf96c5`. + +- `run_agent.py` now delegates to + [`agent/legacy_cli.py`](https://github.com/NousResearch/hermes-agent/blob/f97608f178d1ffeca59860195ab7da295f7c8e5f/agent/legacy_cli.py). + **argparse replaced Fire.** The underscore flag spellings remain accepted + (`query`, `model`, `max_turns`, `save_sample`, enabled/disabled toolsets). + Embedded Fire quotes must be removed from comma-separated toolsets. The + harness argv was exercised against the new source's actual parser function. +- [`run_agent.py::_save_sample_trajectory`](https://github.com/NousResearch/hermes-agent/blob/f97608f178d1ffeca59860195ab7da295f7c8e5f/run_agent.py#L1485) + still writes `sample_.json` in cwd, with `conversations`, timestamp, model, + completed, and query. The ShareGPT turns still use `from`/`value`; assistant + turns use `gpt`. The implementation moved into helpers; the envelope remains + compatible. The new fixture was generated with the actual converter and + sample writer on synthetic input (no inference). +- [`agent/reasoning_params.py`](https://github.com/NousResearch/hermes-agent/blob/f97608f178d1ffeca59860195ab7da295f7c8e5f/agent/reasoning_params.py) + reads `model.reasoning_echo`, an opt-in absent at the old pin. It preserves + reasoning on earlier turns within the native conversation. The kernel's + existing resume-by-brief mechanism is unchanged; it is not a native replay of + the full tool/reasoning history across separate harness invocations. +- [`hermes_cli/runtime_provider_custom.py`](https://github.com/NousResearch/hermes-agent/blob/f97608f178d1ffeca59860195ab7da295f7c8e5f/hermes_cli/runtime_provider_custom.py) + still resolves named `custom_providers` and their `key_env` credential pointers. + The newer `providers` mapping coexists with the supported legacy list form. + `model.provider` and `model.default` in `config.yaml` remain the seed contract. +- [`toolsets.py`](https://github.com/NousResearch/hermes-agent/blob/f97608f178d1ffeca59860195ab7da295f7c8e5f/toolsets.py) + still defines every toolset the harness uses: file, terminal, web, search, + browser, computer_use, code_execution, delegation, cronjob, skills, memory. +- [`agent/prompt_builder.py`](https://github.com/NousResearch/hermes-agent/blob/f97608f178d1ffeca59860195ab7da295f7c8e5f/agent/prompt_builder.py) + now recognizes `AGENTS.override.md` before `AGENTS.md`/`agents.md`, walks the + git-root-to-cwd chain, and avoids accidental install-tree fallback. Its + `.hermes.md`/`HERMES.md`, lowercase Claude/agents files, and cursor rules are + also auto-load surfaces (several already existed at the old pin). Judge + sanitization now covers all these names. + +## Upgrade and rollback + +The run-record schema is unchanged. New endpoint authors persist +`[endpoint=]` in `author_model` and the resolved key-file path in +`author_key_file`. Resuming uses the saved selector, never the fleet's current +`OUTERLOOP_AUTHOR_ENDPOINT`. Retain referenced profile definitions until those +runs finish; changing a profile's URL/key changes that profile's routing, while +changing its served model makes old runs fail validation rather than silently +switch models. + +Legacy records (including records missing backend/model fields), ended runs, +inbox messages, PR branches, and Hermes resume files remain readable without a +backfill. The fixture from kernel `5563c46` exercises the existing record writer's +output through load, repeat, and save/reload retry paths. Existing older run and +panel fixtures remain in the suite. The first tick preserves legacy routes and +validates endpoint settings for new work; it does not migrate parked sessions or +rewrite PRs/messages. Sanitization runs on disposable judge checkouts. + +Rollback is safe for conventional runs. Finish endpoint-backed runs and remove +endpoint selections before rolling back to a kernel that does not understand +`[endpoint=profile]`; that kernel cannot safely resume their model selectors. Reinstall the +Hermes version required by the chosen kernel. The Hermes pin and argparse harness +change must ship together; overriding only the pin to the older Fire version is +not supported. + +## Offline client validation limits + +`test_exact_client_configuration_at_process_boundary` checks the exact argv, +environment, and provider config passed to all three clients, in contained and +uncontained sessions, including client failure. It checks that ambient vendor +credentials/routes are excluded and Codex endpoint sessions never invoke login. +This is configuration coverage, not evidence of the clients' network behavior. +The pinned Linux clients/runtime are not available in the offline test environment +(the installed Claude/Codex binaries have different versions, and Hermes lacks +its pinned runtime). No test claims to prove absence of client-internal vendor +fallback. That requires the pinned clients under network observation. + +Legacy records with no model use native model defaults only. A missing native +Codex model fails preflight; it never adopts an endpoint. No backfill is needed. diff --git a/scripts/tick_deploy.sh b/scripts/tick_deploy.sh index b88d8080..3cfb28a1 100755 --- a/scripts/tick_deploy.sh +++ b/scripts/tick_deploy.sh @@ -134,10 +134,17 @@ fi # Only this ALLOWLIST is read, so .env is structurally per-tick author config # and can never hijack the chain's identity or scheduling. if [ -n "$ENV_TRUSTED" ]; then + # Named profiles forward only coordinates and key-file paths, never keys. + while IFS= read -r _k; do + env_line + env_value + export "$_k=$_v" + done < <(sed -nE 's/^(OUTERLOOP_ENDPOINT_[A-Z][A-Z0-9_]*_(URL|KEY_FILE|MODEL|API))=.*/\1/p' "$ENV_FILE" | sort -u) for _k in OUTERLOOP_CLAUDE_VERSION OUTERLOOP_CODEX_VERSION \ OUTERLOOP_CLAUDE_SHA256 OUTERLOOP_CODEX_SHA256 \ OUTERLOOP_HERMES_REF OUTERLOOP_HERMES_SHA \ OUTERLOOP_CACHE_ROOT REVIEW_BACKEND \ + OUTERLOOP_AUTHOR_ENDPOINT REVIEW_ENDPOINT REVIEW_MODEL \ OUTERLOOP_AUTHOR_BACKEND OUTERLOOP_AUTHOR_MODEL OUTERLOOP_CLAUDE_MODEL \ OUTERLOOP_CLAUDE_BIN OUTERLOOP_CODEX_BIN OUTERLOOP_CODEX_KEY_FILE \ OUTERLOOP_CLAUDE_KEY_FILE OUTERLOOP_STEWARD_KEY_FILE \ diff --git a/src/outerloop/attempt.py b/src/outerloop/attempt.py index a5f3e24b..4dc0e378 100644 --- a/src/outerloop/attempt.py +++ b/src/outerloop/attempt.py @@ -45,6 +45,7 @@ should_dispatch, snapshot_tree, ) +from outerloop.endpoints import author_model_setting, model_key, resolve_endpoint, split_endpoint from outerloop.evalcache import seed_dir from outerloop.github import ( GitError, @@ -175,6 +176,22 @@ def codex_author_config_error(backend: str, model: str, image: str) -> str: passes the PARKED RUN's persisted pair — so backend and model are checked as a unit and never a fleet backend against a run's model. codex writes+executes, so it must be contained (--image) and needs a non-claude model.""" + try: + _, profile = resolve_endpoint(model, backend) + except ValueError as exc: + return str(exc) + if profile: + if backend != "claude" and not image: + return f"author-backend {backend} requires --image (it runs contained)" + if backend == "hermes": + from outerloop.hermes_install import hermes_ready + + repo = os.environ.get("REVIEW_HERMES_REPO", "") + if not repo or not hermes_ready(Path(repo).expanduser()): + return "hermes author needs REVIEW_HERMES_REPO with pinned source and runtime" + return "" + if backend == "hermes": + return "hermes author requires OUTERLOOP_AUTHOR_ENDPOINT" if backend not in ("claude", "codex"): # a typo'd OUTERLOOP_AUTHOR_BACKEND passes the env DEFAULT silently # (argparse validates the flag, not its default) and the climb rejects it @@ -203,7 +220,7 @@ def fleet_author_model(backend: str) -> str: the deployment's Claude model for a claude author (ClaudeModelUnset when that is missing too). Another backend without OUTERLOOP_AUTHOR_MODEL gets "", and codex_author_config_error names the fix.""" - model = os.environ.get("OUTERLOOP_AUTHOR_MODEL", "") + model = author_model_setting(backend, os.environ.get("OUTERLOOP_AUTHOR_MODEL", "")) if not model and backend == "claude": return default_claude_model() return model @@ -224,14 +241,18 @@ def resume_author( else the fleet model when the fleet runs the same backend (so the configured author model applies to legacy claude records too); a claude record under a codex fleet falls back to the claude default, and a codex - record to the fleet model only as a last resort (codex records always - carry their model). + record to the native fleet model only as a last resort. Endpoint fleet + selectors are never inherited by a record missing its route; a missing + native Codex model fails preflight rather than changing providers. The key file is the exact resolved path the run used (so an explicit --key-file survives), falling back to the per-backend resolution for legacy records that never recorded it.""" backend = getattr(record, "author_backend", "") or "claude" model = getattr(record, "author_model", "") if not model: + # A missing route cannot opt into an endpoint via fleet defaults. + if fleet_model and split_endpoint(fleet_model)[1]: + fleet_model = "" if explicit_model: model = explicit_model elif backend == "claude": @@ -3151,7 +3172,11 @@ def _panel_lenses_from_args( parsed = resolve_lenses(args.panel, author_backend, author_model) # the anthropic panel key is read only when a claude lens will use it — # a codex-only panel must not demand an unrelated credential - panel_key = role_key(args.panel_key_file) if any(b == "claude" for _, b, _ in parsed) else "" + panel_key = ( + role_key(args.panel_key_file) + if any(b == "claude" and not split_endpoint(m)[1] for _, b, m in parsed) + else "" + ) lenses = [] secrets: list[str] = [panel_key] if panel_key else [] for kind, backend, model in parsed: @@ -3160,7 +3185,28 @@ def _panel_lenses_from_args( # anthropic panel key, and role separation forbids defaulting to the # AUTHOR's codex key: the judge key is its own, named explicitly claude_panel_path = Path(args.panel_key_file or PANEL_KEY_DEFAULT).expanduser() - if backend == "codex": + _, endpoint = resolve_endpoint(model, backend) + if endpoint: + if not args.image: + raise ValueError(f"a {backend} endpoint panel lens requires --image") + from outerloop.endpoints import validate_judge_key_file + + _, author_profile = resolve_endpoint(author_model, author_backend) + validate_judge_key_file( + endpoint, + author_profile.key_file + if author_profile + else (getattr(args, "key_file", "") or resolve_author_key_file(author_backend)), + claude_panel_path, + ) + lens_key = endpoint.key() + author_key_file = getattr(args, "key_file", "") or resolve_author_key_file( + author_backend + ) + if lens_key == model_key(author_key_file, author_backend, author_model): + raise ValueError("a panel judge key is the author key (role separation)") + secrets.append(lens_key) + elif backend == "codex": lens_key = _judge_lens_key( backend="codex", key_file_env="OUTERLOOP_PANEL_CODEX_KEY_FILE", @@ -3210,6 +3256,9 @@ def _panel_lenses_from_args( except (ValueError, ClaudeModelUnset) as exc: raise ValueError(f"panel entry {kind}:{backend}: {exc}") from exc lenses.append(PanelLens(kind=kind, harness=judge)) + _, author_endpoint = resolve_endpoint(author_model, author_backend) + if author_endpoint and author_endpoint.key() in secrets: + raise ValueError("a panel judge key is the author key (role separation)") return tuple(lenses), tuple(dict.fromkeys(secrets)) @@ -4816,7 +4865,7 @@ def _run_id(value: str) -> str: ) parser.add_argument( "--author-backend", - choices=("claude", "codex"), + choices=("claude", "codex", "hermes"), default=os.environ.get("OUTERLOOP_AUTHOR_BACKEND") or "claude", help="agent backend for the author/editor role (config-driven: default " "from OUTERLOOP_AUTHOR_BACKEND). codex runs contained (apptainer + " @@ -4884,7 +4933,7 @@ def _run_id(value: str) -> str: # OUTERLOOP_CLAUDE_MODEL fails here with the fix named, not with a traceback try: args.model = fleet_author_model(args.author_backend) - except ClaudeModelUnset as exc: + except (ClaudeModelUnset, ValueError) as exc: parser.error(str(exc)) if args.resume: _attach_run_log(run_dir_of(args.run_root, args.resume)) @@ -4935,7 +4984,8 @@ def _run_id(value: str) -> str: try: explicit_model = args.model # what the operator typed, before any env fill-in if not getattr(_wake_record, "author_model", "") and not args.model: - args.model = fleet_author_model(args.author_backend) + native_model = os.environ.get("OUTERLOOP_AUTHOR_MODEL", "") + args.model = "" if split_endpoint(native_model)[1] else native_model wake_backend, wake_model, wake_key_file = resume_author( _wake_record, args.model, args.author_backend, explicit_model ) @@ -4984,7 +5034,7 @@ def _run_id(value: str) -> str: or _wake_stage.get("submitted") or getattr(_wake_record, "pr_url", "") ): - wake_api_key = role_key(wake_key_file, wake_backend) + wake_api_key = model_key(wake_key_file, wake_backend, wake_model) if wake_api_key and wake_api_key in wake_panel_secrets: args.panel_skip = "a panel judge key is this run's author key (role separation)" wake_lenses = () @@ -5004,6 +5054,9 @@ def _run_id(value: str) -> str: model=wake_model, container_image=args.image, codex_extra_args=codex_extra, + hermes_repo=Path(os.environ["REVIEW_HERMES_REPO"]) + if wake_backend == "hermes" + else None, ) wake_secrets = tuple(k for k in (bot_auth.token(), *wake_panel_secrets, wake_api_key) if k) try: @@ -5052,15 +5105,24 @@ def _run_id(value: str) -> str: # a fresh climb authors on the FLEET's configured backend; validate it (codex # writes+executes, so --image + a non-claude model) before any spend. + try: + args.model = author_model_setting(args.author_backend, args.model) + except ValueError as exc: + parser.error(str(exc)) _err = codex_author_config_error(args.author_backend, args.model, args.image) if _err: parser.error(_err) # config-driven: the author key defaults per backend (claude vs codex) so the # tick never threads it — see resolve_author_key_file (result is ~-expanded). - args.key_file = resolve_author_key_file(args.author_backend, args.key_file) + _, author_endpoint = resolve_endpoint(args.model, args.author_backend) + args.key_file = ( + str(author_endpoint.key_file) + if author_endpoint + else resolve_author_key_file(args.author_backend, args.key_file) + ) # same 0600 discipline as the PAT: this key spends real money. A missing # file is tolerated only when Vertex (ADC) covers the claude backend. - api_key = role_key(args.key_file, args.author_backend) + api_key = model_key(args.key_file, args.author_backend, args.model) stamp = datetime.now(UTC).strftime("%Y%m%d-%H%M%S") # the agent id keeps concurrent same-benchmark slots (the width dial's # portfolio case) from minting one run directory in the same second @@ -5130,6 +5192,9 @@ def _run_id(value: str) -> str: model=args.model, container_image=args.image, codex_extra_args=codex_extra, + hermes_repo=Path(os.environ["REVIEW_HERMES_REPO"]) + if args.author_backend == "hermes" + else None, ), spec=spec, panel_lenses=panel_lenses, diff --git a/src/outerloop/cli.py b/src/outerloop/cli.py index d8fa2e0b..04c2671d 100644 --- a/src/outerloop/cli.py +++ b/src/outerloop/cli.py @@ -26,6 +26,7 @@ from typing import TYPE_CHECKING from outerloop import paths +from outerloop.endpoints import author_model_setting, endpoint_config_key from outerloop.harness import HARNESS_INSTALL, default_binary if TYPE_CHECKING: @@ -59,6 +60,9 @@ "OUTERLOOP_HERMES_SHA", "OUTERLOOP_CACHE_ROOT", "REVIEW_BACKEND", + "OUTERLOOP_AUTHOR_ENDPOINT", + "REVIEW_ENDPOINT", + "REVIEW_MODEL", "OUTERLOOP_AUTHOR_BACKEND", "OUTERLOOP_AUTHOR_MODEL", "OUTERLOOP_CLAUDE_MODEL", @@ -120,7 +124,11 @@ def env_file_values( continue key, value = line.split("=", 1) key = key.strip() - if keys is not None and key not in keys: + if ( + keys is not None + and key not in keys + and not ("OUTERLOOP_AUTHOR_ENDPOINT" in keys and endpoint_config_key(key)) + ): continue value = value.strip() if len(value) >= 2 and value[0] == value[-1] and value[0] in "\"'": @@ -397,6 +405,15 @@ def missing_harness_binary(values: Mapping[str, str], environ: Mapping[str, str] """Check only the configured author's host CLI, using the harness's lookup.""" backend = _setting_of("OUTERLOOP_AUTHOR_BACKEND", values, environ).lower() or "claude" key = f"OUTERLOOP_{backend.upper()}_BIN" + if backend == "hermes": + from outerloop.hermes_install import hermes_ready + + repo = _setting_of("REVIEW_HERMES_REPO", values, environ) + return ( + "" + if repo and hermes_ready(Path(repo).expanduser()) + else "hermes author needs REVIEW_HERMES_REPO with pinned source and runtime" + ) if key not in HARNESS_BIN_KEYS: hint = ( f" Hermes is a review backend; install its source with `{HARNESS_INSTALL['hermes']}`." @@ -444,7 +461,11 @@ def missing_panel_model(values: dict[str, str], environ: Mapping[str, str]) -> s from outerloop.panel import resolve_lenses try: - resolve_lenses(panel, backend, _setting_of("OUTERLOOP_AUTHOR_MODEL", values, environ)) + env = {**values, **environ} + model = author_model_setting( + backend, _setting_of("OUTERLOOP_AUTHOR_MODEL", values, environ), env + ) + resolve_lenses(panel, backend, model, environ=env) except ValueError as exc: return str(exc) return "" @@ -461,7 +482,11 @@ def missing_claude_model(values: Mapping[str, str], environ: Mapping[str, str]) return "" roles: list[str] = [] backend = _setting_of("OUTERLOOP_AUTHOR_BACKEND", values, environ).lower() or "claude" - if backend == "claude" and not _setting_of("OUTERLOOP_AUTHOR_MODEL", values, environ): + if ( + backend == "claude" + and not _setting_of("OUTERLOOP_AUTHOR_MODEL", values, environ) + and not _setting_of("OUTERLOOP_AUTHOR_ENDPOINT", values, environ) + ): roles.append("the claude author (no OUTERLOOP_AUTHOR_MODEL)") panel = _configured("OUTERLOOP_PANEL", values, environ) panel = DEFAULT_PANEL if panel is None else panel.strip() @@ -470,7 +495,10 @@ def missing_claude_model(values: Mapping[str, str], environ: Mapping[str, str]) try: lenses = resolve_lenses( - panel, backend, _setting_of("OUTERLOOP_AUTHOR_MODEL", values, environ) + panel, + backend, + _setting_of("OUTERLOOP_AUTHOR_MODEL", values, environ), + environ={**values, **environ}, ) except ValueError: lenses = () # missing_panel_model reports invalid panel configuration @@ -558,6 +586,14 @@ def start(args: argparse.Namespace) -> int: try: values = env_file_values(ENV_FILE, START_KEYS + TICK_ENV_KEYS) # one read for everything problem = "" if args.dry_run else missing_harness_binary(values, os.environ) + try: + author_model_setting( + _setting_of("OUTERLOOP_AUTHOR_BACKEND", values, os.environ) or "claude", + _setting_of("OUTERLOOP_AUTHOR_MODEL", values, os.environ), + {**values, **os.environ}, + ) + except ValueError as exc: + raise StartError(str(exc)) from exc # the model check holds for --dry-run too: it is configuration, not a host lookup problem = problem or missing_claude_model(values, os.environ) problem = problem or missing_panel_model(values, os.environ) @@ -631,7 +667,7 @@ def start(args: argparse.Namespace) -> int: # export from .env each tick are exported here once; the shell wins env = {**os.environ, **path_env} for key, value in values.items(): - if key in TICK_ENV_KEYS: + if key in TICK_ENV_KEYS or endpoint_config_key(key): env.setdefault(key, value) env["OUTERLOOP_COMPUTE"] = "local" if plan.mode == "local" else "slurm" env.pop("OUTERLOOP_TICK_HOST", None) # the plan decided; nothing inherited diff --git a/src/outerloop/compute.py b/src/outerloop/compute.py index 9f7ea059..bd98be6c 100644 --- a/src/outerloop/compute.py +++ b/src/outerloop/compute.py @@ -834,7 +834,19 @@ def _secret_name(name: str) -> bool: job_env = { k: v for k, v in os.environ.items() - if k in ("PATH", "HOME", "LANG", "TMPDIR", "SLURM_TMPDIR", "USER", "LOGNAME") + if k + in ( + "PATH", + "HOME", + "LANG", + "TMPDIR", + "SLURM_TMPDIR", + "USER", + "LOGNAME", + "REVIEW_ENDPOINT", + "REVIEW_MODEL", + "REVIEW_BACKEND", + ) or (k.startswith(("OUTERLOOP_", "REVIEW_HERMES_")) and not _secret_name(k)) } job_env.update(cache_environment(os.environ)) diff --git a/src/outerloop/endpoints.py b/src/outerloop/endpoints.py new file mode 100644 index 00000000..75182549 --- /dev/null +++ b/src/outerloop/endpoints.py @@ -0,0 +1,137 @@ +"""Named, file-authenticated endpoints shared by every role and backend.""" + +from __future__ import annotations + +import os +import re +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path +from urllib.parse import urlsplit + +from outerloop.github import FileTokenProvider + +PROFILE_KEY = re.compile(r"OUTERLOOP_ENDPOINT_[A-Z][A-Z0-9_]*_(URL|KEY_FILE|MODEL|API)\Z") +PROFILE_NAME = re.compile(r"[A-Za-z][A-Za-z0-9_]*\Z") + + +def endpoint_config_key(key: str) -> bool: + return PROFILE_KEY.fullmatch(key) is not None + + +def split_endpoint(model: str) -> tuple[str, str]: + """Only an explicit bracketed selector changes a native model's route.""" + if "[endpoint=" not in model: + return model, "" + match = re.fullmatch(r"([^\[\]]*)\[endpoint=([A-Za-z][A-Za-z0-9_]*)\]", model) + if match is None: + raise ValueError("endpoint selector must be [endpoint=]") + return match[1], match[2].lower() + + +@dataclass(frozen=True) +class EndpointProfile: + name: str + url: str + key_file: Path + model: str + apis: tuple[str, ...] + + def key(self) -> str: + try: + return FileTokenProvider(self.key_file).token() + except (OSError, ValueError) as exc: + raise ValueError(f"endpoint {self.name!r}: {exc}") from exc + + +def endpoint_profile( + name: str, backend: str, model: str = "", environ: Mapping[str, str] | None = None +) -> EndpointProfile: + env = os.environ if environ is None else environ + if backend not in ("claude", "codex", "hermes"): + raise ValueError(f"endpoint {name!r}: unsupported backend {backend!r}") + if not PROFILE_NAME.fullmatch(name): + raise ValueError(f"invalid endpoint profile name {name!r}") + prefix = f"OUTERLOOP_ENDPOINT_{name.upper()}_" + values = { + suffix: env.get(prefix + suffix, "").strip() + for suffix in ("URL", "KEY_FILE", "MODEL", "API") + } + if not any(values.values()): + raise ValueError(f"unknown endpoint profile {name!r}") + for suffix, value in values.items(): + if not value: + raise ValueError(f"endpoint {name!r}: missing {prefix}{suffix}") + apis = tuple(part.strip() for part in values["API"].split(",")) + required = {"claude": "anthropic", "codex": "responses", "hermes": "chat"}[backend] + if any(api not in ("anthropic", "responses", "chat") for api in apis): + raise ValueError(f"endpoint {name!r}: API must list anthropic, responses, or chat") + if required not in apis: + raise ValueError(f"endpoint {name!r}: {backend} requires API {required}") + url = urlsplit(values["URL"]) + if ( + url.scheme not in ("http", "https") + or not url.hostname + or url.username + or url.password + or url.query + or url.fragment + ): + raise ValueError( + f"endpoint {name!r}: URL must be an HTTP(S) base URL " + "without credentials, query or fragment" + ) + path = Path(values["KEY_FILE"]).expanduser() + if not path.is_absolute(): + raise ValueError(f"endpoint {name!r}: KEY_FILE must be absolute") + if ( + "[" in values["MODEL"] + or "]" in values["MODEL"] + or any(c in values["MODEL"] for c in "\r\n") + ): + raise ValueError(f"endpoint {name!r}: invalid served model") + if model and model != values["MODEL"]: + raise ValueError( + f"endpoint {name!r}: model {model!r} does not match served model {values['MODEL']!r}" + ) + profile = EndpointProfile(name.lower(), values["URL"], path, values["MODEL"], apis) + profile.key() # Validate before any session or intake claim. + return profile + + +def resolve_endpoint( + model: str, backend: str, name: str = "", environ: Mapping[str, str] | None = None +) -> tuple[str, EndpointProfile | None]: + served, selected = split_endpoint(model) + if name and selected and name.lower() != selected: + raise ValueError("conflicting endpoint selectors") + selected = selected or name + profile = endpoint_profile(selected, backend, served, environ) if selected else None + return (profile.model if profile else served), profile + + +def author_model_setting(backend: str, model: str, environ: Mapping[str, str] | None = None) -> str: + env = os.environ if environ is None else environ + served, profile = resolve_endpoint( + model, backend, env.get("OUTERLOOP_AUTHOR_ENDPOINT", "").strip(), env + ) + return f"{served}[endpoint={profile.name}]" if profile else served + + +def model_key(key_file: str | Path, backend: str, model: str) -> str: + from outerloop.role_runner import role_key + + _, profile = resolve_endpoint(model, backend) + return profile.key() if profile else role_key(key_file, backend) + + +def validate_judge_key_file( + profile: EndpointProfile, author_path: str | Path, claude_panel_path: str | Path +) -> None: + """Profiles obey the same file isolation as conventional judge credentials.""" + for label, path in (("author", author_path), ("claude panel", claude_panel_path)): + other = Path(path).expanduser() + if profile.key_file.resolve() == other.resolve() or ( + other.exists() and profile.key_file.samefile(other) + ): + raise ValueError(f"endpoint judge key file is the {label} key file (role separation)") diff --git a/src/outerloop/harness.py b/src/outerloop/harness.py index e825b2de..640b979b 100644 --- a/src/outerloop/harness.py +++ b/src/outerloop/harness.py @@ -27,6 +27,7 @@ from pathlib import Path from typing import Any, Protocol +from outerloop.endpoints import EndpointProfile from outerloop.hermes_install import hermes_ready, hermes_runtime from outerloop.image import apptainer_from_env @@ -573,6 +574,7 @@ class ClaudeCodeHarness: # Claude-on-Vertex (ADC) instead of the Anthropic API key; the api_key is # ignored when set. Contained sessions get the ADC file bind-mounted. vertex: VertexConfig | None = None + endpoint: EndpointProfile | None = None CONTAINER_CLAUDE = "/opt/agent/claude" CONTAINER_ADC = "/opt/agent/adc.json" @@ -664,7 +666,33 @@ def run( # children included) in one process group we can kill as a unit — # a timed-out session must not leave orphans holding the API key # and writing into the clone. - if self.vertex is not None: + if self.endpoint: + env = session_env(self.api_key, "ANTHROPIC_AUTH_TOKEN", session_home) + env.update( + { + # Claude Code appends /v1/messages itself; profiles use the + # OpenAI-style base, so drop a trailing /v1 + "ANTHROPIC_BASE_URL": self.endpoint.url.rstrip("/").removesuffix("/v1"), + "CLAUDE_CODE_USE_VERTEX": "0", + "CLAUDE_CODE_USE_BEDROCK": "0", + "CLAUDE_CODE_USE_FOUNDRY": "0", + "CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC": "1", + "ANTHROPIC_SMALL_FAST_MODEL": self.model, + **{ + f"ANTHROPIC_DEFAULT_{size}_MODEL": self.model + for size in ("OPUS", "SONNET", "HAIKU") + }, + } + ) + if self.container_image: + env.update( + { + f"APPTAINERENV_{k}": v + for k, v in list(env.items()) + if k not in SESSION_ENV_ALLOWLIST and k != "HOME" + } + ) + elif self.vertex is not None: # vertex auth: ADC only — ANTHROPIC_API_KEY is deliberately # absent so the CLI cannot fall back to direct billing env = session_env("", "ANTHROPIC_API_KEY", session_home) @@ -952,6 +980,7 @@ class CodexHarness: api_key: str binary: str = "codex" + endpoint: EndpointProfile | None = None # empty -> codex's configured default (a wrong id 404s); pin only a verified one model: str = "" # the deployment's container or ephemeral runner is the boundary; codex's own @@ -1094,11 +1123,36 @@ def run( finally: os.close(codex_fd) os.close(home_fd) - # Codex authenticates from ~/.codex/auth.json, not OPENAI_API_KEY alone - # (the responses endpoint 401s on env-only). Write auth.json - # into the scrubbed per-run HOME with `codex login --with-api-key` - # (key on stdin, never argv) before exec. - login_error = self._login(session_home) + # Native OpenAI sessions log in via stdin; endpoint sessions use env_key. + if self.endpoint: + codex_dir = session_home / ".codex" + try: + codex_dir.mkdir(mode=0o700, exist_ok=True) + except OSError: + return _error_result("workspace-error", detail="could not create codex config dir") + config = ( + 'model_provider = "outerloop_endpoint"\n' + "[model_providers.outerloop_endpoint]\n" + 'name = "Outerloop endpoint"\n' + f"base_url = {json.dumps(self.endpoint.url)}\n" + 'env_key = "OUTERLOOP_SESSION_KEY"\n' + 'wire_api = "responses"\n' + "requires_openai_auth = false\n" + ) + if not _write_private_fixed(codex_dir / "config.toml", config): + return _error_result("workspace-error", detail="could not seed codex config") + self._purge_auth(session_home) + else: + config_path = session_home / ".codex/config.toml" + previous = _read_no_follow(config_path) or "" + if previous.startswith('model_provider = "outerloop_endpoint"\n'): + try: + config_path.unlink() + except OSError: + return _error_result( + "workspace-error", detail="could not clear endpoint config" + ) + login_error = None if self.endpoint else self._login(session_home) if login_error is not None: return login_error # --output-last-message target lives inside the per-run home (0700), @@ -1125,9 +1179,10 @@ def run( else codex_argv ) try: - env = session_env(self.api_key, "OPENAI_API_KEY", session_home) + key_env = "OUTERLOOP_SESSION_KEY" if self.endpoint else "OPENAI_API_KEY" + env = session_env(self.api_key, key_env, session_home) if self.container_image: - env["APPTAINERENV_OPENAI_API_KEY"] = self.api_key + env[f"APPTAINERENV_{key_env}"] = self.api_key process = subprocess.Popen( command, cwd=workspace, @@ -1189,8 +1244,8 @@ def _hermes_command( disabled_toolsets: tuple[str, ...], extra_args: tuple[str, ...], ) -> list[str]: - """Argv for one headless hermes run (`run_agent.py`, fire-style flags, - hermes-agent v0.20.1). + """Argv for one headless hermes run (`run_agent.py`, argparse flags, + hermes-agent v2026.9.24). The BRIEF is never in argv — it is written to a file and `query` is only a short pointer instruction. The API key is never in argv either: hermes @@ -1210,13 +1265,11 @@ def _hermes_command( argv.append(f"--model={model}") if base_url: argv.append(f"--base_url={base_url}") - # Embedded quotes are load-bearing: fire literal-evals flag values, so a - # bare `a,b` becomes a Python TUPLE and hermes's .split(",") crashes. - # `"a,b"` evals to the string hermes expects. + # argparse receives these directly; shell/Fire quoting would become literal. if enabled_toolsets: - argv.append(f'--enabled_toolsets="{",".join(enabled_toolsets)}"') + argv.append(f"--enabled_toolsets={','.join(enabled_toolsets)}") if disabled_toolsets: - argv.append(f'--disabled_toolsets="{",".join(disabled_toolsets)}"') + argv.append(f"--disabled_toolsets={','.join(disabled_toolsets)}") return [*argv, *extra_args] @@ -1237,7 +1290,7 @@ def _parse_hermes_result( messages = sample elif isinstance(sample, dict): # run_agent.py --save_sample wraps the ShareGPT turns under - # "conversations" (hermes v0.20.1); accept "messages"/"trajectory" + # "conversations" (hermes v2026.9.24); accept "messages"/"trajectory" # too for other paths. Missing this key makes num_turns==0 and # drops a real verdict as a bogus error. wrapped = ( @@ -1308,6 +1361,7 @@ class HermesHarness: # none, and it refuses to run "unconfigured"); when set, the harness # pre-seeds a minimal config in the per-run home provider: str = "" + endpoint: EndpointProfile | None = None # approvals.deny: fnmatch globs hermes refuses before any yolo/mode-off # bypass (headless: a clean deny, never a hang). NOTE it matches SHELL # COMMANDS (the terminal tool), NOT the write_file/patch tool calls. @@ -1380,9 +1434,18 @@ def run( hermes_dir = session_home / ".hermes" try: hermes_dir.mkdir(mode=0o700, exist_ok=True) - config_lines = ["model:\n", f' provider: "{self.provider}"\n'] + config_lines = ["model:\n", f" provider: {json.dumps(self.provider)}\n"] if self.model: - config_lines.insert(1, f' default: "{self.model}"\n') + config_lines.insert(1, f" default: {json.dumps(self.model)}\n") + if self.endpoint: + config_lines.append(" reasoning_echo: true\n") + config_lines.append( + "custom_providers:\n" + f" - name: {json.dumps(self.provider)}\n" + f" base_url: {json.dumps(self.endpoint.url)}\n" + f" key_env: {json.dumps(self.key_env)}\n" + " api_mode: chat_completions\n" + ) if self.approvals_deny: config_lines.append("approvals:\n deny:\n") config_lines += [f' - "{glob}"\n' for glob in self.approvals_deny] @@ -1405,7 +1468,7 @@ def run( repo, query, self.model, - self.base_url, + "" if self.endpoint else self.base_url, self.max_turns, self.enabled_toolsets, self.disabled_toolsets, @@ -1439,10 +1502,12 @@ def run( ] try: env = session_env(self.api_key, self.key_env, session_home) + env["TERMINAL_CWD"] = str(workspace.resolve()) if self.container_image: # --cleanenv drops the host env except APPTAINERENV_*: the key # travels via the environment, never argv env[f"APPTAINERENV_{self.key_env}"] = env[self.key_env] + env["APPTAINERENV_TERMINAL_CWD"] = env["TERMINAL_CWD"] # cwd is the per-run home, NOT the workspace: --save_sample writes # its trajectory JSON to cwd, and artifacts must never land in the # clone (they would enter the diff). diff --git a/src/outerloop/harnesses.toml b/src/outerloop/harnesses.toml index ffeba3f2..7f543636 100644 --- a/src/outerloop/harnesses.toml +++ b/src/outerloop/harnesses.toml @@ -10,5 +10,5 @@ version = "0.130.0" sha256 = "16779e7b7857508a768a36d7d4e084eec336ec23946ed70a9b09489b8f861190" [hermes] -ref = "v2026.8.13" -sha = "f80f453ae0679347e38abc917c7f94f717bf96c5" +ref = "v2026.9.24" +sha = "f97608f178d1ffeca59860195ab7da295f7c8e5f" diff --git a/src/outerloop/panel.py b/src/outerloop/panel.py index 970e197c..4d714460 100644 --- a/src/outerloop/panel.py +++ b/src/outerloop/panel.py @@ -15,9 +15,11 @@ from __future__ import annotations import logging +from collections.abc import Mapping from dataclasses import dataclass from pathlib import Path +from outerloop.endpoints import resolve_endpoint, split_endpoint from outerloop.harness import Harness, backend_id from outerloop.review import Finding, PullRequest, build_agent_brief from outerloop.role_runner import run_role @@ -36,7 +38,7 @@ def parse_lenses(panel: str, default_backend: str = "claude") -> tuple[tuple[str, str, str], ...]: - """Parse a panel spec — comma-separated ``kind[:backend[:model]]`` — into + """Parse a panel spec — comma-separated ``kind[:backend[:model[endpoint=profile]]]`` — into (kind, backend, model) triples, or raise ValueError. A lens that names no backend takes `default_backend`: callers pass the author's backend, so a codex deployment gets codex judges by default and a claude deployment @@ -64,12 +66,13 @@ def parse_lenses(panel: str, default_backend: str = "claude") -> tuple[tuple[str raise ValueError( f"panel entry {entry!r}: unknown backend {backend!r} (claude, codex, hermes)" ) + split_endpoint(model) # Validate syntax without reading deployment credentials. entries.append((kind, backend, model)) return tuple(entries) def resolve_lenses( - panel: str, author_backend: str, author_model: str + panel: str, author_backend: str, author_model: str, *, environ: Mapping[str, str] | None = None ) -> tuple[tuple[str, str, str], ...]: """Resolve lenses against their author; other backends require an explicit model. @@ -83,9 +86,18 @@ def resolve_lenses( # model nobody chose resolved = [] for kind, backend, model in parsed: + served, profile = resolve_endpoint(model, backend, environ=environ) + if profile: + resolved.append((kind, backend, f"{served}[endpoint={profile.name}]")) + continue if not model: if backend == author_backend: # inherit, even when the author itself runs its CLI's default + if split_endpoint(author_model)[1]: + raise ValueError( + f"panel lens {kind}:{backend} must select its own judge endpoint " + "([endpoint=]) or an explicit model" + ) model = author_model else: raise ValueError( diff --git a/src/outerloop/review_agent.py b/src/outerloop/review_agent.py index 8ea38532..1a962545 100644 --- a/src/outerloop/review_agent.py +++ b/src/outerloop/review_agent.py @@ -23,6 +23,7 @@ backend_id, budget_exhausted, outage, + redact, ) from outerloop.posting import ( EXPECTED_FAILURES, @@ -37,7 +38,7 @@ format_review, skip_reason, ) -from outerloop.role_runner import run_role +from outerloop.role_runner import redact_role_result, run_role from outerloop.roles import review_result_from_role, reviewer_spec from outerloop.rolespec import RoleSpec @@ -48,7 +49,19 @@ # .claude/settings.json hooks that execute commands), so the CLIs rename them # before any session starts. Renamed — not deleted — so a judge can still read # them as data. Backend-agnostic defense in depth behind claude's --bare. -INSTRUCTION_FILES = ("CLAUDE.md", "AGENTS.md", ".claude", ".mcp.json") +INSTRUCTION_FILES = ( + "CLAUDE.md", + "claude.md", + "AGENTS.md", + "agents.md", + "AGENTS.override.md", + ".hermes.md", + "HERMES.md", + ".cursorrules", + ".cursor", + ".claude", + ".mcp.json", +) SANITIZED_SUFFIX = ".pr-data" @@ -169,13 +182,16 @@ def run_agent_review( from outerloop.syscall import tool_command brief = build_agent_brief(pr, today, syscall_cmd=tool_command(workspace), lens=lens) - role_result = run_role(spec, harness, brief, workspace) + role_result = redact_role_result(run_role(spec, harness, brief, workspace), harness) review = review_result_from_role(role_result) if review is None: # No verdict: an errored or refused session, not a clean read. An # API outage or a budget-exhausted session (walltime/turns) says so # on the thread; other failures are logged, advisory-silent. - detail = role_result.error or role_result.session.stop_reason + detail = redact( + role_result.error or role_result.session.stop_reason, + (getattr(harness, "api_key", ""),), + ) log.warning("agent review produced no verdict on %s#%s: %s", repo, number, detail) # `detail` is already api-key-redacted by the harness (it owns # its own secret), so no secrets are passed here. @@ -246,7 +262,8 @@ def run_agent_review( ) return round_label except EXPECTED_FAILURES as exc: # advisory: never fail the target repo's CI - log.warning("agent review did not complete: %s: %s", type(exc).__name__, exc) + detail = redact(f"{type(exc).__name__}: {exc}", (getattr(harness, "api_key", ""),)) + log.warning("agent review did not complete: %s", detail) if emit_path is not None: # the invariant holds here too: the workflow backstop would cover # a missing file, but with a generic detail — the real failure is @@ -257,7 +274,7 @@ def run_agent_review( repo, number, kind="skip-stub", - detail=f"{type(exc).__name__}: {exc}", + detail=detail, reviewed_by=backend_id(harness), ) return None diff --git a/src/outerloop/review_agent_cli.py b/src/outerloop/review_agent_cli.py index daede2e0..27288c39 100644 --- a/src/outerloop/review_agent_cli.py +++ b/src/outerloop/review_agent_cli.py @@ -12,6 +12,7 @@ import sys from pathlib import Path +from outerloop.endpoints import resolve_endpoint from outerloop.github import EnvTokenProvider, GitHubClient from outerloop.harness import ClaudeModelUnset, Harness from outerloop.review_agent import ( @@ -53,6 +54,24 @@ def resolve_reviewer_harness(spec: RoleSpec) -> tuple[Harness | None, str, str]: never disagree about a value.""" backend = os.environ.get("REVIEW_BACKEND", "claude").lower() review_model = os.environ.get("REVIEW_MODEL", "").strip() + try: + _, endpoint = resolve_endpoint( + review_model, backend, os.environ.get("REVIEW_ENDPOINT", "").strip() + ) + if endpoint: + repo = os.environ.get("REVIEW_HERMES_REPO", "").strip() + harness = build_harness( + "", + spec, + backend=backend, + model=review_model or None, + endpoint=endpoint.name, + binary=os.environ.get("REVIEW_BINARY") or None, + hermes_repo=Path(repo).expanduser() if repo else None, + ) + return harness, "", backend + except ValueError as exc: + return None, str(exc), backend hermes_provider = os.environ.get("REVIEW_HERMES_PROVIDER", "").lower() or "openrouter" key_var = { "claude": "ANTHROPIC_REVIEWER_KEY", diff --git a/src/outerloop/role_runner.py b/src/outerloop/role_runner.py index b604a5e2..a12a2540 100644 --- a/src/outerloop/role_runner.py +++ b/src/outerloop/role_runner.py @@ -17,10 +17,11 @@ from __future__ import annotations import logging -from dataclasses import dataclass +from dataclasses import dataclass, replace from pathlib import Path from typing import Any +from outerloop.endpoints import resolve_endpoint from outerloop.harness import ( ClaudeCodeHarness, CodexHarness, @@ -94,6 +95,7 @@ def build_harness( codex_extra_args: tuple[str, ...] = (), hermes_repo: Path | None = None, hermes_provider: str = "", + endpoint: str = "", ) -> Harness: """Construct the harness for any role on any backend — the ONE deployment wiring (`spec.tools` → native flags, `spec.budget` → turns/walltime, @@ -115,6 +117,9 @@ def build_harness( deployment has a jail (the cluster), pass none where the runner itself is the ephemeral boundary (CI). The tokenless split keeps credentials out of the session either way.""" + model, profile = resolve_endpoint(model or "", backend, endpoint) + if profile: + api_key = profile.key() if backend == "codex": # the web, when the spec grants it: the config override, because # `codex exec` does not accept `--search` @@ -123,6 +128,7 @@ def build_harness( api_key=api_key, binary=binary or "codex", model=model or "", # "" -> codex's configured default; pin a verified id + endpoint=profile, sandbox="danger-full-access", timeout_s=spec.budget.walltime_s, container_image=container_image, @@ -131,9 +137,13 @@ def build_harness( if backend == "hermes": if hermes_repo is None: raise ValueError("hermes backend needs hermes_repo (the pinned clone)") - if (hermes_provider or "openrouter") not in _HERMES_PROVIDERS: + if not profile and (hermes_provider or "openrouter") not in _HERMES_PROVIDERS: raise ValueError(f"unknown hermes provider: {hermes_provider!r}") - seed, key_env = _HERMES_PROVIDERS[hermes_provider or "openrouter"] + seed, key_env = ( + ("outerloop_endpoint", "OUTERLOOP_SESSION_KEY") + if profile + else _HERMES_PROVIDERS[hermes_provider or "openrouter"] + ) # `terminal` (the shell) is keyed on the SAME signal claude uses — the # spec granting the Bash tool — not on can_execute, so every backend # gives a role the same shell/no-shell whether or not those two ever @@ -146,6 +156,7 @@ def build_harness( key_env=key_env, repo_dir=hermes_repo, provider=seed, + endpoint=profile, model=model or "", max_turns=spec.budget.max_turns, timeout_s=spec.budget.walltime_s, @@ -170,7 +181,8 @@ def build_harness( # Vertex (ADC) billing when the deployment configures it; the env # contract has ONE owner (harness.vertex_from_env), so every claude # role on every CLI flips together and the API key stays the fallback - vertex=vertex_from_env(), + vertex=None if profile else vertex_from_env(), + endpoint=profile, ) @@ -187,6 +199,24 @@ class RoleResult: error: str = "" +def redact_role_result(result: RoleResult, harness: Harness) -> RoleResult: + """Scrub verdict strings structurally, including JSON-escaped credentials.""" + from outerloop.harness import redact + + secrets = (getattr(harness, "api_key", ""),) + + def clean(value: Any) -> Any: + if isinstance(value, str): + return redact(value, secrets) + if isinstance(value, dict): + return {clean(key): clean(item) for key, item in value.items()} + if isinstance(value, list): + return [clean(item) for item in value] + return value + + return replace(result, data=clean(result.data), error=clean(result.error)) + + def run_role( spec: RoleSpec, harness: Harness, @@ -227,4 +257,4 @@ def run_role( return RoleResult(ok=False, session=session, error=f"invalid verdict: {exc}") if data is None: return RoleResult(ok=False, session=session, error="judge produced no verdict") - return RoleResult(ok=True, session=session, data=data) + return redact_role_result(RoleResult(ok=True, session=session, data=data), harness) diff --git a/src/outerloop/runstate.py b/src/outerloop/runstate.py index f98e5890..56481fd5 100644 --- a/src/outerloop/runstate.py +++ b/src/outerloop/runstate.py @@ -133,8 +133,8 @@ class RunRecord: # The author this run was STARTED with ("" backend = legacy/claude). A wake or # follow-up reproduces the run's OWN author from these, not the current fleet # default, so a fleet backend flip never resumes a run on the wrong backend, - # model, or key. backend and model are a PAIR — a claude backend needs a - # claude model and vice versa — so both are persisted together. + # model, or key. Endpoint-backed models retain their [endpoint=profile] selector; + # the harness strips it only when constructing the backend session. author_backend: str = "" author_model: str = "" # The resolved author key FILE PATH (not the key) this run used, so a wake or diff --git a/src/outerloop/tick.py b/src/outerloop/tick.py index 2df5897f..94c7d54c 100644 --- a/src/outerloop/tick.py +++ b/src/outerloop/tick.py @@ -2412,7 +2412,7 @@ def _author_config_error(spec: ServiceSpec) -> str: backend = os.environ.get("OUTERLOOP_AUTHOR_BACKEND") or "claude" try: model = fleet_author_model(backend) - except ClaudeModelUnset as exc: + except (ClaudeModelUnset, ValueError) as exc: return str(exc) return codex_author_config_error(backend, model, spec.image) @@ -2432,6 +2432,7 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: return "" try: from outerloop.attempt import PANEL_KEY_DEFAULT, resolve_author_key_file + from outerloop.endpoints import author_model_setting from outerloop.github import FileTokenProvider from outerloop.panel import resolve_lenses @@ -2439,10 +2440,53 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: lenses = resolve_lenses( spec.panel, os.environ.get("OUTERLOOP_AUTHOR_BACKEND", "").strip() or "claude", - os.environ.get("OUTERLOOP_AUTHOR_MODEL", "").strip(), + author_model_setting( + os.environ.get("OUTERLOOP_AUTHOR_BACKEND") or "claude", + os.environ.get("OUTERLOOP_AUTHOR_MODEL", "").strip(), + ), ) except ValueError as exc: return str(exc) + from outerloop.attempt import fleet_author_model + from outerloop.endpoints import resolve_endpoint + + author_backend = os.environ.get("OUTERLOOP_AUTHOR_BACKEND") or "claude" + _, author_endpoint = resolve_endpoint( + author_model_setting(author_backend, os.environ.get("OUTERLOOP_AUTHOR_MODEL", "")), + author_backend, + ) + traditional = [] + for kind, backend, model in lenses: + _, profile = resolve_endpoint(model, backend) + if profile is None: + traditional.append((kind, backend, model)) + continue + if not spec.image or not Path(spec.image).is_file(): + return f"a {backend} endpoint panel lens requires a real container image" + author_backend = os.environ.get("OUTERLOOP_AUTHOR_BACKEND") or "claude" + from outerloop.endpoints import model_key, validate_judge_key_file + + validate_judge_key_file( + profile, + author_endpoint.key_file + if author_endpoint + else resolve_author_key_file(author_backend), + spec.panel_key_file or PANEL_KEY_DEFAULT, + ) + author_key = model_key( + resolve_author_key_file(author_backend), + author_backend, + fleet_author_model(author_backend), + ) + if profile.key() == author_key: + return "a panel judge key is the author key (role separation)" + if backend == "hermes": + from outerloop.hermes_install import hermes_ready + + repo = os.environ.get("REVIEW_HERMES_REPO", "") + if not repo or not hermes_ready(Path(repo).expanduser()): + return "hermes panel needs REVIEW_HERMES_REPO with pinned source and runtime" + lenses = tuple(traditional) # non-claude (shelled) lenses: mirror the climb's rules exactly, per # backend — image required, the judge's OWN key (set + absolute + # neither the author's nor the claude panel key + readable), and for @@ -2483,7 +2527,9 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: "key file (an anthropic key must never reach another " "provider's login)" ) - FileTokenProvider(key_path).token() + key = FileTokenProvider(key_path).token() + if author_endpoint and key == author_endpoint.key(): + return "a panel judge key is the author key (role separation)" if lens_backend == "hermes": repo = os.environ.get("REVIEW_HERMES_REPO", "").strip() from outerloop.hermes_install import hermes_ready @@ -2536,7 +2582,9 @@ def _panel_preflight_error(spec: ServiceSpec) -> str: # time, so the preflight and the climb agree. from outerloop.role_runner import role_key - role_key(path) + key = role_key(path) + if author_endpoint and key == author_endpoint.key(): + return "a panel judge key is the author key (role separation)" return "" except Exception as exc: # never raises: an unexpected failure (partial deploy, ELOOP, unset diff --git a/tests/fixtures/README.md b/tests/fixtures/README.md new file mode 100644 index 00000000..c6986fb5 --- /dev/null +++ b/tests/fixtures/README.md @@ -0,0 +1,18 @@ +# Endpoint compatibility fixtures + +- `author_route_legacy.json`: produced by `runstate.RunRecord` and + `runstate.save_record` from kernel commit `5563c46` (the parent tree before the + endpoint change), with synthetic `owner/repo` and key-file coordinates. +- `hermes_sample_20260924.json`: produced from synthetic user, assistant/tool-call, + tool-result, and final-assistant messages by + `agent.agent_runtime_helpers.convert_to_trajectory_format` and + `run_agent._save_sample_trajectory` at Hermes + `f97608f178d1ffeca59860195ab7da295f7c8e5f`. Timestamp normalized; no model calls. + The pure functions were extracted with Python AST to avoid starting Hermes or + importing its optional runtime integrations. + +The other fixtures carry their originating kernel commit in their filenames. + +- `author_route_missing_model.json` and `author_route_missing_route.json`: synthetic + derivatives of `author_route_legacy.json` with model or all route fields omitted, + exercising pre-field writers through load/save and the new kernel wake entry point. diff --git a/tests/fixtures/author_route_legacy.json b/tests/fixtures/author_route_legacy.json new file mode 100644 index 00000000..86c298d8 --- /dev/null +++ b/tests/fixtures/author_route_legacy.json @@ -0,0 +1,31 @@ +{ + "agent_id": "agent-01", + "author_backend": "codex", + "author_key_file": "/keys/author", + "author_model": "gpt-5.6-terra", + "auto_bless_base": "", + "auto_bless_reason": "", + "auto_bless_reason_kind": "", + "auto_blessed_head": "", + "auto_publish_head": "", + "benchmark": "", + "created": 1.0, + "deadline": 0.0, + "ending": "", + "ending_note": "", + "experiment_job_id": "", + "inbox_seq": 0, + "issue_number": 0, + "pr_url": "", + "resume_session_id": "", + "run_id": "legacy-author", + "run_job_id": "", + "stage": {}, + "state": "parked", + "target": "owner/repo", + "task_title": "Check the model route", + "terminal_seen": 0.0, + "updated": 1.0, + "wake_attempts": 0, + "workspace_shed": 0.0 +} \ No newline at end of file diff --git a/tests/fixtures/author_route_missing_model.json b/tests/fixtures/author_route_missing_model.json new file mode 100644 index 00000000..a8e09bff --- /dev/null +++ b/tests/fixtures/author_route_missing_model.json @@ -0,0 +1,30 @@ +{ + "agent_id": "agent-01", + "author_backend": "codex", + "author_key_file": "/keys/author", + "auto_bless_base": "", + "auto_bless_reason": "", + "auto_bless_reason_kind": "", + "auto_blessed_head": "", + "auto_publish_head": "", + "benchmark": "", + "created": 1.0, + "deadline": 0.0, + "ending": "", + "ending_note": "", + "experiment_job_id": "", + "inbox_seq": 0, + "issue_number": 0, + "pr_url": "", + "resume_session_id": "", + "run_id": "legacy-author", + "run_job_id": "", + "stage": {}, + "state": "parked", + "target": "owner/repo", + "task_title": "Check the model route", + "terminal_seen": 0.0, + "updated": 1.0, + "wake_attempts": 0, + "workspace_shed": 0.0 +} diff --git a/tests/fixtures/author_route_missing_route.json b/tests/fixtures/author_route_missing_route.json new file mode 100644 index 00000000..2749394e --- /dev/null +++ b/tests/fixtures/author_route_missing_route.json @@ -0,0 +1,28 @@ +{ + "agent_id": "agent-01", + "auto_bless_base": "", + "auto_bless_reason": "", + "auto_bless_reason_kind": "", + "auto_blessed_head": "", + "auto_publish_head": "", + "benchmark": "", + "created": 1.0, + "deadline": 0.0, + "ending": "", + "ending_note": "", + "experiment_job_id": "", + "inbox_seq": 0, + "issue_number": 0, + "pr_url": "", + "resume_session_id": "", + "run_id": "legacy-author", + "run_job_id": "", + "stage": {}, + "state": "parked", + "target": "owner/repo", + "task_title": "Check the model route", + "terminal_seen": 0.0, + "updated": 1.0, + "wake_attempts": 0, + "workspace_shed": 0.0 +} diff --git a/tests/fixtures/hermes_sample_20260924.json b/tests/fixtures/hermes_sample_20260924.json new file mode 100644 index 00000000..2e96b47e --- /dev/null +++ b/tests/fixtures/hermes_sample_20260924.json @@ -0,0 +1,28 @@ +{ + "conversations": [ + { + "from": "system", + "value": "You are a function calling AI model. You are provided with function signatures within XML tags. You may call one or more functions to assist with the user query. If available tools are not relevant in assisting with user query, just respond in natural conversational language. Don't make assumptions about what values to plug into functions. After calling & executing the functions, you will be provided with function results within XML tags. Here are the available tools:\n\n[]\n\nFor each function call return a JSON object, with the following pydantic model json schema for each:\n{'title': 'FunctionCall', 'type': 'object', 'properties': {'name': {'title': 'Name', 'type': 'string'}, 'arguments': {'title': 'Arguments', 'type': 'object'}}, 'required': ['name', 'arguments']}\nEach function call should be enclosed within XML tags.\nExample:\n\n{'name': ,'arguments': }\n" + }, + { + "from": "human", + "value": "Inspect the workspace." + }, + { + "from": "gpt", + "value": "\nRead the file first.\n\n\n{\"name\": \"read_file\", \"arguments\": {\"path\": \"README.md\"}}\n" + }, + { + "from": "tool", + "value": "\n{\"tool_call_id\": \"call_1\", \"name\": \"read_file\", \"content\": \"Example project\"}\n" + }, + { + "from": "gpt", + "value": "\n\nChecked the workspace." + } + ], + "timestamp": "2026-09-24T00:00:00", + "model": "open-model", + "completed": true, + "query": "Inspect the workspace." +} diff --git a/tests/test_attempt.py b/tests/test_attempt.py index 39909b2a..157e59d8 100644 --- a/tests/test_attempt.py +++ b/tests/test_attempt.py @@ -988,7 +988,7 @@ def test_codex_author_config_error() -> None: assert "codex/openai model" in codex_author_config_error("codex", "claude-opus-5", "img.sif") assert "codex/openai model" in codex_author_config_error("codex", "", "img.sif") # an unknown backend (typo'd env default) is rejected, not silently accepted - assert "unknown author backend" in codex_author_config_error("hermes", "m", "img.sif") + assert "unknown author backend" in codex_author_config_error("typo", "m", "img.sif") def test_resolve_author_key_file(monkeypatch, tmp_path) -> None: diff --git a/tests/test_default_claude_model.py b/tests/test_default_claude_model.py index e1353356..ebad70b2 100644 --- a/tests/test_default_claude_model.py +++ b/tests/test_default_claude_model.py @@ -393,6 +393,7 @@ def capture_author(backend, model, image): monkeypatch.setattr(attempt, "codex_author_config_error", capture_author) monkeypatch.setattr(attempt, "role_key", lambda *a: "panel-key") + monkeypatch.setattr("outerloop.role_runner.role_key", lambda *a: "panel-key") real_panel = attempt._panel_lenses_from_args def capture_panel(args, **kwargs): diff --git a/tests/test_endpoints.py b/tests/test_endpoints.py new file mode 100644 index 00000000..a0624fed --- /dev/null +++ b/tests/test_endpoints.py @@ -0,0 +1,530 @@ +"""Endpoint wiring is checked at the process boundary without model calls.""" + +from __future__ import annotations + +import json +import os +import subprocess +import tomllib +from pathlib import Path +from types import SimpleNamespace +from typing import Any + +import pytest +import yaml + +from outerloop import harness as harness_mod +from outerloop.attempt import codex_author_config_error, fleet_author_model, resume_author +from outerloop.cli import TICK_ENV_KEYS, env_file_values, missing_claude_model, missing_panel_model +from outerloop.endpoints import endpoint_profile, model_key, resolve_endpoint, split_endpoint +from outerloop.harness import CodexHarness, _parse_hermes_result +from outerloop.panel import parse_lenses, resolve_lenses +from outerloop.review_agent_cli import resolve_reviewer_harness +from outerloop.role_runner import build_harness +from outerloop.roles import author_spec, reviewer_spec + + +@pytest.fixture +def profile(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> dict[str, str]: + key = tmp_path / "endpoint-key" + key.write_text("endpoint-secret") + key.chmod(0o600) + env = { + "OUTERLOOP_ENDPOINT_LOCAL_URL": "https://llm.example.internal/v1", + "OUTERLOOP_ENDPOINT_LOCAL_KEY_FILE": str(key), + "OUTERLOOP_ENDPOINT_LOCAL_MODEL": "open-model", + "OUTERLOOP_ENDPOINT_LOCAL_API": "anthropic,responses,chat", + } + for name, value in env.items(): + monkeypatch.setenv(name, value) + return env + + +@pytest.mark.parametrize("backend", ["claude", "codex", "hermes"]) +def test_profile_for_every_backend(profile: dict[str, str], backend: str) -> None: + served, endpoint = resolve_endpoint("open-model[endpoint=LOCAL]", backend) + assert served == "open-model" and endpoint is not None + assert endpoint.name == "local" + assert endpoint.key() == "endpoint-secret" + assert resolve_endpoint("[endpoint=local]", backend)[0] == "open-model" + assert "endpoint-secret" not in repr(endpoint) + with pytest.raises(ValueError, match="does not match"): + resolve_endpoint("wrong-model[endpoint=local]", backend) + + +@pytest.mark.parametrize( + ("suffix", "value", "error"), + [ + ("URL", "", "missing"), + ("MODEL", "", "missing"), + ("KEY_FILE", "/does/not/exist", "credential file"), + ("KEY_FILE", "relative-key", "absolute"), + ("URL", "https://user:password@llm.example.internal/v1", "without credentials"), + ("URL", "https://llm.example.internal/v1?key=secret", "without credentials"), + ("URL", "file:///tmp/server", "HTTP"), + ], +) +def test_invalid_profile(profile: dict[str, str], suffix: str, value: str, error: str) -> None: + profile[f"OUTERLOOP_ENDPOINT_LOCAL_{suffix}"] = value + with pytest.raises(ValueError, match=error): + endpoint_profile("local", "hermes", environ=profile) + + +def test_bad_names_keys_and_backend(profile: dict[str, str]) -> None: + with pytest.raises(ValueError, match="unknown endpoint"): + endpoint_profile("absent", "claude", environ=profile) + with pytest.raises(ValueError, match="unsupported backend"): + endpoint_profile("local", "other", environ=profile) + for model in ( + "open-model[endpoint=]", + "open-model[endpoint=bad-name]", + "x[endpoint=x][endpoint=local]", + ): + with pytest.raises(ValueError, match="selector"): + split_endpoint(model) + key = Path(profile["OUTERLOOP_ENDPOINT_LOCAL_KEY_FILE"]) + key.chmod(0o644) + with pytest.raises(ValueError, match="chmod 600"): + endpoint_profile("local", "claude", environ=profile) + key.chmod(0o600) + key.write_text("") + with pytest.raises(ValueError, match="empty"): + endpoint_profile("local", "claude", environ=profile) + + +def test_lens_endpoint_grammar(profile: dict[str, str]) -> None: + spec = "verify:hermes:open-model[endpoint=local],review:codex:[endpoint=local]" + assert parse_lenses(spec)[0] == ("verify", "hermes", "open-model[endpoint=local]") + assert resolve_lenses(spec, "claude", "claude-model")[1] == ( + "review", + "codex", + "open-model[endpoint=local]", + ) + # Judges must select their own endpoint/credential, never inherit the author's. + with pytest.raises(ValueError, match="judge endpoint"): + resolve_lenses("review", "codex", "open-model[endpoint=local]") + with pytest.raises(ValueError, match="unknown endpoint"): + resolve_lenses("verify:hermes:[endpoint=missing]", "codex", "") + + +def test_author_and_start_preflight( + profile: dict[str, str], monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setenv("OUTERLOOP_AUTHOR_ENDPOINT", "local") + monkeypatch.delenv("OUTERLOOP_AUTHOR_MODEL", raising=False) + assert fleet_author_model("claude") == "open-model[endpoint=local]" + assert codex_author_config_error("claude", "open-model[endpoint=local]", "") == "" + assert "requires --image" in codex_author_config_error( + "codex", "open-model[endpoint=local]", "" + ) + assert "unknown endpoint" in codex_author_config_error("codex", "[endpoint=missing]", "image") + env = { + **profile, + "OUTERLOOP_AUTHOR_ENDPOINT": "local", + "OUTERLOOP_PANEL": "verify:claude:[endpoint=local]", + } + assert missing_claude_model(env, {}) == "" + assert missing_panel_model(env, {}) == "" + env["OUTERLOOP_PANEL"] = "verify:hermes:[endpoint=missing]" + assert "unknown endpoint" in missing_panel_model(env, {}) + + +@pytest.mark.parametrize("backend", ["claude", "codex", "hermes"]) +@pytest.mark.parametrize("contained", [False, True]) +@pytest.mark.parametrize("failed", [False, True]) +def test_exact_client_configuration_at_process_boundary( + profile: dict[str, str], + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + backend: str, + contained: bool, + failed: bool, +) -> None: + seen: dict[str, Any] = {} + workspace = tmp_path / "workspace" + workspace.mkdir() + monkeypatch.setattr(harness_mod, "hermes_ready", lambda _: True) + monkeypatch.setenv("ANTHROPIC_API_KEY", "ambient-secret") + monkeypatch.setenv("OUTERLOOP_VERTEX_PROJECT", "ambient-project") + monkeypatch.setenv("APPTAINERENV_ANTHROPIC_API_KEY", "ambient-secret") + + class Process: + returncode = 1 if failed else 0 + + def __init__(self, command: list[str], **kwargs: Any): + seen.update(argv=command, env=kwargs["env"]) + home = Path(kwargs["env"]["HOME"]) + for name in (".codex/config.toml", ".hermes/config.yaml"): + path = home / name + if path.is_file(): + seen["config"] = path.read_text() + (home / "sample_test.json").write_text( + json.dumps({"conversations": [{"from": "gpt", "value": "done"}]}) + ) + + def communicate(self, **kwargs: Any) -> tuple[str, str]: + if failed: + return "", "HTTP 503 endpoint unavailable" + return '{"type":"result","subtype":"success","result":"done"}', "" + + monkeypatch.setattr(harness_mod.subprocess, "Popen", Process) + monkeypatch.setattr( + CodexHarness, "_login", lambda *_: pytest.fail("custom provider must not log in to OpenAI") + ) + harness = build_harness( + "ignored-key", + author_spec(), + backend=backend, + endpoint="local", + binary="/opt/agent-cli", + hermes_repo=tmp_path / "hermes", + container_image="/opt/image.sif" if contained else "", + ) + result = harness.run("brief", workspace) + if failed: + assert result.is_error + assert "endpoint-secret" not in repr(seen["argv"]) + assert "ambient-secret" not in repr(seen) + assert "ignored-key" not in repr(seen) + env = seen["env"] + key_env = "ANTHROPIC_AUTH_TOKEN" if backend == "claude" else "OUTERLOOP_SESSION_KEY" + assert env[key_env] == "endpoint-secret" + assert ( + not { + "OPENAI_API_KEY", + "OPENAI_BASE_URL", + "OPENROUTER_API_KEY", + "ANTHROPIC_API_KEY", + "GOOGLE_APPLICATION_CREDENTIALS", + } + & env.keys() + ) + if contained: + assert env[f"APPTAINERENV_{key_env}"] == "endpoint-secret" + if backend == "claude": + # Claude Code appends /v1/messages itself + assert env["ANTHROPIC_BASE_URL"] == "https://llm.example.internal" + assert "ANTHROPIC_API_KEY" not in env + assert "GOOGLE_APPLICATION_CREDENTIALS" not in env + assert env["CLAUDE_CODE_USE_VERTEX"] == "0" + assert env["CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC"] == "1" + for size in ("OPUS", "SONNET", "HAIKU"): + assert env[f"ANTHROPIC_DEFAULT_{size}_MODEL"] == "open-model" + assert env["ANTHROPIC_SMALL_FAST_MODEL"] == "open-model" + elif backend == "codex": + config = tomllib.loads(seen["config"]) + provider = config["model_providers"][config["model_provider"]] + assert provider["base_url"] == profile["OUTERLOOP_ENDPOINT_LOCAL_URL"] + assert provider["env_key"] == key_env + assert provider["wire_api"] == "responses" + assert provider["requires_openai_auth"] is False + else: + config = yaml.safe_load(seen["config"]) + assert config["model"]["reasoning_echo"] is True + provider = config["custom_providers"][0] + assert provider["name"] == config["model"]["provider"] + assert provider["api_mode"] == "chat_completions" + assert len(config["custom_providers"]) == 1 + assert provider["key_env"] == key_env + assert provider["base_url"] == profile["OUTERLOOP_ENDPOINT_LOCAL_URL"] + assert not any(arg.startswith("--base_url") for arg in seen["argv"]) + assert "--enabled_toolsets=file,terminal,web,search" in seen["argv"] + assert env["TERMINAL_CWD"] == str(workspace.resolve()) + assert "endpoint-secret" not in seen.get("config", "") + + +@pytest.mark.parametrize("backend", ["claude", "codex", "hermes"]) +def test_reviewer_endpoint( + profile: dict[str, str], monkeypatch: pytest.MonkeyPatch, backend: str +) -> None: + monkeypatch.setenv("REVIEW_BACKEND", backend) + monkeypatch.setenv("REVIEW_ENDPOINT", "local") + monkeypatch.setenv("REVIEW_HERMES_REPO", "/opt/hermes") + monkeypatch.delenv("REVIEW_MODEL", raising=False) + harness, error, label = resolve_reviewer_harness(reviewer_spec()) + assert not error and label == backend + assert getattr(harness, "model", "") == "open-model" + monkeypatch.setenv("REVIEW_ENDPOINT", "missing") + assert "unknown endpoint" in resolve_reviewer_harness(reviewer_spec())[1] + + +def test_dynamic_deploy_allowlists(profile: dict[str, str], tmp_path: Path) -> None: + path = tmp_path / ".env" + settings = {**profile, "REVIEW_ENDPOINT": "local", "OUTERLOOP_AUTHOR_ENDPOINT": "local"} + path.write_text( + "\n".join(f"{k}={v}" for k, v in settings.items()) + + "\nOUTERLOOP_ENDPOINT_LOCAL_KEY=must-not-forward\n" + ) + assert env_file_values(path, TICK_ENV_KEYS) == settings + # Run the actual deployment config block without its unrelated update/install actions. + script = Path("scripts/tick_deploy.sh").read_text() + helpers = script[script.index("env_line() {") : script.index("# --- 2. deploy")] + start = script.index('if [ -n "$ENV_TRUSTED" ]; then', script.index("# --- config knobs")) + block = script[start : script.index("# Host-side caches", start)] + result = subprocess.run( + ["bash", "-c", helpers + block + "\n/usr/bin/env"], + env={"PATH": os.environ["PATH"], "ENV_TRUSTED": "1", "ENV_FILE": str(path)}, + capture_output=True, + text=True, + check=True, + ) + forwarded = dict(line.split("=", 1) for line in result.stdout.splitlines() if "=" in line) + for key, value in settings.items(): + assert forwarded[key] == value + assert "OUTERLOOP_ENDPOINT_LOCAL_KEY" not in forwarded + + +def test_saved_author_route_survives_fleet_change( + profile: dict[str, str], monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setenv("OUTERLOOP_AUTHOR_ENDPOINT", "new_fleet") + record = SimpleNamespace( + author_backend="codex", + author_model="open-model[endpoint=local]", + author_key_file="/old/key", + ) + for _ in range(3): # first read, repeat, and retry after an interrupted session + backend, model, key_file = resume_author(record, "different-model", "claude") + assert model == "open-model[endpoint=local]" and backend == "codex" + assert model_key(key_file, backend, model) == "endpoint-secret" + legacy = json.loads(Path("tests/fixtures/author_route_legacy.json").read_text()) + for _ in range(3): + assert resume_author(SimpleNamespace(**legacy), "different-model", "claude") == ( + "codex", + "gpt-5.6-terra", + "/keys/author", + ) + + +def test_pinned_hermes_sample() -> None: + sample = json.loads(Path("tests/fixtures/hermes_sample_20260924.json").read_text()) + result = _parse_hermes_result("noisy stdout", sample, 0) + assert not result.is_error + assert result.num_turns == 2 + assert result.final_text.endswith("Checked the workspace.") + + +def test_record_compatibility_roundtrip(tmp_path, profile): + from dataclasses import replace + + from outerloop.runstate import RECORD_NAME, load_record, run_dir, save_record + + fixture = Path("tests/fixtures/author_route_legacy.json").read_text() + directory = run_dir(tmp_path, "legacy-author") + directory.mkdir(parents=True) + (directory / RECORD_NAME).write_text(fixture) + legacy = load_record(tmp_path, "legacy-author") + assert resume_author(legacy, "open-model[endpoint=local]", "hermes")[:2] == ( + "codex", + "gpt-5.6-terra", + ) + assert load_record(tmp_path, "legacy-author") == legacy + save_record(tmp_path, legacy, now=legacy.updated) + assert load_record(tmp_path, "legacy-author") == legacy + # An interrupted writer's temporary file does not replace the committed record. + (directory / (RECORD_NAME + ".interrupted")).write_text('{"author_model":') + assert load_record(tmp_path, "legacy-author") == legacy + endpoint = replace(legacy, author_model="open-model[endpoint=local]") + save_record(tmp_path, endpoint, now=legacy.updated) + assert load_record(tmp_path, "legacy-author").author_model == "open-model[endpoint=local]" + assert ( + model_key(endpoint.author_key_file, endpoint.author_backend, endpoint.author_model) + == "endpoint-secret" + ) + + +def test_tick_endpoint_preflight(profile, tmp_path, monkeypatch): + from outerloop.tick import ServiceSpec, _author_config_error, _panel_preflight_error + + image = tmp_path / "image.sif" + image.touch() + author_key = tmp_path / "author-key" + author_key.write_text("separate-author-secret") + author_key.chmod(0o600) + monkeypatch.setenv("OUTERLOOP_AUTHOR_BACKEND", "codex") + monkeypatch.setenv("OUTERLOOP_AUTHOR_MODEL", "gpt-5.6-terra") + monkeypatch.setenv("OUTERLOOP_CODEX_KEY_FILE", str(author_key)) + spec = ServiceSpec( + account="", + partition="", + run_root=tmp_path, + home=tmp_path, + panel="verify:codex:[endpoint=local]", + image=str(image), + panel_key_file="", + ) + assert _panel_preflight_error(spec) == "" + author_key.write_text("endpoint-secret") + assert "role separation" in _panel_preflight_error(spec) + monkeypatch.setenv("OUTERLOOP_AUTHOR_ENDPOINT", "missing") + assert "unknown endpoint" in _author_config_error(spec) + monkeypatch.setenv("OUTERLOOP_AUTHOR_ENDPOINT", "local") + monkeypatch.setenv("OUTERLOOP_AUTHOR_MODEL", "") + monkeypatch.setenv("OUTERLOOP_AUTHOR_BACKEND", "hermes") + monkeypatch.setenv("REVIEW_HERMES_REPO", "/opt/hermes-agent") + monkeypatch.setattr("outerloop.hermes_install.hermes_ready", lambda _: True) + assert _author_config_error(spec) == "" + + +def test_panel_endpoint_builds_on_own_key(profile, tmp_path): + from outerloop.attempt import _panel_lenses_from_args + + key = tmp_path / "author-key" + key.write_text("own-author-secret") + key.chmod(0o600) + args = SimpleNamespace( + key_file=str(key), + panel="verify:codex:[endpoint=local],review:claude:open-model[endpoint=local]", + author_backend="codex", + model="gpt-5.6-terra", + panel_key_file="/missing/native-key", + image="/opt/image.sif", + claude_bin="/opt/claude", + codex_bin="/opt/codex", + ) + lenses, secrets = _panel_lenses_from_args(args) + assert len(lenses) == 2 and secrets == ("endpoint-secret",) + assert all(getattr(lens.harness, "model", "") == "open-model" for lens in lenses) + args.model = "open-model[endpoint=local]" + with pytest.raises(ValueError, match="role separation"): + _panel_lenses_from_args(args) + + +@pytest.mark.parametrize("model", ["claude-sonnet-4@20250514", "org/model:free", "x@y"]) +def test_native_model_ids_are_unchanged(model): + assert split_endpoint(model) == (model, "") + assert parse_lenses(f"review:claude:{model}")[0][2] == model + + +@pytest.mark.parametrize( + "backend,api", [("claude", "anthropic"), ("codex", "responses"), ("hermes", "chat")] +) +def test_api_compatibility(profile, backend, api): + profile["OUTERLOOP_ENDPOINT_LOCAL_API"] = api + assert endpoint_profile("local", backend, environ=profile).apis == (api,) + profile["OUTERLOOP_ENDPOINT_LOCAL_API"] = "chat" if api != "chat" else "responses" + with pytest.raises(ValueError, match="requires API"): + endpoint_profile("local", backend, environ=profile) + del profile["OUTERLOOP_ENDPOINT_LOCAL_API"] + with pytest.raises(ValueError, match=r"missing.*API"): + endpoint_profile("local", backend, environ=profile) + + +@pytest.mark.parametrize("same_as", ["author", "panel"]) +@pytest.mark.parametrize("alias", ["direct", "symlink", "hardlink"]) +def test_endpoint_key_file_separation(profile, tmp_path, monkeypatch, same_as, alias): + from outerloop.attempt import _panel_lenses_from_args + from outerloop.tick import ServiceSpec, _panel_preflight_error + + endpoint_key = Path(profile["OUTERLOOP_ENDPOINT_LOCAL_KEY_FILE"]) + shared = endpoint_key + if alias != "direct": + shared = tmp_path / "alias" + if alias == "symlink": + shared.symlink_to(endpoint_key) + else: + shared.hardlink_to(endpoint_key) + own = tmp_path / "own" + own.write_text("separate-secret") + own.chmod(0o600) + author = shared if same_as == "author" else own + panel = shared if same_as == "panel" else own + image = tmp_path / "image.sif" + image.touch() + monkeypatch.setenv("OUTERLOOP_AUTHOR_BACKEND", "codex") + monkeypatch.setenv("OUTERLOOP_AUTHOR_MODEL", "gpt-native") + monkeypatch.setenv("OUTERLOOP_CODEX_KEY_FILE", str(author)) + args = SimpleNamespace( + panel="review:codex:[endpoint=local]", + author_backend="codex", + model="gpt-native", + key_file=str(author), + panel_key_file=str(panel), + image=str(image), + ) + with pytest.raises( + ValueError, match=f"{same_as if same_as == 'author' else 'claude panel'} key file" + ): + _panel_lenses_from_args(args) + spec = ServiceSpec( + account="", + partition="", + run_root=tmp_path, + home=tmp_path, + panel=args.panel, + image=str(image), + panel_key_file=str(panel), + ) + assert "key file (role separation)" in _panel_preflight_error(spec) + + +@pytest.mark.parametrize( + "fixture", ["author_route_missing_model.json", "author_route_missing_route.json"] +) +def test_missing_route_never_inherits_endpoint(fixture, tmp_path, monkeypatch): + from outerloop.runstate import RECORD_NAME, load_record, run_dir, save_record + + monkeypatch.setenv("OUTERLOOP_CLAUDE_MODEL", "claude-native") + directory = run_dir(tmp_path, "legacy-author") + directory.mkdir(parents=True) + (directory / RECORD_NAME).write_text(Path("tests/fixtures", fixture).read_text()) + for _ in range(2): + record = load_record(tmp_path, "legacy-author") + backend, model, _ = resume_author(record, "open-model[endpoint=missing]", "claude") + assert not split_endpoint(model)[1] + assert model == ("claude-native" if backend == "claude" else "") + save_record(tmp_path, record, now=record.updated) + + +def test_new_kernel_wakes_missing_route_record(tmp_path, monkeypatch): + from outerloop import attempt + from outerloop.runstate import RECORD_NAME, run_dir + + directory = run_dir(tmp_path, "legacy-author") + directory.mkdir(parents=True) + data = json.loads(Path("tests/fixtures/author_route_missing_route.json").read_text()) + data["stage"] = {"phase": "author-sleep"} + (directory / RECORD_NAME).write_text(json.dumps(data)) + monkeypatch.setenv("OUTERLOOP_AUTHOR_ENDPOINT", "absent_fleet_profile") + monkeypatch.setenv("OUTERLOOP_AUTHOR_BACKEND", "claude") + monkeypatch.delenv("OUTERLOOP_AUTHOR_MODEL", raising=False) + monkeypatch.setenv("OUTERLOOP_CLAUDE_MODEL", "claude-native") + image = tmp_path / "image.sif" + image.touch() + monkeypatch.setattr( + "sys.argv", + [ + "climb", + "--resume", + "legacy-author", + "--run-root", + str(tmp_path), + "--image", + str(image), + "--panel-skip", + "test", + ], + ) + monkeypatch.setattr( + attempt, "resolve_bot_auth", lambda *args: SimpleNamespace(token=lambda: "bot-secret") + ) + monkeypatch.setattr(attempt, "_lease_held_by_another_job", lambda *args: "") + monkeypatch.setattr("outerloop.tick.dispatch_wake_armed", lambda *args: True) + monkeypatch.setattr(attempt, "model_key", lambda *args: "native-key") + seen: dict[str, Any] = {} + monkeypatch.setattr( + attempt, "build_harness", lambda *args, **kwargs: seen.update(kwargs) or object() + ) + + class Resumed(Exception): + pass + + def resume(*args, **kwargs): + assert kwargs["harness"] is not None + raise Resumed + + monkeypatch.setattr(attempt, "resume_run", resume) + monkeypatch.setattr(attempt, "_release_own_lease", lambda *args: None) + with pytest.raises(Resumed): + attempt.main() + assert seen["backend"] == "claude" and seen["model"] == "claude-native" diff --git a/tests/test_hermes_harness.py b/tests/test_hermes_harness.py index 75cdbc8e..609314c5 100644 --- a/tests/test_hermes_harness.py +++ b/tests/test_hermes_harness.py @@ -1,7 +1,7 @@ """HermesHarness command construction and output parsing. Hermes is not run here; these pin the argv shape (verified against -hermes-agent v0.20.1 source) and the defensive trajectory parsing.""" +hermes-agent v2026.9.24 source) and the defensive trajectory parsing.""" from __future__ import annotations @@ -75,9 +75,9 @@ def test_command_shape_and_toolsets() -> None: ] assert "uv" not in cmd assert "--save_sample" in cmd - # embedded quotes so fire literal-evals a STRING, not a tuple - assert '--enabled_toolsets="file"' in cmd - assert '--disabled_toolsets="terminal,web"' in cmd + # argparse receives comma-separated strings without embedded quotes + assert "--enabled_toolsets=file" in cmd + assert "--disabled_toolsets=terminal,web" in cmd assert "--max_turns=40" in cmd assert not any("--base_url" in part for part in cmd) # empty -> hermes default @@ -298,8 +298,8 @@ def test_parse_sharegpt_trajectory() -> None: def test_parse_conversations_wrapper() -> None: # run_agent.py --save_sample wraps the turns under "conversations" - # (run_agent.py:8404, v0.20.1). Missing this key was read as zero turns and - # dropped a real verdict as a bogus error — the whole "produced no verdict" bug. + # (run_agent.py::_save_sample_trajectory, v2026.9.24). A missing wrapper + # used to discard a real verdict as a zero-turn error. sample = { "conversations": [ {"from": "human", "value": "the brief"}, diff --git a/tests/test_install_harness.py b/tests/test_install_harness.py index e34d5fe7..3a977546 100644 --- a/tests/test_install_harness.py +++ b/tests/test_install_harness.py @@ -379,7 +379,7 @@ def test_hermes_clone_retry_and_pin_verification(tmp_path, moved_tag): exit 1 fi [ ! -e "$REPO/partial" ] - expected="clone --depth 1 --branch v2026.8.13" + expected="clone --depth 1 --branch v2026.9.24" [ "$*" = "$expected https://github.com/NousResearch/hermes-agent $REPO" ] mkdir -p "$REPO/.git" elif [ "$3" = rev-parse ]; then diff --git a/tests/test_local_compute.py b/tests/test_local_compute.py index a5714a33..6548d631 100644 --- a/tests/test_local_compute.py +++ b/tests/test_local_compute.py @@ -1061,3 +1061,26 @@ def test_unannounced_reservation_keeps_live_submitter_or_grace_period( ) is live ) + + +def test_endpoint_settings_cross_local_job_boundary(tmp_path, monkeypatch): + import json + + out = tmp_path / "env.json" + settings = { + "OUTERLOOP_ENDPOINT_LOCAL_URL": "https://llm.example.internal/v1", + "OUTERLOOP_ENDPOINT_LOCAL_KEY_FILE": "/keys/local", + "OUTERLOOP_ENDPOINT_LOCAL_MODEL": "open-model", + "OUTERLOOP_AUTHOR_ENDPOINT": "local", + "REVIEW_ENDPOINT": "judge", + "REVIEW_MODEL": "open-model", + "REVIEW_BACKEND": "hermes", + } + for name, value in settings.items(): + monkeypatch.setenv(name, value) + monkeypatch.setenv("OUTERLOOP_ENDPOINT_LOCAL_KEY", "do-not-forward") + monkeypatch.setenv("APPTAINERENV_OUTERLOOP_SESSION_KEY", "do-not-forward") + LocalCompute().submit(_spec(command=f"/usr/bin/env > {out}")) + env = dict(line.split("=", 1) for line in out.read_text().splitlines() if "=" in line) + assert all(env.get(k) == v for k, v in settings.items()) + assert "do-not-forward" not in json.dumps(env) diff --git a/tests/test_review_agent.py b/tests/test_review_agent.py index 5c761843..8fe1f7ed 100644 --- a/tests/test_review_agent.py +++ b/tests/test_review_agent.py @@ -11,6 +11,8 @@ from pathlib import Path from typing import Any +import pytest + from outerloop.harness import SessionResult from outerloop.review import build_agent_brief from outerloop.review_agent import run_agent_review @@ -45,6 +47,8 @@ class _Harness: + api_key: str = "" + def __init__(self, final_text: str, *, is_error: bool = False, detail: str = "") -> None: self._text, self._err, self._detail = final_text, is_error, detail self.briefs: list[str] = [] @@ -496,3 +500,68 @@ def test_skip_stub_names_its_opinion(tmp_path: Path) -> None: ) (comment,) = client.comments assert "advisory review (second opinion — terra)" in comment + + +def test_sanitize_hermes_instruction_aliases(tmp_path): + from outerloop.review_agent import sanitize_checkout + + names = ( + "AGENTS.override.md", + "agents.md", + "claude.md", + ".hermes.md", + "HERMES.md", + ".cursorrules", + ) + for name in names: + (tmp_path / name).write_text("untrusted instructions") + (tmp_path / ".cursor" / "rules").mkdir(parents=True) + (tmp_path / ".cursor" / "rules" / "rule.mdc").write_text("untrusted instructions") + assert sanitize_checkout(tmp_path) == (len(names) + 1, 0) + assert sanitize_checkout(tmp_path) == (0, 0) + for name in names: + assert not (tmp_path / name).exists() + assert (tmp_path / (name + ".pr-data")).is_file() + + +@pytest.mark.parametrize("backend", ["claude", "codex", "hermes"]) +def test_endpoint_verdict_redacted_before_emit_post_and_summarizer(tmp_path, monkeypatch, backend): + from outerloop.review import build_summarizer_brief + from outerloop.role_runner import build_harness + from outerloop.roles import reviewer_spec + + secret = 'endpoint-"secret\\with-escapes' + data = json.loads(_FINDINGS) + data["notes"] = secret + for field in ("summary", "detail"): + data["findings"][0][field] = secret + for emit in (False, True): + writer = _Harness(json.dumps(data)) + harness = build_harness( + secret, reviewer_spec(), backend=backend, model="test-model", hermes_repo=tmp_path + ) + monkeypatch.setattr( + type(harness), "run", lambda self, *args, writer=writer: writer.run(*args) + ) + client = _Client() + path = tmp_path / "envelope.json" + run_agent_review( + client, # type: ignore[arg-type] + "owner/repo", + 1, + harness, + tmp_path, + bot_login="bot", + emit_path=path if emit else None, + ) + if emit: + envelope = json.loads(path.read_text()) + assert envelope["data"]["notes"] == "[redacted]" + assert envelope["data"]["findings"][0]["summary"] == "[redacted]" + assert envelope["data"]["findings"][0]["detail"] == "[redacted]" + brief = build_summarizer_brief([envelope], syscall_cmd="tool") + assert secret not in brief + assert "[redacted]" in brief + else: + assert secret not in str(client.reviews + client.comments) + assert "[redacted]" in str(client.reviews + client.comments) diff --git a/tests/test_start.py b/tests/test_start.py index a3d76764..e64224c8 100644 --- a/tests/test_start.py +++ b/tests/test_start.py @@ -702,10 +702,9 @@ def test_start_refuses_without_the_claude_model(clean_env, monkeypatch, capsys, assert main(argv) == 0 -def test_hermes_is_a_review_backend_not_an_author(tmp_path): +def test_hermes_author_requires_pinned_runtime(tmp_path): problem = cli.missing_harness_binary({"OUTERLOOP_AUTHOR_BACKEND": "hermes"}, {}) - assert "unsupported author backend 'hermes'" in problem - assert "scripts/install_hermes.sh" in problem + assert "REVIEW_HERMES_REPO with pinned source and runtime" in problem # ---------------------------------------------------------------- uv From 6d5aa0017bdb8ac6c5c1a01b9bb2f2d70607fda3 Mon Sep 17 00:00:00 2001 From: Mengye Ren Date: Mon, 28 Sep 2026 12:18:19 -0400 Subject: [PATCH 2/2] Endpoints doc: Claude Code in the example; record the live Codex finding and the /v1 handling Co-Authored-By: Claude Opus 5.5 (1M context) --- docs/endpoints.md | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/docs/endpoints.md b/docs/endpoints.md index 225fef05..ebd72a64 100644 --- a/docs/endpoints.md +++ b/docs/endpoints.md @@ -12,11 +12,11 @@ OUTERLOOP_ENDPOINT_AUTHOR_API=chat OUTERLOOP_ENDPOINT_JUDGE_URL=https://llm.example.internal/v1 OUTERLOOP_ENDPOINT_JUDGE_KEY_FILE=/keys/model-judge OUTERLOOP_ENDPOINT_JUDGE_MODEL=open-model -OUTERLOOP_ENDPOINT_JUDGE_API=chat,responses +OUTERLOOP_ENDPOINT_JUDGE_API=chat,anthropic OUTERLOOP_AUTHOR_BACKEND=hermes OUTERLOOP_AUTHOR_ENDPOINT=author -OUTERLOOP_PANEL=verify:hermes:[endpoint=judge],review:codex:open-model[endpoint=judge] +OUTERLOOP_PANEL=verify:hermes:[endpoint=judge],review:claude:open-model[endpoint=judge] REVIEW_HERMES_REPO=/opt/hermes-agent OUTERLOOP_IMAGE=/opt/agent.sif ``` @@ -54,7 +54,8 @@ For example, an Anthropic-compatible client may need `https://llm.example.internal` while OpenAI-compatible clients need `https://llm.example.internal/v1`. -- **Claude Code:** `ANTHROPIC_BASE_URL` and `ANTHROPIC_AUTH_TOKEN`; the profile +- **Claude Code:** `ANTHROPIC_BASE_URL` (the profile URL without a trailing `/v1`, + since Claude Code appends `/v1/messages`) and `ANTHROPIC_AUTH_TOKEN`; the profile model also sets the default Opus, Sonnet, Haiku, and small-fast model variables. Nonessential traffic is disabled. Vertex, Bedrock, and Foundry are disabled; API keys, ADC, and ambient authentication are excluded from the session env. @@ -62,8 +63,10 @@ For example, an Anthropic-compatible client may need `base_url`, `env_key`, `wire_api = "responses"`, and `requires_openai_auth = false`. Custom-provider sessions skip OpenAI login. This follows the [official configuration reference](https://developers.openai.com/codex/config-reference). - Configuration is tested here; live self-hosted Responses interoperability is - not verified by this change. + In a live test against a self-hosted vLLM server, requests reached the local + Responses endpoint and succeeded, but the session filed no verdict: the served + model's tool calls did not come back through that server's Responses API. Use + Claude Code or Hermes for self-hosted endpoints until this is resolved. - **Hermes:** the per-run `.hermes/config.yaml` selects a named `custom_providers` list entry with `base_url`, `key_env`, and `api_mode: chat_completions`. **`reasoning_echo: true` belongs under `model`,