From fd217dd832fa7229db1fba386e1e6ecd50f21ab9 Mon Sep 17 00:00:00 2001 From: Mengye Ren Date: Sat, 26 Sep 2026 15:45:44 -0400 Subject: [PATCH 1/5] Per-target GPU lanes from the deployment's settings OUTERLOOP_GPU_LANES maps a target to a partition, account, GPU type and extra sbatch flags, so one repo's GPU jobs (evals, author launches, sweeps) can go to a dedicated node while others keep the fleet lane. A typed lane requests --gres=gpu::N. Extra flags may not repeat any flag the kernel sets, since sbatch lets the later one win. Co-Authored-By: Claude Opus 5.5 (1M context) --- CHANGELOG.md | 8 ++ docs/install.md | 16 +++ scripts/tick_deploy.sh | 2 +- src/outerloop/attempt.py | 15 ++- src/outerloop/cli.py | 1 + src/outerloop/compute.py | 9 +- src/outerloop/dispatch.py | 4 + src/outerloop/gpu_lanes.py | 88 ++++++++++++++ src/outerloop/measure.py | 35 +++++- src/outerloop/tick.py | 23 +++- tests/test_gpu_lanes.py | 239 +++++++++++++++++++++++++++++++++++++ tests/test_start.py | 2 +- tests/test_tick.py | 4 +- 13 files changed, 433 insertions(+), 13 deletions(-) create mode 100644 src/outerloop/gpu_lanes.py create mode 100644 tests/test_gpu_lanes.py diff --git a/CHANGELOG.md b/CHANGELOG.md index d7d168ce..34bba6c4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,14 @@ Versions follow [SemVer](https://semver.org). ## [Unreleased] +### Added + +- Optional per-target GPU lanes route evals and author launches to deployment-specific partitions, accounts, GPU types, and sbatch flags. + +### Upgrading + +- No action needed; OUTERLOOP_GPU_LANES is optional. + ## [0.2.1] - 2026-09-25 ### Upgrading diff --git a/docs/install.md b/docs/install.md index 0c7729b2..d274ee14 100644 --- a/docs/install.md +++ b/docs/install.md @@ -464,6 +464,22 @@ default, and a lens that names no model runs the author's model when it shares the author's backend, and must name an explicit model on any other backend); the author backend is `OUTERLOOP_AUTHOR_BACKEND`/`OUTERLOOP_AUTHOR_MODEL`. +For a target-specific GPU lane, add a JSON mapping to the deployment's `.env`: + +```bash +OUTERLOOP_GPU_LANES='{"owner/repo":{"partition":"b200","account":"torch_pr_36_mren","gpu_type":"b200","extra":["--comment=preemption=yes"]}}' +``` + +Replace `owner/repo` with the reserved node's target. This Torch lane submits +`--account=torch_pr_36_mren --partition=b200 --gres=gpu:b200:N --comment=preemption=yes` +for that target's GPU evals and author launches (including arrays and re-measures). +Other targets, such as gpt-speedrun, keep the fleet's H200 lane; CPU jobs are +unchanged. Only `partition` is required; an omitted `account` uses the CPU/default +account, and omitted `gpu_type` preserves untyped per-node GPU requests. `extra` +is a list of `--name=value` flags; kernel-owned account, partition, gres, gpus*, +time, mem, and cpus* flags are rejected, as are unknown keys and malformed JSON. +These are cluster settings, not target contract fields. + `OUTERLOOP_CLAUDE_MODEL` names the model for every Claude role (author, panel judges, steward) when no explicit or inherited model covers that role: there is no built-in default, and `start` refuses when a role needs it, naming diff --git a/scripts/tick_deploy.sh b/scripts/tick_deploy.sh index 4f58e4be..75308e52 100755 --- a/scripts/tick_deploy.sh +++ b/scripts/tick_deploy.sh @@ -194,7 +194,7 @@ if [ -n "$ENV_TRUSTED" ]; then OUTERLOOP_VERTEX_ADC OUTERLOOP_VERTEX_SMALL_MODEL \ OUTERLOOP_TARGET \ OUTERLOOP_GITHUB_APP_FILE OUTERLOOP_BOT_LOGIN OUTERLOOP_BOT_ALIASES \ - OUTERLOOP_GPU_PARTITION OUTERLOOP_GPU_ACCOUNT \ + OUTERLOOP_GPU_PARTITION OUTERLOOP_GPU_ACCOUNT OUTERLOOP_GPU_LANES \ OUTERLOOP_QOS OUTERLOOP_APPTAINER_BIN \ OUTERLOOP_IMAGE \ OUTERLOOP_PANEL OUTERLOOP_PANEL_KEY_FILE \ diff --git a/src/outerloop/attempt.py b/src/outerloop/attempt.py index 1115304b..79d81e77 100644 --- a/src/outerloop/attempt.py +++ b/src/outerloop/attempt.py @@ -809,6 +809,8 @@ def _dispatch_settings(args: argparse.Namespace) -> DispatchSettings: gpu_partition=getattr(args, "gpu_partition", ""), gpu_account=getattr(args, "gpu_account", ""), seed_cache=seed_dir(Path(args.run_root), target) if target else None, + target=target, + gpu_lanes=getattr(args, "gpu_lanes", {}), ) @@ -826,6 +828,8 @@ def with_seed(dispatch: DispatchSettings, run_root: Path, target: str) -> Dispat """These settings with the target's seed cache filled in from the record's target when the CLI gave none (wake and follow-up jobs carry the run id, not the target).""" + if isinstance(dispatch, DispatchSettings): + dispatch = dc_replace(dispatch, target=target) # tolerant of any settings object: a backend that knows no seed (or a # test double) is left exactly as it is if not target or getattr(dispatch, "seed_cache", "unknown") is not None: @@ -844,7 +848,8 @@ def _make_launcher( the sealed snapshot (write_eval_job's copy-out handles artifacts), and a partially-submitted batch is reaped rather than orphaned. `gpus` is the benchmark's: an author's experiments run on the same lane as its evals.""" - account, partition = dispatch.placement(gpus) + lane = dispatch.lane(gpus) + account, partition = lane.account, lane.partition def launcher(sha: str, request: SyscallRequest) -> str: from outerloop.dispatch import eval_job_spec, write_eval_job @@ -878,6 +883,8 @@ def launcher(sha: str, request: SyscallRequest) -> str: partition=partition, eval_minutes=launch.minutes, gpus=gpus, + gpu_type=lane.gpu_type, + extra=lane.extra, nice=LAUNCH_NICE, array=array_spec(launch), ) @@ -4874,6 +4881,12 @@ def _run_id(value: str) -> str: "--hypothesis-b64", default="", help="base64 task hypothesis (issue text, fenced)" ) args = parser.parse_args() + from outerloop.gpu_lanes import gpu_lanes_from_env + + try: + args.gpu_lanes = gpu_lanes_from_env() + except ValueError as exc: + parser.error(str(exc)) logging.basicConfig(level=logging.INFO, format="%(asctime)s %(message)s") if not args.model and not args.resume: # resolved after parsing, never at parser build: a deployment without diff --git a/src/outerloop/cli.py b/src/outerloop/cli.py index 40532550..37327593 100644 --- a/src/outerloop/cli.py +++ b/src/outerloop/cli.py @@ -69,6 +69,7 @@ "OUTERLOOP_BOT_ALIASES", "OUTERLOOP_GPU_PARTITION", "OUTERLOOP_GPU_ACCOUNT", + "OUTERLOOP_GPU_LANES", "OUTERLOOP_QOS", "OUTERLOOP_APPTAINER_BIN", "OUTERLOOP_IMAGE", diff --git a/src/outerloop/compute.py b/src/outerloop/compute.py index 101eb63a..6a0b5f64 100644 --- a/src/outerloop/compute.py +++ b/src/outerloop/compute.py @@ -111,6 +111,8 @@ class JobSpec: array: str = "" extra: tuple[str, ...] = () + gpu_type: str = "" + def to_argv(self) -> list[str]: if bool(self.command) == bool(self.script): raise ValueError("exactly one of command/script must be set") @@ -132,7 +134,12 @@ def to_argv(self) -> list[str]: # and Slurm submit plugins commonly classify a job by its # per-node GRES — the per-job form has been rejected on a GPU # partition as "CPU job setup is not valid" - argv.append(f"--gpus-per-node={self.gpus}") + # Typed --gres is also per-node. + argv.append( + f"--gres=gpu:{self.gpu_type}:{self.gpus}" + if self.gpu_type + else f"--gpus-per-node={self.gpus}" + ) if self.qos: argv.append(f"--qos={self.qos}") if self.nice: diff --git a/src/outerloop/dispatch.py b/src/outerloop/dispatch.py index b279cc81..788dab27 100644 --- a/src/outerloop/dispatch.py +++ b/src/outerloop/dispatch.py @@ -532,6 +532,8 @@ def eval_job_spec( gpus: int = 0, nice: int = 0, array: str = "", + gpu_type: str = "", + extra: tuple[str, ...] = (), ) -> JobSpec: """The JobSpec for one dispatched eval: the hint CLAMPED to our ceiling plus setup slack — a contract value above EVAL_JOB_MINUTES_CEILING must @@ -561,6 +563,8 @@ def eval_job_spec( gpus=gpus, nice=nice, array=array, + gpu_type=gpu_type, + extra=extra, ) diff --git a/src/outerloop/gpu_lanes.py b/src/outerloop/gpu_lanes.py new file mode 100644 index 00000000..413c8155 --- /dev/null +++ b/src/outerloop/gpu_lanes.py @@ -0,0 +1,88 @@ +"""Deployment-owned GPU lanes; targets never choose scheduler flags.""" + +from __future__ import annotations + +import json +import os +import re +from dataclasses import dataclass + +KERNEL_FLAGS = frozenset( + { + "account", + "partition", + "gres", + "time", + "qos", + "nice", + "array", + "dependency", + "begin", + "job-name", + "output", + "error", + "wrap", + "parsable", + "chdir", + } +) + + +@dataclass(frozen=True) +class GpuLane: + partition: str + account: str = "" + gpu_type: str = "" + extra: tuple[str, ...] = () + + +def gpu_lanes_from_env() -> dict[str, GpuLane]: + """Read and validate once at each composition root, before submitting work.""" + return parse_gpu_lanes(os.environ.get("OUTERLOOP_GPU_LANES", "{}")) + + +def parse_gpu_lanes(raw: str) -> dict[str, GpuLane]: + """Reject malformed config with one setting-labelled startup error.""" + try: + data = json.loads(raw) + if not isinstance(data, dict): + raise ValueError("expected an object mapping owner/repo to lanes") + lanes = {} + for target, lane in data.items(): + if not re.fullmatch(r"[^/\s]+/[^/\s]+", target): + raise ValueError(f"invalid target {target!r}; expected owner/repo") + if not isinstance(lane, dict) or lane.keys() - { + "partition", + "account", + "gpu_type", + "extra", + }: + raise ValueError( + f"{target}: expected a lane object with only partition/account/gpu_type/extra" + ) + if not isinstance(lane.get("partition"), str) or not lane["partition"].strip(): + raise ValueError(f"{target}: partition must be a nonempty string") + for key in ("account", "gpu_type"): + if key in lane and not isinstance(lane[key], str): + raise ValueError(f"{target}: {key} must be a string") + gpu_type = lane.get("gpu_type", "") + if gpu_type and not re.fullmatch(r"[A-Za-z0-9_.-]+", gpu_type): + raise ValueError(f"{target}: gpu_type must be a single GPU type") + extra = lane.get("extra", []) + if not isinstance(extra, list): + raise ValueError(f"{target}: extra must be a list of strings") + for flag in extra: + if not isinstance(flag, str) or not re.fullmatch( + r"--[a-z][a-z0-9-]*=[^\x00\r\n]*", flag + ): + raise ValueError(f"{target}: extra flags must have the form --name=value") + name = flag[2:].split("=", 1)[0] + # later flags win in sbatch, so an extra must not repeat one the kernel sets + if name in KERNEL_FLAGS or name.startswith(("gpus", "cpus", "mem")): + raise ValueError(f"{target}: kernel-owned extra flag --{name}") + lanes[target] = GpuLane( + lane["partition"], lane.get("account", ""), gpu_type, tuple(extra) + ) + return lanes + except (ValueError, TypeError) as exc: + raise ValueError(f"OUTERLOOP_GPU_LANES: {exc}") from None diff --git a/src/outerloop/measure.py b/src/outerloop/measure.py index 127b7daa..d72dd944 100644 --- a/src/outerloop/measure.py +++ b/src/outerloop/measure.py @@ -21,7 +21,7 @@ import json import logging import time -from dataclasses import dataclass +from dataclasses import dataclass, field from pathlib import Path from typing import Any @@ -31,6 +31,7 @@ read_eval_result, write_eval_job, ) +from outerloop.gpu_lanes import GpuLane from outerloop.orchestrator import EvalError log = logging.getLogger(__name__) @@ -282,6 +283,9 @@ class DispatchedMeasurer: seed_cache: Path | None = None qos: str = "" + gpu_type: str = "" + gpu_extra: tuple[str, ...] = () + def _placement(self, m: Measure) -> tuple[str, str]: if m.gpus <= 0 or not self.compute.has_lanes: return self.account, self.partition @@ -400,6 +404,8 @@ def _dispatch(self, m: Measure) -> str: partition=partition, eval_minutes=self.eval_minutes, gpus=m.gpus, + gpu_type=self.gpu_type if m.gpus and self.compute.has_lanes else "", + extra=self.gpu_extra if m.gpus and self.compute.has_lanes else (), ) job_id = self.compute.submit(spec) (self._ev(m) / "submitted").write_text(job_id) @@ -530,7 +536,23 @@ class DispatchSettings: seed_cache: Path | None = None qos: str = "" + gpu_lanes: dict[str, GpuLane] = field(default_factory=dict) + target: str = "" + + def lane(self, gpus: int) -> GpuLane: + if gpus <= 0 or not self.compute.has_lanes: + return GpuLane(self.partition, self.account) + lane = self.gpu_lanes.get(self.target) + if lane is not None: + return GpuLane(lane.partition, lane.account or self.account, lane.gpu_type, lane.extra) + account, partition = self._fleet_placement(gpus) + return GpuLane(partition, account) + def placement(self, gpus: int) -> tuple[str, str]: + lane = self.lane(gpus) + return lane.account, lane.partition + + def _fleet_placement(self, gpus: int) -> tuple[str, str]: """(account, partition) for a job needing `gpus` GPUs. Raises when a GPU job has no lane — a queue that can never run is worse than a loud refusal. A backend without lanes (local compute) runs every job, @@ -552,6 +574,11 @@ def measurer( out; `eval_minutes` is the benchmark's contract hint (clamped in the job spec). GPUs are per MEASURE (Measure.gpus): the measurer carries the lane and places each measure when it dispatches it.""" + lane = ( + self.lane(1) + if self.target in self.gpu_lanes + else GpuLane(self.gpu_partition, self.gpu_account) + ) return DispatchedMeasurer( compute=self.compute, run_dir=run_dir, @@ -562,8 +589,10 @@ def measurer( partition=self.partition, eval_minutes=eval_minutes, run_tag=run_tag, - gpu_partition=self.gpu_partition, - gpu_account=self.gpu_account, + gpu_partition=lane.partition, + gpu_account=lane.account, + gpu_type=lane.gpu_type, + gpu_extra=lane.extra, # target-wide, beside the run dirs: every attempt on one base # shares its cached baseline measurement (Benchmark.baseline) baseline_cache=run_dir.parent / "baselines", diff --git a/src/outerloop/tick.py b/src/outerloop/tick.py index cd7bcea2..aec22ab9 100644 --- a/src/outerloop/tick.py +++ b/src/outerloop/tick.py @@ -41,6 +41,7 @@ quote_command, ) from outerloop.disk import DEFAULT_MIN_FREE_BYTES, check_disk +from outerloop.gpu_lanes import GpuLane, gpu_lanes_from_env from outerloop.harness import DEFAULT_MAX_TURNS, ClaudeModelUnset, default_claude_model, redact from outerloop.housekeeping import shed_ended_workspaces from outerloop.ledger_branch import RESEARCH_LOG_BRANCH as RESEARCH_LOG_BRANCH @@ -237,6 +238,8 @@ class ServiceSpec: # here (see MAX_ATTEMPT_JOB_MINUTES). Raise together with job_partition. max_job_minutes: int = MAX_ATTEMPT_JOB_MINUTES qos: str = "" + # Preflight only; wake jobs inherit the environment and resolve their own target. + gpu_lanes: dict[str, GpuLane] = field(default_factory=dict) # Generous vs the ~2 h job walltimes plus queue wait, tight enough that @@ -297,7 +300,7 @@ def _gpu_lane_error(contract: Any, benchmark: str, spec: ServiceSpec) -> str: counts, not just the climbed one: the suite gate measures siblings. A backend without lanes (local compute) runs GPU jobs on the default placement, so the check does not apply.""" - if spec.gpu_partition or not spec.has_lanes: + if spec.gpu_partition or spec.target in spec.gpu_lanes or not spec.has_lanes: return "" gpu_benches = [ b.name for b in getattr(contract, "benchmarks", []) if int(getattr(b, "gpus", 0) or 0) @@ -925,7 +928,11 @@ def write_wake_spec(root: Path, spec: ServiceSpec) -> None: """Publish the tick's wake recipe for the jobs that park runs: a park submits its own wake (`arm_wake`) with exactly the tick's settings, so dispatched wakes stay one recipe with one owner.""" - data = {k: (str(v) if isinstance(v, Path) else v) for k, v in asdict(spec).items()} + data = { + k: (str(v) if isinstance(v, Path) else v) + for k, v in asdict(spec).items() + if k != "gpu_lanes" # deployment config stays out of the persisted wake recipe + } tmp = root / f".{WAKE_SPEC_NAME}.{os.getpid()}.tmp" tmp.write_text(json.dumps(data)) os.replace(tmp, root / WAKE_SPEC_NAME) @@ -3264,10 +3271,13 @@ def _default_image() -> str: return os.path.expanduser("~/outerloop-images/agent-py312.sif") -def _service_spec_from_env(root: Path) -> tuple[Any, ServiceSpec | None]: +def _service_spec_from_env( + root: Path, *, gpu_lanes: dict[str, GpuLane] | None = None +) -> tuple[Any, ServiceSpec | None]: """GitHub client + ServiceSpec from the chain environment, or Nones when the environment is incomplete (the tick then runs without parked servicing, and logs what is absent).""" + lanes = gpu_lanes_from_env() if gpu_lanes is None else gpu_lanes pat_file = os.environ.get("OUTERLOOP_PAT_FILE", "") app_file = os.environ.get("OUTERLOOP_GITHUB_APP_FILE", "") account = os.environ.get("OUTERLOOP_ACCOUNT", "") @@ -3330,6 +3340,7 @@ def _service_spec_from_env(root: Path) -> tuple[Any, ServiceSpec | None]: job_partition=os.environ.get("OUTERLOOP_JOB_PARTITION", ""), gpu_partition=os.environ.get("OUTERLOOP_GPU_PARTITION", ""), gpu_account=os.environ.get("OUTERLOOP_GPU_ACCOUNT", ""), + gpu_lanes=lanes, max_job_minutes=_max_job_minutes_from_env(), has_lanes=compute_from_env().has_lanes, ) @@ -3404,6 +3415,10 @@ def main() -> int: "OUTERLOOP_CADENCE_MIN via the chain's own parser (default 30)", ) args = parser.parse_args() + try: + gpu_lanes = gpu_lanes_from_env() + except ValueError as exc: + parser.error(str(exc)) logging.basicConfig(level=logging.INFO, format="%(asctime)s %(message)s") args.root.mkdir(parents=True, exist_ok=True) @@ -3424,7 +3439,7 @@ def main() -> int: compute = compute_from_env() def run_once() -> None: - github, service_spec = _service_spec_from_env(args.root) + github, service_spec = _service_spec_from_env(args.root, gpu_lanes=gpu_lanes) now = time.time() dispatcher, wake_live = _wake_dispatcher_from_env(compute, service_spec, now, args.root) # parks arm their own wake from this recipe; without it the sweep delivers. diff --git a/tests/test_gpu_lanes.py b/tests/test_gpu_lanes.py new file mode 100644 index 00000000..293d168d --- /dev/null +++ b/tests/test_gpu_lanes.py @@ -0,0 +1,239 @@ +"""Deployment lane validation and the actual eval/author submission seams.""" + +import json +from dataclasses import replace + +import pytest + +from outerloop.attempt import LAUNCH_NICE, _make_launcher, with_seed +from outerloop.compute import CommandResult, JobSpec, SlurmCompute, gpus_in_gres +from outerloop.gpu_lanes import GpuLane, parse_gpu_lanes +from outerloop.measure import DispatchSettings, Measure +from outerloop.syscall import Launch, SyscallRequest + +RAW = json.dumps( + { + "owner/repo": { + "partition": "b200", + "account": "torch_pr_36_mren", + "gpu_type": "b200", + "extra": ["--comment=preemption=yes"], + } + } +) + + +def settings(submitted): + def runner(argv, timeout_s): + submitted.append(argv) + return CommandResult(0, "123\n", "") + + return DispatchSettings( + SlurmCompute(runner=runner), + "/image.sif", + "cpu-account", + "cpu", + gpu_partition="h200", + gpu_account="fleet-account", + gpu_lanes=parse_gpu_lanes(RAW), + target="owner/repo", + ) + + +def test_parse_valid(): + assert parse_gpu_lanes(RAW) == { + "owner/repo": GpuLane("b200", "torch_pr_36_mren", "b200", ("--comment=preemption=yes",)) + } + assert parse_gpu_lanes('{"o/r":{"partition":"b200"}}') == {"o/r": GpuLane("b200")} + assert parse_gpu_lanes("{}") == {} + + +@pytest.mark.parametrize( + "raw", + [ + "{bad", + "[]", + '{"o/r":{"partition":"b200","typo":"x"}}', + '{"o/r":{"account":"a"}}', + '{"o/r":{"partition":4}}', + '{"o/r":{"partition":"p","account":null}}', + '{"o/r":{"partition":"p","gpu_type":4}}', + '{"o/r":{"partition":"p","extra":"--comment=x"}}', + '{"o/r":{"partition":"p","extra":[4]}}', + '{"o/r":{"partition":"p","gpu_type":"b200:8"}}', + '{"repo":{"partition":"p"}}', + ], +) +def test_parse_invalid(raw): + with pytest.raises(ValueError, match=r"^OUTERLOOP_GPU_LANES:"): + parse_gpu_lanes(raw) + + +@pytest.mark.parametrize( + "flag", + [ + "--account=a", + "--partition=p", + "--gres=gpu:8", + "--gpus=8", + "--gpus-per-node=8", + "--time=99", + "--mem=2G", + "--cpus-per-task=8", + "--output=/tmp/x", + "--dependency=afterok:1", + "--qos=high", + "--nice=0", + "--mem-per-gpu=10G", + "--comment", + "-A=a", + "--comment=x\n--account=a", + ], +) +def test_parse_forbidden_extra(flag): + with pytest.raises(ValueError, match=r"^OUTERLOOP_GPU_LANES:"): + parse_gpu_lanes(json.dumps({"o/r": {"partition": "p", "extra": [flag]}})) + + +def test_placement_and_resume(tmp_path): + d = settings([]) + assert d.placement(2) == ("torch_pr_36_mren", "b200") + assert d.placement(0) == ("cpu-account", "cpu") + assert d.lane(0).extra == () + assert replace(d, target="owner/gpt-speedrun").placement(2) == ("fleet-account", "h200") + assert replace(d, gpu_partition="").placement(1) == ("torch_pr_36_mren", "b200") + minimal = replace(d, gpu_lanes={"owner/repo": GpuLane("b200")}) + assert minimal.placement(1) == ("cpu-account", "b200") + # Wake CLI has only a run id; bind the persisted target even with a seed override. + wake = replace(d, target="", seed_cache=tmp_path) + assert with_seed(wake, tmp_path, "owner/repo").lane(1) == d.lane(1) + + +def assert_b200(argv): + assert "--partition=b200" in argv + assert "--account=torch_pr_36_mren" in argv + assert "--gres=gpu:b200:2" in argv + assert "--comment=preemption=yes" in argv + assert not any(arg.startswith("--gpus") for arg in argv) + + +def test_eval_and_author_array_submissions(tmp_path): + submitted: list[list[str]] = [] + d = settings(submitted) + measurer = d.measurer(tmp_path, tmp_path / "repo", 10, "run") + measurer._dispatch(Measure("candidate", "a" * 40, "true", "score", gpus=2)) + assert_b200(submitted[-1]) + assert not any(arg.startswith("--nice") for arg in submitted[-1]) + launcher = _make_launcher(d, tmp_path, tmp_path / "repo", "run", gpus=2) + request = SyscallRequest((Launch("probe", "true", 10, array=4, concurrency=2),)) + assert launcher("a" * 40, request) == "afterany:123" + assert_b200(submitted[-1]) + assert "--array=0-3%2" in submitted[-1] + assert f"--nice={LAUNCH_NICE}" in submitted[-1] + + +def test_unlaned_author_golden_argv(tmp_path): + submitted: list[list[str]] = [] + d = replace(settings(submitted), target="owner/gpt-speedrun") + _make_launcher(d, tmp_path, tmp_path / "repo", "run", gpus=2)( + "a" * 40, SyscallRequest((Launch("probe", "true", 10),)) + ) + assert submitted[-1] == [ + "sbatch", + "--parsable", + "--job-name=run-launch-probe", + "--time=20", + "--cpus-per-task=16", + "--mem=128G", + "--output=/dev/null", + "--account=fleet-account", + "--partition=h200", + "--gpus-per-node=2", + "--nice=5000", + str(tmp_path / "eval-launch-probe" / "job.sh"), + ] + + +def test_typed_gres_and_legacy_argv(): + spec = JobSpec("j", "a", "p", 10, command="true", gpus=2) + assert spec.to_argv() == [ + "sbatch", + "--parsable", + "--job-name=j", + "--time=10", + "--cpus-per-task=1", + "--mem=2G", + "--output=/dev/null", + "--account=a", + "--partition=p", + "--gpus-per-node=2", + "--wrap=true", + ] + assert "--gres=gpu:b200:2" in replace(spec, gpu_type="b200").to_argv() + assert gpus_in_gres("gres/gpu:b200:2") == 2 + + +@pytest.mark.parametrize("module", ["outerloop.tick", "outerloop.attempt"]) +def test_bad_json_is_one_startup_error(monkeypatch, capsys, tmp_path, module): + import importlib + + monkeypatch.setenv("OUTERLOOP_GPU_LANES", "{bad") + monkeypatch.setattr( + "sys.argv", + [module, "--root" if module == "outerloop.tick" else "--run-root", str(tmp_path)], + ) + with pytest.raises(SystemExit) as exc: + importlib.import_module(module).main() + assert exc.value.code == 2 + error = capsys.readouterr().err + assert error.count("OUTERLOOP_GPU_LANES:") == 1 + assert "Traceback" not in error + + +def test_cpu_submission_ignores_target_gpu_flags(tmp_path): + submitted: list[list[str]] = [] + d = settings(submitted) + d.measurer(tmp_path, tmp_path / "repo", 10, "run")._dispatch( + Measure("cpu", "a" * 40, "true", "score") + ) + argv = submitted[-1] + assert "--account=cpu-account" in argv and "--partition=cpu" in argv + assert not any(a.startswith(("--gres", "--gpus", "--comment")) for a in argv) + + +def test_tick_preflight_accepts_override_without_fleet_lane(monkeypatch, tmp_path): + from types import SimpleNamespace + + from outerloop.tick import _gpu_lane_error, _service_spec_from_env + + image = tmp_path / "image.sif" + image.touch() + pat = tmp_path / "pat" + pat.write_text("test-token") + monkeypatch.setattr( + "os.environ", + { + "OUTERLOOP_PAT_FILE": str(pat), + "OUTERLOOP_IMAGE": str(image), + "OUTERLOOP_HOME": str(tmp_path), + "OUTERLOOP_TARGET": "owner/repo", + "OUTERLOOP_BOT_LOGIN": "bot", + "OUTERLOOP_GPU_LANES": RAW, + }, + ) + _, spec = _service_spec_from_env(tmp_path) + assert spec is not None + assert spec.gpu_partition == "" and spec.gpu_account == "" + assert spec.gpu_lanes["owner/repo"].partition == "b200" + contract = SimpleNamespace(benchmarks=[SimpleNamespace(name="bench", gpus=2)]) + assert _gpu_lane_error(contract, "bench", spec) == "" + + # The legacy wake recipe stays byte-for-byte the same with overrides configured. + from outerloop.tick import WAKE_SPEC_NAME, load_wake_spec, write_wake_spec + + write_wake_spec(tmp_path, replace(spec, gpu_lanes={})) + legacy = (tmp_path / WAKE_SPEC_NAME).read_bytes() + write_wake_spec(tmp_path, spec) + assert (tmp_path / WAKE_SPEC_NAME).read_bytes() == legacy + restored = load_wake_spec(tmp_path) + assert restored is not None and restored.gpu_lanes == {} diff --git a/tests/test_start.py b/tests/test_start.py index a3d76764..8f40e809 100644 --- a/tests/test_start.py +++ b/tests/test_start.py @@ -1019,7 +1019,7 @@ def test_resident_tick_takes_no_root_lease(tmp_path, monkeypatch): monkeypatch.setenv("OUTERLOOP_COMPUTE", "slurm") monkeypatch.delenv("OUTERLOOP_TICK_HOST", raising=False) monkeypatch.setattr(sys, "argv", ["tick", "--root", str(tmp_path)]) - monkeypatch.setattr(mod, "_service_spec_from_env", lambda root: (None, None)) + monkeypatch.setattr(mod, "_service_spec_from_env", lambda root, **kwargs: (None, None)) monkeypatch.setattr(mod, "tick", lambda *a, **k: mod.TickReport()) assert mod.main() == 0 assert not (tmp_path / "TICK").exists() diff --git a/tests/test_tick.py b/tests/test_tick.py index 1d52d321..e26d2b51 100644 --- a/tests/test_tick.py +++ b/tests/test_tick.py @@ -4463,7 +4463,7 @@ def test_tick_main_releases_root_lease(tmp_path, monkeypatch, loop, ending): monkeypatch.setattr( sys, "argv", ["tick", "--root", str(tmp_path)] + (["--loop"] if loop else []) ) - monkeypatch.setattr(mod, "_service_spec_from_env", lambda root: (None, None)) + monkeypatch.setattr(mod, "_service_spec_from_env", lambda root, **kwargs: (None, None)) seen = [] def one_tick(*args, **kwargs): @@ -4613,7 +4613,7 @@ def test_sigterm_handler_is_installed_before_the_lease(tmp_path, monkeypatch): monkeypatch.setenv("OUTERLOOP_COMPUTE", "local") monkeypatch.setattr(sys, "argv", ["tick", "--root", str(tmp_path), "--loop"]) - monkeypatch.setattr(mod, "_service_spec_from_env", lambda root: (None, None)) + monkeypatch.setattr(mod, "_service_spec_from_env", lambda root, **kwargs: (None, None)) seen = [] def acquire(*args, **kwargs): From fd5b82745848c1071bf25ce69a1509270d2e8cb8 Mon Sep 17 00:00:00 2001 From: Mengye Ren Date: Sat, 26 Sep 2026 15:57:41 -0400 Subject: [PATCH 2/5] Generic examples in the lanes docs and tests; refused-flag list matches the code Co-Authored-By: Claude Opus 5.5 (1M context) --- docs/install.md | 15 ++++++++------- tests/test_gpu_lanes.py | 18 +++++++++--------- 2 files changed, 17 insertions(+), 16 deletions(-) diff --git a/docs/install.md b/docs/install.md index d274ee14..a405a1e1 100644 --- a/docs/install.md +++ b/docs/install.md @@ -467,17 +467,18 @@ backend); the author backend is For a target-specific GPU lane, add a JSON mapping to the deployment's `.env`: ```bash -OUTERLOOP_GPU_LANES='{"owner/repo":{"partition":"b200","account":"torch_pr_36_mren","gpu_type":"b200","extra":["--comment=preemption=yes"]}}' +OUTERLOOP_GPU_LANES='{"owner/repo":{"partition":"gpu-large","account":"my-account","gpu_type":"b200","extra":["--comment=reserved"]}}' ``` -Replace `owner/repo` with the reserved node's target. This Torch lane submits -`--account=torch_pr_36_mren --partition=b200 --gres=gpu:b200:N --comment=preemption=yes` -for that target's GPU evals and author launches (including arrays and re-measures). -Other targets, such as gpt-speedrun, keep the fleet's H200 lane; CPU jobs are +This lane submits `--account=my-account --partition=gpu-large --gres=gpu:b200:N +--comment=reserved` for that target's GPU evals and author launches (including +arrays and re-measures). Other targets keep the fleet GPU lane; CPU jobs are unchanged. Only `partition` is required; an omitted `account` uses the CPU/default account, and omitted `gpu_type` preserves untyped per-node GPU requests. `extra` -is a list of `--name=value` flags; kernel-owned account, partition, gres, gpus*, -time, mem, and cpus* flags are rejected, as are unknown keys and malformed JSON. +is a list of `--name=value` flags. A flag the kernel sets itself (account, +partition, gres, gpus*, cpus*, mem*, time, qos, nice, array, dependency, begin, +job-name, output, error, wrap, parsable, chdir) is rejected, since sbatch lets the +later flag win; so are unknown keys and malformed JSON. These are cluster settings, not target contract fields. `OUTERLOOP_CLAUDE_MODEL` names the model for every Claude role (author, diff --git a/tests/test_gpu_lanes.py b/tests/test_gpu_lanes.py index 293d168d..743dbb28 100644 --- a/tests/test_gpu_lanes.py +++ b/tests/test_gpu_lanes.py @@ -15,9 +15,9 @@ { "owner/repo": { "partition": "b200", - "account": "torch_pr_36_mren", + "account": "lab-account", "gpu_type": "b200", - "extra": ["--comment=preemption=yes"], + "extra": ["--comment=reserved"], } } ) @@ -42,7 +42,7 @@ def runner(argv, timeout_s): def test_parse_valid(): assert parse_gpu_lanes(RAW) == { - "owner/repo": GpuLane("b200", "torch_pr_36_mren", "b200", ("--comment=preemption=yes",)) + "owner/repo": GpuLane("b200", "lab-account", "b200", ("--comment=reserved",)) } assert parse_gpu_lanes('{"o/r":{"partition":"b200"}}') == {"o/r": GpuLane("b200")} assert parse_gpu_lanes("{}") == {} @@ -97,11 +97,11 @@ def test_parse_forbidden_extra(flag): def test_placement_and_resume(tmp_path): d = settings([]) - assert d.placement(2) == ("torch_pr_36_mren", "b200") + assert d.placement(2) == ("lab-account", "b200") assert d.placement(0) == ("cpu-account", "cpu") assert d.lane(0).extra == () - assert replace(d, target="owner/gpt-speedrun").placement(2) == ("fleet-account", "h200") - assert replace(d, gpu_partition="").placement(1) == ("torch_pr_36_mren", "b200") + assert replace(d, target="owner/other-repo").placement(2) == ("fleet-account", "h200") + assert replace(d, gpu_partition="").placement(1) == ("lab-account", "b200") minimal = replace(d, gpu_lanes={"owner/repo": GpuLane("b200")}) assert minimal.placement(1) == ("cpu-account", "b200") # Wake CLI has only a run id; bind the persisted target even with a seed override. @@ -111,9 +111,9 @@ def test_placement_and_resume(tmp_path): def assert_b200(argv): assert "--partition=b200" in argv - assert "--account=torch_pr_36_mren" in argv + assert "--account=lab-account" in argv assert "--gres=gpu:b200:2" in argv - assert "--comment=preemption=yes" in argv + assert "--comment=reserved" in argv assert not any(arg.startswith("--gpus") for arg in argv) @@ -134,7 +134,7 @@ def test_eval_and_author_array_submissions(tmp_path): def test_unlaned_author_golden_argv(tmp_path): submitted: list[list[str]] = [] - d = replace(settings(submitted), target="owner/gpt-speedrun") + d = replace(settings(submitted), target="owner/other-repo") _make_launcher(d, tmp_path, tmp_path / "repo", "run", gpus=2)( "a" * 40, SyscallRequest((Launch("probe", "true", 10),)) ) From 35b4cd85710318c60f56236c4862085e250d58ee Mon Sep 17 00:00:00 2001 From: Mengye Ren Date: Sat, 26 Sep 2026 16:06:01 -0400 Subject: [PATCH 3/5] Lanes docs and tests name no specific hardware Co-Authored-By: Claude Opus 5.5 (1M context) --- docs/install.md | 4 ++-- tests/test_gpu_lanes.py | 42 ++++++++++++++++++++--------------------- 2 files changed, 23 insertions(+), 23 deletions(-) diff --git a/docs/install.md b/docs/install.md index a405a1e1..5ee80145 100644 --- a/docs/install.md +++ b/docs/install.md @@ -467,10 +467,10 @@ backend); the author backend is For a target-specific GPU lane, add a JSON mapping to the deployment's `.env`: ```bash -OUTERLOOP_GPU_LANES='{"owner/repo":{"partition":"gpu-large","account":"my-account","gpu_type":"b200","extra":["--comment=reserved"]}}' +OUTERLOOP_GPU_LANES='{"owner/repo":{"partition":"gpu-large","account":"my-account","gpu_type":"a100","extra":["--comment=reserved"]}}' ``` -This lane submits `--account=my-account --partition=gpu-large --gres=gpu:b200:N +This lane submits `--account=my-account --partition=gpu-large --gres=gpu:a100:N --comment=reserved` for that target's GPU evals and author launches (including arrays and re-measures). Other targets keep the fleet GPU lane; CPU jobs are unchanged. Only `partition` is required; an omitted `account` uses the CPU/default diff --git a/tests/test_gpu_lanes.py b/tests/test_gpu_lanes.py index 743dbb28..74e9fbba 100644 --- a/tests/test_gpu_lanes.py +++ b/tests/test_gpu_lanes.py @@ -14,9 +14,9 @@ RAW = json.dumps( { "owner/repo": { - "partition": "b200", + "partition": "gpu-large", "account": "lab-account", - "gpu_type": "b200", + "gpu_type": "a100", "extra": ["--comment=reserved"], } } @@ -33,7 +33,7 @@ def runner(argv, timeout_s): "/image.sif", "cpu-account", "cpu", - gpu_partition="h200", + gpu_partition="gpu-fleet", gpu_account="fleet-account", gpu_lanes=parse_gpu_lanes(RAW), target="owner/repo", @@ -42,9 +42,9 @@ def runner(argv, timeout_s): def test_parse_valid(): assert parse_gpu_lanes(RAW) == { - "owner/repo": GpuLane("b200", "lab-account", "b200", ("--comment=reserved",)) + "owner/repo": GpuLane("gpu-large", "lab-account", "a100", ("--comment=reserved",)) } - assert parse_gpu_lanes('{"o/r":{"partition":"b200"}}') == {"o/r": GpuLane("b200")} + assert parse_gpu_lanes('{"o/r":{"partition":"gpu-large"}}') == {"o/r": GpuLane("gpu-large")} assert parse_gpu_lanes("{}") == {} @@ -53,14 +53,14 @@ def test_parse_valid(): [ "{bad", "[]", - '{"o/r":{"partition":"b200","typo":"x"}}', + '{"o/r":{"partition":"gpu-large","typo":"x"}}', '{"o/r":{"account":"a"}}', '{"o/r":{"partition":4}}', '{"o/r":{"partition":"p","account":null}}', '{"o/r":{"partition":"p","gpu_type":4}}', '{"o/r":{"partition":"p","extra":"--comment=x"}}', '{"o/r":{"partition":"p","extra":[4]}}', - '{"o/r":{"partition":"p","gpu_type":"b200:8"}}', + '{"o/r":{"partition":"p","gpu_type":"a100:8"}}', '{"repo":{"partition":"p"}}', ], ) @@ -97,22 +97,22 @@ def test_parse_forbidden_extra(flag): def test_placement_and_resume(tmp_path): d = settings([]) - assert d.placement(2) == ("lab-account", "b200") + assert d.placement(2) == ("lab-account", "gpu-large") assert d.placement(0) == ("cpu-account", "cpu") assert d.lane(0).extra == () - assert replace(d, target="owner/other-repo").placement(2) == ("fleet-account", "h200") - assert replace(d, gpu_partition="").placement(1) == ("lab-account", "b200") - minimal = replace(d, gpu_lanes={"owner/repo": GpuLane("b200")}) - assert minimal.placement(1) == ("cpu-account", "b200") + assert replace(d, target="owner/other-repo").placement(2) == ("fleet-account", "gpu-fleet") + assert replace(d, gpu_partition="").placement(1) == ("lab-account", "gpu-large") + minimal = replace(d, gpu_lanes={"owner/repo": GpuLane("gpu-large")}) + assert minimal.placement(1) == ("cpu-account", "gpu-large") # Wake CLI has only a run id; bind the persisted target even with a seed override. wake = replace(d, target="", seed_cache=tmp_path) assert with_seed(wake, tmp_path, "owner/repo").lane(1) == d.lane(1) -def assert_b200(argv): - assert "--partition=b200" in argv +def assert_lane(argv): + assert "--partition=gpu-large" in argv assert "--account=lab-account" in argv - assert "--gres=gpu:b200:2" in argv + assert "--gres=gpu:a100:2" in argv assert "--comment=reserved" in argv assert not any(arg.startswith("--gpus") for arg in argv) @@ -122,12 +122,12 @@ def test_eval_and_author_array_submissions(tmp_path): d = settings(submitted) measurer = d.measurer(tmp_path, tmp_path / "repo", 10, "run") measurer._dispatch(Measure("candidate", "a" * 40, "true", "score", gpus=2)) - assert_b200(submitted[-1]) + assert_lane(submitted[-1]) assert not any(arg.startswith("--nice") for arg in submitted[-1]) launcher = _make_launcher(d, tmp_path, tmp_path / "repo", "run", gpus=2) request = SyscallRequest((Launch("probe", "true", 10, array=4, concurrency=2),)) assert launcher("a" * 40, request) == "afterany:123" - assert_b200(submitted[-1]) + assert_lane(submitted[-1]) assert "--array=0-3%2" in submitted[-1] assert f"--nice={LAUNCH_NICE}" in submitted[-1] @@ -147,7 +147,7 @@ def test_unlaned_author_golden_argv(tmp_path): "--mem=128G", "--output=/dev/null", "--account=fleet-account", - "--partition=h200", + "--partition=gpu-fleet", "--gpus-per-node=2", "--nice=5000", str(tmp_path / "eval-launch-probe" / "job.sh"), @@ -169,8 +169,8 @@ def test_typed_gres_and_legacy_argv(): "--gpus-per-node=2", "--wrap=true", ] - assert "--gres=gpu:b200:2" in replace(spec, gpu_type="b200").to_argv() - assert gpus_in_gres("gres/gpu:b200:2") == 2 + assert "--gres=gpu:a100:2" in replace(spec, gpu_type="a100").to_argv() + assert gpus_in_gres("gres/gpu:a100:2") == 2 @pytest.mark.parametrize("module", ["outerloop.tick", "outerloop.attempt"]) @@ -224,7 +224,7 @@ def test_tick_preflight_accepts_override_without_fleet_lane(monkeypatch, tmp_pat _, spec = _service_spec_from_env(tmp_path) assert spec is not None assert spec.gpu_partition == "" and spec.gpu_account == "" - assert spec.gpu_lanes["owner/repo"].partition == "b200" + assert spec.gpu_lanes["owner/repo"].partition == "gpu-large" contract = SimpleNamespace(benchmarks=[SimpleNamespace(name="bench", gpus=2)]) assert _gpu_lane_error(contract, "bench", spec) == "" From 8ee5e076755623da35be788ac12b5d2601c59bfc Mon Sep 17 00:00:00 2001 From: Mengye Ren Date: Sat, 26 Sep 2026 16:58:47 -0400 Subject: [PATCH 4/5] Install guide: typed lanes request GPUs with --gres Co-Authored-By: Claude Opus 5.5 (1M context) --- docs/install.md | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/docs/install.md b/docs/install.md index 5ee80145..69f9ebc5 100644 --- a/docs/install.md +++ b/docs/install.md @@ -495,7 +495,8 @@ so it also needs the image cluster, evals run inside the Apptainer image at `OUTERLOOP_IMAGE` (default `~/outerloop-images/agent-py312.sif`) in a jail that binds only the checked-out tree — an eval that needs data must fetch it into the tree, and -GPU jobs are requested per node (`--gpus-per-node`). The tick has three +GPU jobs are requested per node: `--gpus-per-node=N`, or `--gres=gpu::N` +for a lane with a GPU type (see `OUTERLOOP_GPU_LANES`). The tick has three scheduling knobs. `OUTERLOOP_CADENCE_MIN`, read from the `.env`, is how often the chain ticks (minutes; default 30). Two finer ones are read from the tick's own environment (set at launch, not the per-tick `.env`): `OUTERLOOP_MIN_TICK_MINUTES` From 86b028f53fd76f8e856efd047af91af84e0708e7 Mon Sep 17 00:00:00 2001 From: Mengye Ren Date: Wed, 30 Sep 2026 09:55:10 -0400 Subject: [PATCH 5/5] GPU lanes: node, task and exclusivity flags are the kernel's too Co-Authored-By: Claude Opus 5.5 (1M context) --- src/outerloop/gpu_lanes.py | 6 +++++- tests/test_gpu_lanes.py | 4 ++++ 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/src/outerloop/gpu_lanes.py b/src/outerloop/gpu_lanes.py index 413c8155..48183ee9 100644 --- a/src/outerloop/gpu_lanes.py +++ b/src/outerloop/gpu_lanes.py @@ -78,7 +78,11 @@ def parse_gpu_lanes(raw: str) -> dict[str, GpuLane]: raise ValueError(f"{target}: extra flags must have the form --name=value") name = flag[2:].split("=", 1)[0] # later flags win in sbatch, so an extra must not repeat one the kernel sets - if name in KERNEL_FLAGS or name.startswith(("gpus", "cpus", "mem")): + # node, task and exclusivity counts multiply a per-node GPU request past + # what the caps charge, so they are the kernel's too + if name in KERNEL_FLAGS or name.startswith( + ("gpus", "cpus", "mem", "nodes", "ntasks", "exclusive", "tres-per") + ): raise ValueError(f"{target}: kernel-owned extra flag --{name}") lanes[target] = GpuLane( lane["partition"], lane.get("account", ""), gpu_type, tuple(extra) diff --git a/tests/test_gpu_lanes.py b/tests/test_gpu_lanes.py index 74e9fbba..11527388 100644 --- a/tests/test_gpu_lanes.py +++ b/tests/test_gpu_lanes.py @@ -88,6 +88,10 @@ def test_parse_invalid(raw): "--comment", "-A=a", "--comment=x\n--account=a", + "--nodes=2", + "--ntasks-per-node=4", + "--exclusive=user", + "--tres-per-task=gres/gpu:2", ], ) def test_parse_forbidden_extra(flag):