From b7cf05bc9293b9ea214f05dd19a340a777ab2dfc Mon Sep 17 00:00:00 2001 From: "Ma, Guokai" Date: Thu, 3 Sep 2026 14:31:06 +0800 Subject: [PATCH 1/2] Raise modal CI timeouts to absorb slower sandbox provisioning The modal-torch-latest GPU job is killed at its 75-minute cap before the test suite finishes. On master this now happens in roughly half the runs. The suite is not what grew. Phase timings taken from the job logs show the regression is confined to Modal sandbox provisioning, which went from about 0.1 min in late August to 18-22 min from 2026-08-31 onward, while pytest itself stayed in its usual 38-49 min band for a near-identical test count: run provisioning env setup pytest tests 08-22 12:53 0.1m 6.0m 37.5m 1150 passed 08-29 00:40 0.2m 6.2m 39.3m 1194 passed 09-01 11:47 17.8m 6.6m 48.4m 1241 passed 09-02 04:00 22.0m 4.8m 38.5m 1241 passed Raise the outer GitHub job budget to 90 minutes. Raise the inner Modal sandbox lifetime to 70 minutes as well: its clock starts when the container starts, so it has to cover env setup plus pytest, which has already been observed at 55 min, leaving only 5 minutes of headroom against the old 3600s value. This only buys back the margin the provisioning regression consumed. The regression itself still needs investigation on the Modal side. Signed-off-by: Ma, Guokai --- .github/workflows/modal-torch-latest.yml | 2 +- ci/test_torch_latest.py | 4 ++-- ci/torch_latest.py | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/.github/workflows/modal-torch-latest.yml b/.github/workflows/modal-torch-latest.yml index e96d156fe886..f64fe87dd87d 100644 --- a/.github/workflows/modal-torch-latest.yml +++ b/.github/workflows/modal-torch-latest.yml @@ -125,7 +125,7 @@ jobs: deploy: name: modal-torch-latest / DeepSpeedAI CI runs-on: ubuntu-latest - timeout-minutes: 75 + timeout-minutes: 90 permissions: contents: read needs: collect-tests diff --git a/ci/test_torch_latest.py b/ci/test_torch_latest.py index f765a89f43d3..a6ce4ca2abf1 100644 --- a/ci/test_torch_latest.py +++ b/ci/test_torch_latest.py @@ -407,7 +407,7 @@ def test_remote_plan_is_structural_and_preserves_order_and_scope(): def test_sandbox_kwargs_are_fixed_and_secret_free(): kwargs = torch_latest.build_sandbox_kwargs("image") assert kwargs["gpu"] == "l40s:2" - assert kwargs["timeout"] == 3600 + assert kwargs["timeout"] == 4200 assert kwargs["secrets"] == [] assert kwargs["network_file_systems"] == {} assert kwargs["volumes"] == {} @@ -547,7 +547,7 @@ def test_workflow_keeps_github_execution_trusted_and_preserves_modes(): assert "HF_TOKEN" not in text assert "modal==1.2.6" in text assert "timeout-minutes: 20" in text - assert "timeout-minutes: 75" in text + assert "timeout-minutes: 90" in text assert text.count("persist-credentials: false") == 2 assert text.count("lfs: false") == 2 assert text.count("submodules: false") == 2 diff --git a/ci/torch_latest.py b/ci/torch_latest.py index baf6099c094f..11c98de57229 100644 --- a/ci/torch_latest.py +++ b/ci/torch_latest.py @@ -67,7 +67,7 @@ } PYTORCH_CUDA_128_INDEX_URL = "https://download.pytorch.org/whl/cu128" APP_NAME = "deepspeedai-torch-latest-ci" -SANDBOX_TIMEOUT_SECONDS = 3600 +SANDBOX_TIMEOUT_SECONDS = 4200 MAX_TEST_LIST_BYTES = 64 * 1024 MAX_TEST_TARGETS = 1024 MAX_DISPLAY_BYTES_PER_COMMAND = 16 * 1024 * 1024 From c59ce79cdce8411ee1f4753d6eb6f327b525a52d Mon Sep 17 00:00:00 2001 From: "Ma, Guokai" Date: Thu, 3 Sep 2026 15:22:08 +0800 Subject: [PATCH 2/2] Split the modal CI budget into acquisition and test phases A single wall-clock budget cannot tell "we never got a GPU" apart from "the tests ran long". Both currently surface as the same job timeout, and a run that waits 22 minutes for capacity spends that time out of the budget the tests still need. Sandbox.create() returns before the container exists, so the wait for the l40s:2 reservation surfaces on the first exec. Bound that wait on its own: - acquisition: 30 min (SANDBOX_ACQUIRE_TIMEOUT_SECONDS). Past that the run aborts with SandboxStartTimeout, which says no test ran, instead of holding a runner for the rest of the budget. - tests: 70 min, unchanged. The Sandbox lifetime clock starts when the container starts, so this budget is always fully available once a GPU is reserved, no matter how long acquisition took. - the job timeout becomes a backstop at 105 min, covering 30 + 70 plus runner setup and cleanup. The startup duration is now printed with flush=True. Sandbox output is otherwise block-buffered and lost when the job is killed, which is why the timed-out runs show a silent gap rather than any progress. Observed acquisition times were 12s, 0.9, 1.3, 7.9, 10.8, 17.8 and 22.0 min, so 30 min leaves headroom over the worst case while still failing fast. Signed-off-by: Ma, Guokai --- .github/workflows/modal-torch-latest.yml | 2 +- ci/test_torch_latest.py | 46 ++++++++++++++++++++++-- ci/torch_latest.py | 43 ++++++++++++++++++++++ 3 files changed, 88 insertions(+), 3 deletions(-) diff --git a/.github/workflows/modal-torch-latest.yml b/.github/workflows/modal-torch-latest.yml index f64fe87dd87d..c5d66fbdb295 100644 --- a/.github/workflows/modal-torch-latest.yml +++ b/.github/workflows/modal-torch-latest.yml @@ -125,7 +125,7 @@ jobs: deploy: name: modal-torch-latest / DeepSpeedAI CI runs-on: ubuntu-latest - timeout-minutes: 90 + timeout-minutes: 105 permissions: contents: read needs: collect-tests diff --git a/ci/test_torch_latest.py b/ci/test_torch_latest.py index a6ce4ca2abf1..71d45358d67e 100644 --- a/ci/test_torch_latest.py +++ b/ci/test_torch_latest.py @@ -13,6 +13,7 @@ import subprocess import sys import tempfile +import threading from pathlib import Path from types import SimpleNamespace @@ -100,11 +101,13 @@ def __init__( fail_label: str | None = None, cleanup_failure: bool = False, wait_failure: bool = False, + never_starts: bool = False, ): self.candidate_sha = candidate_sha self.fail_label = fail_label self.cleanup_failure = cleanup_failure self.wait_failure = wait_failure + self.never_starts = never_starts self.exec_calls = [] self.processes = [] self.terminated = False @@ -112,6 +115,9 @@ def __init__( def exec(self, *args, **kwargs): self.exec_calls.append((args, kwargs)) + if self.never_starts: + # A container that never gets a GPU never returns from its first exec. + threading.Event().wait() lines = [self.candidate_sha + "\n"] if "rev-parse" in args and "HEAD^{commit}" in args else ["ok\n"] label_failure = self.fail_label and self.fail_label in " ".join(args) process = FakeProcess(lines, return_code=9 if label_failure else 0) @@ -135,9 +141,10 @@ def _fake_modal( cleanup_failure: bool = False, wait_failure: bool = False, create_failure: bool = False, + never_starts: bool = False, ): state = SimpleNamespace(image_calls=[], app_calls=[], create_calls=[]) - sandbox = FakeSandbox(candidate_sha, fail_label, cleanup_failure, wait_failure) + sandbox = FakeSandbox(candidate_sha, fail_label, cleanup_failure, wait_failure, never_starts) class Image: @@ -408,6 +415,7 @@ def test_sandbox_kwargs_are_fixed_and_secret_free(): kwargs = torch_latest.build_sandbox_kwargs("image") assert kwargs["gpu"] == "l40s:2" assert kwargs["timeout"] == 4200 + assert torch_latest.SANDBOX_ACQUIRE_TIMEOUT_SECONDS == 1800 assert kwargs["secrets"] == [] assert kwargs["network_file_systems"] == {} assert kwargs["volumes"] == {} @@ -444,6 +452,40 @@ def test_controller_creates_one_sandbox_without_forwarding_secrets_and_cleans_up shutil.rmtree(root, ignore_errors=True) +def test_await_sandbox_start_gives_up_when_the_container_never_runs(): + # Catches a controller that blocks forever on a GPU reservation that is never satisfied. + sandbox = FakeSandbox("a" * 40, never_starts=True) + error = _expect_error( + torch_latest.await_sandbox_start, + sandbox, + 0.05, + exception=torch_latest.SandboxStartTimeout, + ) + assert "no test ran" in str(error) + + +def test_await_sandbox_start_reports_startup_duration(): + sandbox = FakeSandbox("a" * 40) + assert torch_latest.await_sandbox_start(sandbox, 30) >= 0 + + +def test_controller_aborts_without_running_tests_when_sandbox_never_starts(): + # Catches a controller that spends the whole job budget waiting, or that runs commands + # against a Sandbox that never started, or that leaks the Sandbox when startup times out. + root, path = _selection_file("tests/unit/v1\n") + original = torch_latest.SANDBOX_ACQUIRE_TIMEOUT_SECONDS + torch_latest.SANDBOX_ACQUIRE_TIMEOUT_SECONDS = 0.05 + try: + env = _valid_env(path) + fake, _, sandbox = _fake_modal("a" * 40, never_starts=True) + _expect_error(torch_latest.run_controller, env, fake, exception=torch_latest.SandboxStartTimeout) + assert sandbox.terminated + assert not any("pytest" in " ".join(args) for args, _ in sandbox.exec_calls) + finally: + torch_latest.SANDBOX_ACQUIRE_TIMEOUT_SECONDS = original + shutil.rmtree(root, ignore_errors=True) + + def test_controller_propagates_command_and_cleanup_failures(): root, path = _selection_file("tests/unit/v1\n") try: @@ -547,7 +589,7 @@ def test_workflow_keeps_github_execution_trusted_and_preserves_modes(): assert "HF_TOKEN" not in text assert "modal==1.2.6" in text assert "timeout-minutes: 20" in text - assert "timeout-minutes: 90" in text + assert "timeout-minutes: 105" in text assert text.count("persist-credentials: false") == 2 assert text.count("lfs: false") == 2 assert text.count("submodules: false") == 2 diff --git a/ci/torch_latest.py b/ci/torch_latest.py index 11c98de57229..dda978a508e9 100644 --- a/ci/torch_latest.py +++ b/ci/torch_latest.py @@ -22,6 +22,8 @@ import shutil import stat import subprocess +import threading +import time from dataclasses import dataclass, replace from pathlib import Path, PurePosixPath from typing import Any, Mapping, Sequence @@ -68,6 +70,7 @@ PYTORCH_CUDA_128_INDEX_URL = "https://download.pytorch.org/whl/cu128" APP_NAME = "deepspeedai-torch-latest-ci" SANDBOX_TIMEOUT_SECONDS = 4200 +SANDBOX_ACQUIRE_TIMEOUT_SECONDS = 1800 MAX_TEST_LIST_BYTES = 64 * 1024 MAX_TEST_TARGETS = 1024 MAX_DISPLAY_BYTES_PER_COMMAND = 16 * 1024 * 1024 @@ -115,6 +118,15 @@ def __init__(self, primary: BaseException, cleanup: BaseException): self.cleanup = cleanup +class SandboxStartTimeout(RuntimeError): + """The Sandbox never started, so no test ever ran.""" + + def __init__(self, timeout_seconds: float): + super().__init__(f"Sandbox did not start within {timeout_seconds:g}s, so no test ran. This is a capacity " + f"problem rather than a test failure: the GPU reservation was never satisfied.") + self.timeout_seconds = timeout_seconds + + def validate_repository(value: str) -> str: if not isinstance(value, str) or not _REPOSITORY_RE.fullmatch(value): raise ValueError("repository must be an ASCII owner/name pair") @@ -582,6 +594,35 @@ def _cleanup_sandbox(sandbox: Any) -> None: raise observation_error.with_traceback(observation_error.__traceback__) +def await_sandbox_start(sandbox: Any, timeout_seconds: float | None = None) -> float: + """Block until the Sandbox container is running, and return how long that took. + + ``Sandbox.create`` returns before the container exists, so the wait for a free GPU surfaces on the first + ``exec`` instead. Bounding that wait on its own keeps an unsatisfied reservation from consuming the whole + job budget, and keeps the Sandbox lifetime budget available for the tests that follow. + """ + if timeout_seconds is None: + timeout_seconds = SANDBOX_ACQUIRE_TIMEOUT_SECONDS + started_at = time.monotonic() + probe_result: list[BaseException | None] = [] + + def probe() -> None: + try: + sandbox.exec("true").wait() + probe_result.append(None) + except BaseException as exc: # surfaced on the calling thread below + probe_result.append(exc) + + probe_thread = threading.Thread(target=probe, daemon=True) + probe_thread.start() + probe_thread.join(timeout_seconds) + if probe_thread.is_alive(): + raise SandboxStartTimeout(timeout_seconds) + if probe_result and probe_result[0] is not None: + raise probe_result[0] + return time.monotonic() - started_at + + def run_controller(env: Mapping[str, str], modal_module: Any | None = None) -> int: inputs = resolve_controller_inputs(env) if inputs.selection_mode == "none": @@ -604,6 +645,8 @@ def run_controller(env: Mapping[str, str], modal_module: Any | None = None) -> i cleanup_error: BaseException | None = None try: sandbox = modal_module.Sandbox.create(app=app, **build_sandbox_kwargs(image)) + startup_seconds = await_sandbox_start(sandbox) + print(f"Sandbox started after {startup_seconds:.0f}s", flush=True) for command in build_remote_commands(inputs): run_sandbox_command(sandbox, modal_module, command) except BaseException as exc: