diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 0d264445c..5ecf884b2 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -139,7 +139,7 @@ Two directors, selected by role: Auto mode is toggled by CLI flags (`--auto` / `--no-auto`); there is currently no in-session key to toggle it (default on; constrained envelope — workspace writes and unconstrained shell auto-allow; installs, recursive rm, force/uncontained worktree changes, sensitive-path and opaque-wrapper shell still ask; contained non-force `git worktree add`/`remove`/`prune` and `list` auto-allow; shell file-mutation denied). It is not a separate edit/plan mode. - **SubAgentDirector** (delegated work, `src/subagent/index.ts`) — Drives a dispatched worker until a turn arrives with no tool calls, then replies with the final assistant text and ends the run. A tool-less turn **after tools** completes only with the four-heading envelope (Summary, Findings, Blockers, Paths). Assistant text that prints explicit `` markup is treated as attempted tool use, not narration: one **verbatim-tool-call** nudge asks the worker to re-issue a real `tool_call` and does not count toward the tool-less spiral. A missing envelope otherwise nudges once (**incomplete-report**) and a second tool-less turn still without the envelope salvages as **incomplete-report-stop** (once; later tool-less turns wait). Explore/read-only workers that used tools then replied with findings remain normal completes; `requireEvidence` (off by default, set per director) additionally requires at least one read before a tool-less spawn-only reply can complete. `requirePlanSubstance` (counsel or `intent=plan`, not `modelRole === "plan"`) additionally requires Findings to contain files/paths, acceptance criteria, non-goals, risks, and ordered steps with a non-placeholder line each — four headings with stub Findings are incomplete-report, not an attachable plan. After real tool work, wrap-up Findings that are not placeholder or outline-only complete instead of being salvaged as a stub plan. Reads done through `run_shell` count as evidence too — `src/subagent/shell-evidence.ts` classifies shell reads (`cat`, `grep`, `sed` without `-i`, …) over the same subject expansion the auto-shell policy uses — but there is no corresponding shell-write evidence or file-write requirement: a run that never touches a file still completes normally once it replies with the envelope. There is no turn budget. Operator/parent cancel after any progress returns a **cancelled** salvage report (partial findings + tool activity) instead of a bare cancel string; cancel before progress still surfaces as cancelled-by-operator. There is no repetition/no-progress/never-acted/never-edited hard stop and no fingerprint-based re-dispatch block — a genuinely stuck worker runs until it completes, stalls, hits an opt-in wall-clock deadline, or is cancelled. - `spawn_agent` starts each worker and records it in the caller's fleet mailbox. On the TUI primary, mailbox mail is the collect path: occupancy takes uncollected terminals and re-enters the parent as system inbound. Nested orchestrators collect through mailbox mail the same way. A mounted `wait_agents` (exec primary) collects worker reports directly instead. Already-collected waits return status without a second report or error body. Wait JSON includes `stop_reason` from the session when present so a salvage that is wait-`done` is not mistaken for a clean complete, and so parent-initiated interrupt (`interrupted`) is not mistaken for operator-cancel (`cancelled`). Deadline salvage prepends an advisory parent hint suggesting continuation plus a longer deadline if more wall-clock time is warranted. Failed and incomplete-report salvage tell the parent to diagnose from the report or error and MAY spawn one successor with a changed brief. A parent-initiated interrupt is a resumable pause: wait unblocks with `stop_reason: interrupted` (often while the session is still running and has no report); the parent should `resume_agent` or re-wait, and must not spawn a successor against a still-live worker. Successor only if that session is no longer resumable. Operator-cancelled salvage asks the parent to synthesize Findings and Paths and wait for the operator instead of auto-starting another specialist. Identical re-dispatch of the same brief stays refused at the prompt / spawn-handoff layer; there is no fingerprint-based re-dispatch hard-block. Deadline hints are advisory only — an identical re-dispatch is still admitted at runtime. Parent hints are prepended on salvage reports returned to the parent. The runtime does not auto-spawn successors. + `spawn_agent` starts each worker and records it in the caller's fleet mailbox. On the TUI primary, mailbox mail is the collect path: occupancy takes uncollected terminals and re-enters the parent as system inbound. Nested orchestrators collect through mailbox mail the same way. A mounted `wait_agents` (exec primary) collects worker reports directly instead. Already-collected waits return status without a second report or error body. Wait JSON includes `stop_reason` from the session when present so a salvage that is wait-`done` is not mistaken for a clean complete, and so parent-initiated interrupt (`interrupted`) is not mistaken for operator-cancel (`cancelled`). Deadline salvage prepends an advisory parent hint suggesting continuation plus a longer deadline if more wall-clock time is warranted. Failed and incomplete-report salvage tell the parent to diagnose from the report or error and MAY spawn one successor with a changed brief. A parent-initiated interrupt is a resumable pause: wait unblocks with `stop_reason: interrupted` (often while the session is still running and has no report); the parent should `resume_agent` or re-wait, and must not spawn a successor against a still-live worker. Successor only if that session is no longer resumable. Operator-cancelled salvage asks the parent to synthesize Findings and Paths and wait for the operator instead of auto-starting another specialist. Identical re-dispatch of the same brief stays refused at the prompt / spawn-handoff layer, except a recoverable/continuable child failure (`continuable: true`) MAY spawn one successor with the same brief; there is no fingerprint-based re-dispatch hard-block. Deadline hints are advisory only — an identical re-dispatch is still admitted at runtime. Parent hints are prepended on salvage reports returned to the parent. The runtime does not auto-spawn successors. #### Model-family policy (`src/agent/model-family-policy.ts`) diff --git a/src/agent/prompts.test.ts b/src/agent/prompts.test.ts index 3f311556f..c8c6cb6d7 100644 --- a/src/agent/prompts.test.ts +++ b/src/agent/prompts.test.ts @@ -223,7 +223,7 @@ Scope and conventions: Orchestration: - One focused task per spawned worker. Fan-out width follows independent lanes (one lane per PR/path/ownership). Break multi-step or parallel work into those dispatches with distinct lenses; prefer \`spawn_agent\` (fire several in one turn when jobs are independent), then reply with who is running and end the turn — workers keep running while you are idle. Mailbox mail arrives as inbound when a worker finishes; read it and do not poll. \`list_agents\` shows the fleet without blocking; after a parked ask is surfaced, answer with \`send_input\` and do not poll \`list_agents\`. - Pass the typed spawn contract and keep it tight: \`intent\`, \`success_criteria\` (done-when; required for implement/review and their default directors), \`do_not\` (scope fence), and \`report_focus\`. Free-form \`prompt\` without \`success_criteria\` fail-closes for implement/review and their default directors. -- After workers return, classify fail / incomplete-report vs parent-initiated interrupt vs operator-cancel vs clean complete. Fail-path (\`status: failed\` or salvage \`incomplete-report\`): diagnose from the report or error and MAY spawn one successor with a changed brief. Parent-initiated interrupt (\`interrupt_agent\` / \`send_input\` with \`interrupt:true\` unblocks wait with \`stop_reason: interrupted\`): the worker is often still running and often has no report — \`resume_agent\`, or idle for its mailbox mail; do not \`spawn_agent\` a successor against a still-live worker. Successor only if that session is no longer resumable. Operator-cancel (\`stop_reason\` cancelled): wait for the operator; do not auto-retry. Identical brief: refuse. Merge Summary/Findings into a coherent answer for the operator; do not paste raw fleet-agent dumps. +- After workers return, classify fail / incomplete-report vs parent-initiated interrupt vs operator-cancel vs clean complete. Fail-path (\`status: failed\` or salvage \`incomplete-report\`): diagnose from the report or error and MAY spawn one successor with a changed brief. Parent-initiated interrupt (\`interrupt_agent\` / \`send_input\` with \`interrupt:true\` unblocks wait with \`stop_reason: interrupted\`): the worker is often still running and often has no report — \`resume_agent\`, or idle for its mailbox mail; do not \`spawn_agent\` a successor against a still-live worker. Successor only if that session is no longer resumable. Operator-cancel (\`stop_reason\` cancelled): wait for the operator; do not auto-retry. Identical brief: refuse. Recoverable/continuable child failure (\`continuable: true\`) MAY spawn one successor with the same brief; identical brief is still refused otherwise. Merge Summary/Findings into a coherent answer for the operator; do not paste raw fleet-agent dumps. - Use manage_tasks for your own coordination checklist; spawning workers is \`spawn_agent\`, not manage_tasks. - If context is compacted automatically, do not stop tasks early due to token fear; persist progress via manage_tasks and worker reports.`); }); diff --git a/src/agent/prompts.ts b/src/agent/prompts.ts index 460b54bc9..2b5637708 100644 --- a/src/agent/prompts.ts +++ b/src/agent/prompts.ts @@ -217,7 +217,7 @@ const GUIDELINE_SUB_BLOCKS: Record< (ctx.waitAgentsMounted ? " or re-wait" : ", or idle for its mailbox mail") + - "; do not `spawn_agent` a successor against a still-live worker. Successor only if that session is no longer resumable. Operator-cancel (`stop_reason` cancelled): wait for the operator; do not auto-retry. Identical brief: refuse. Merge Summary/Findings into a coherent answer for the operator; do not paste raw fleet-agent dumps.", + "; do not `spawn_agent` a successor against a still-live worker. Successor only if that session is no longer resumable. Operator-cancel (`stop_reason` cancelled): wait for the operator; do not auto-retry. Identical brief: refuse. Recoverable/continuable child failure (`continuable: true`) MAY spawn one successor with the same brief; identical brief is still refused otherwise. Merge Summary/Findings into a coherent answer for the operator; do not paste raw fleet-agent dumps.", "- Use manage_tasks for your own coordination checklist; spawning workers is `spawn_agent`, not manage_tasks.", "- If context is compacted automatically, do not stop tasks early due to token fear; persist progress via manage_tasks and worker reports.", ]; diff --git a/src/inference-error-message.ts b/src/inference-error-message.ts index 0cd6126cd..83206c65d 100644 --- a/src/inference-error-message.ts +++ b/src/inference-error-message.ts @@ -68,6 +68,31 @@ export function classifyInferenceErrorCategory( : error.category; } +/** + * Whether a normalized provider-failure category is transient enough that a + * parent may spawn one successor with the same brief (CL-8978). Allowlist: + * retryable/timeout — including 429 overload, which normalizes to retryable. + * Fatal categories win explicitly: credential, quota, and context-overflow + * failures must never read as continuable. + */ +const FATAL_PROVIDER_FAILURE_CATEGORIES: ReadonlySet = new Set([ + "credential_failure", + "quota_exhausted", + "context_overflow", +]); + +const RECOVERABLE_PROVIDER_FAILURE_CATEGORIES: ReadonlySet = new Set([ + "retryable", + "timeout", +]); + +export function isRecoverableProviderFailureCategory( + category: string, +): boolean { + if (FATAL_PROVIDER_FAILURE_CATEGORIES.has(category)) return false; + return RECOVERABLE_PROVIDER_FAILURE_CATEGORIES.has(category); +} + function codexUsageLimitLine(error: InferenceErrorLike): string | undefined { // Match normalizeCodexUsageLimitError: never brand a known non-Codex source. if ( diff --git a/src/prompts.test.ts b/src/prompts.test.ts index 8a2ca78d1..16d32fe8b 100644 --- a/src/prompts.test.ts +++ b/src/prompts.test.ts @@ -194,6 +194,9 @@ test("primary chat prompt classifies fail-path successor vs interrupt resume vs expect(guidelines).toContain("wait for the operator"); expect(guidelines).toContain("do not auto-retry"); expect(guidelines).toContain("Identical brief: refuse"); + expect(guidelines).toContain("continuable"); + expect(guidelines).toContain("MAY spawn one successor with the same brief"); + expect(guidelines).toContain("identical brief is still refused otherwise"); expect(guidelines).toContain("resume_agent"); expect(guidelines).toContain("still-live worker"); expect(guidelines).not.toContain("interrupted-incomplete"); diff --git a/src/subagent/agent-fleet.ts b/src/subagent/agent-fleet.ts index cdcc2fa9d..faca14e67 100644 --- a/src/subagent/agent-fleet.ts +++ b/src/subagent/agent-fleet.ts @@ -107,7 +107,10 @@ import { } from "./authority.js"; import { formatSubAgentSpawnAuthFailureMessage } from "./inference-auth-failure.js"; -import { isResolvedProviderFailureError } from "../inference-error-message.js"; +import { + isRecoverableProviderFailureCategory, + isResolvedProviderFailureError, +} from "../inference-error-message.js"; import { errorMessage } from "../agent/error-message.js"; import { isSubAgentCancelError } from "./dispose.js"; import { @@ -125,6 +128,8 @@ interface FleetRecord { error?: string; stopReason?: string; providerFailure?: true; + /** CL-8978: transient provider failure — the parent may spawn one successor. */ + recoverableFailure?: true; /** Set once a wait_agents caller has been handed this result. */ collected?: boolean; /** Set once a waiter or occupancy take handed report/error. */ @@ -158,6 +163,8 @@ interface FleetOverlay { tombstoned?: boolean; hint?: string; providerFailure?: true; + /** CL-8978: transient provider failure — the parent may spawn one successor. */ + recoverableFailure?: true; } const RECOVERY_HINT = @@ -265,6 +272,17 @@ class FleetMailbox { existing.providerFailure = true; } + /** + * CL-8978: stamp a transient (retryable/timeout/overload) provider failure + * alongside sessions.fail. Survives session eviction like providerFailure — + * snapshot projects it even once the payload is tombstoned. + */ + markRecoverable(id: string): void { + const existing = this.records.get(id); + if (existing === undefined) return; + existing.recoverableFailure = true; + } + markQueued(id: string): void { const existing = this.records.get(id); if (existing === undefined || existing.collected === true) return; @@ -462,6 +480,9 @@ class FleetMailbox { : {}), ...(stopReason !== undefined ? { stopReason } : {}), ...(overlay.providerFailure === true ? { providerFailure: true } : {}), + ...(overlay.recoverableFailure === true + ? { recoverableFailure: true } + : {}), ...(ask !== undefined ? { question: ask.question, questionId: ask.questionId } : {}), @@ -600,7 +621,10 @@ export const waitAgentsToolDefinition: ToolDefinition = { `slot), "running", and "awaiting_director". interrupt_agent unblocks this wait immediately with ` + `status "interrupted" (a parent-initiated pause — resume_agent, do not spawn_agent a successor against the still-live worker). ` + `close_agent also unblocks with status "interrupted" but is permanent. Terminal JSON includes stop_reason when the session recorded one ` + - `(interrupted, cancelled, incomplete-report, and similar). awaiting_director is not terminal: re-wait while still pending re-delivers the same question. ` + + `(interrupted, cancelled, incomplete-report, and similar). A "failed" entry with "continuable": true is a recoverable transient ` + + `provider failure (retryable/timeout/overload) — terminal, not a timeout and not a stall: do not re-wait it, and you may spawn at most ` + + `one successor with the same brief. "failed" without the marker (auth, quota, context-overflow, or other errors) is not continuable — ` + + `do not respawn it. awaiting_director is not terminal: re-wait while still pending re-delivers the same question. ` + `Answer with send_input (soft). Do not call this in a tight zero-progress loop: a timeout means the targets are still ` + `queued, running, or awaiting a director answer, not "try again right away" — do other work, reply to the operator, or change the brief. Calling again with the ` + `same targets is a real timed wait, not a spin, but wastes turns if nothing has changed. ` + @@ -1530,6 +1554,17 @@ export function createSpawnAgentTool(deps: AgentFleetDeps): AgentTool { if (isProviderFailure || providerFailureObserved) { deps.fleetRecords.markProviderFailure(session.id); } + // CL-8978: a classified transient failure stays wait-terminal + // failed, but carries a continuable marker so the parent can + // spawn one successor instead of stalling on the failure. + // Fatal categories (credential/quota/context-overflow) and + // unclassified throws never mark — no auto-retry is added here. + if ( + isResolvedProviderFailureError(err) && + isRecoverableProviderFailureCategory(err.category) + ) { + deps.fleetRecords.markRecoverable(session.id); + } deps.sessions.fail(session.id, failReason); }) .finally(() => { diff --git a/src/subagent/fleet-dry-drive.ts b/src/subagent/fleet-dry-drive.ts index 65b3a67ac..06c3ef860 100644 --- a/src/subagent/fleet-dry-drive.ts +++ b/src/subagent/fleet-dry-drive.ts @@ -34,6 +34,8 @@ export interface FleetDryMailboxRecord { readonly description?: string; readonly hint?: string; readonly providerFailure?: true; + /** CL-8978: transient provider failure — the parent may spawn one successor. */ + readonly recoverableFailure?: true; readonly stopReason?: string; } @@ -58,9 +60,25 @@ export interface CollectedWorkerReport { error?: string; hint?: string; provider_failure?: true; + /** + * CL-8978: failed entries from a transient provider failure carry this + * marker plus single-successor guidance in continue_with. Capped affordance: + * at most one respawn with the same brief, never a retry loop. + */ + continuable?: true; + continue_with?: string; stop_reason?: string; } +/** + * Single-successor guidance for a failed+continuable entry. The marker is + * advisory only — no runtime auto-retry backs it. + */ +export const RECOVERABLE_FAILURE_CONTINUE_GUIDANCE = + "This worker failed with a transient provider error (retryable/timeout/overload) " + + "and is terminal — do not re-wait it. You may spawn at most one successor with " + + "the same brief; do not retry in a loop."; + export function shouldDriveOpenTasks(input: { previousRunning?: number | undefined; running?: number | undefined; @@ -169,6 +187,12 @@ export function projectMailboxRecord( ...(error !== undefined ? { error } : {}), ...(taken.hint !== undefined ? { hint: taken.hint } : {}), ...(taken.providerFailure === true ? { provider_failure: true } : {}), + ...(taken.status === "failed" && taken.recoverableFailure === true + ? { + continuable: true as const, + continue_with: RECOVERABLE_FAILURE_CONTINUE_GUIDANCE, + } + : {}), ...(taken.stopReason !== undefined ? { stop_reason: taken.stopReason } : {}), diff --git a/src/subagent/run-recoverable-failure.test.ts b/src/subagent/run-recoverable-failure.test.ts new file mode 100644 index 000000000..43cd9e699 --- /dev/null +++ b/src/subagent/run-recoverable-failure.test.ts @@ -0,0 +1,314 @@ +import { describe, expect, test } from "bun:test"; +import { + createFleetMailbox, + createSpawnAgentTool, + createWaitAgentsTool, + waitAgentsToolDefinition, + type AgentFleetDeps, +} from "./agent-fleet.js"; +import { unlimitedAdmissionQueue } from "./admission.js"; +import { isLiveWaitStatus } from "./lifecycle.js"; +import { + driveMailboxMail, + occupancyShouldYieldWait, +} from "./mailbox-mail-drive.js"; +import { createResolvedProviderFailureError } from "../inference-error-message.js"; +import { createPermissionGate } from "../permission/gate.js"; +import { createSubAgentSessionStore } from "./session-store.js"; +import type { RunSubAgentParams, RunSubAgentResult } from "./types.js"; + +const testPermissionGate = createPermissionGate({ + approvals: [], + interactive: false, + skipPermissions: true, + reactorGated: false, +}); + +const provider = { + providerName: "test-provider", + baseURL: "http://localhost", + model: "test-model", +}; + +function makeDeps( + run: (params: RunSubAgentParams) => Promise, +): AgentFleetDeps { + const sessions = createSubAgentSessionStore(); + return { + permissionGate: testPermissionGate, + cwd: "/tmp", + getWorkdirBase: () => "/tmp/workdir", + provider, + run, + sessions, + fleetRecords: createFleetMailbox(sessions), + admission: unlimitedAdmissionQueue(), + }; +} + +async function callToolRaw( + tool: + | ReturnType + | ReturnType, + args: Record, +): Promise<{ content: string; isError?: boolean }> { + if (tool.kind !== "full") + throw new Error(`expected full tool, got ${tool.kind}`); + const result = await tool.handler( + { + id: `call-${Math.random()}`, + name: tool.definition.name, + arguments: args, + }, + new AbortController().signal, + ); + const content = + typeof result.content === "string" + ? result.content + : JSON.stringify(result.content); + return { + content, + ...(result.isError !== undefined ? { isError: result.isError } : {}), + }; +} + +async function callTool( + tool: + | ReturnType + | ReturnType, + args: Record, +): Promise> { + const { content } = await callToolRaw(tool, args); + return JSON.parse(content) as Record; +} + +function retryableAfterToolsFailure(): Error { + // runSubAgentInner throws without an outer retry once any tool already ran, + // so a retryable provider fault after tool use lands in the agent-fleet + // catch as a ResolvedProviderFailureError with category "retryable". + return createResolvedProviderFailureError("test-provider", { + category: "retryable", + message: "upstream overloaded, retry later", + statusCode: 429, + }); +} + +function waitUntilMailboxTerminal( + mailbox: ReturnType, + sessions: ReturnType, + id: string, +): Promise { + return new Promise((resolve) => { + const done = (): boolean => { + const snap = mailbox.peek(id); + return snap !== undefined && !isLiveWaitStatus(snap.status); + }; + if (done()) { + resolve(); + return; + } + const unsub = sessions.subscribe(() => { + if (done()) { + unsub(); + resolve(); + } + }); + if (done()) { + unsub(); + resolve(); + } + }); +} + +describe("CL-8978 recoverable subagent failure", () => { + test("retryable-after-tools failure is wait-terminal failed with a continuable marker, and the parent can spawn/wait a successor", async () => { + const deps = makeDeps(async () => { + throw retryableAfterToolsFailure(); + }); + const spawn = createSpawnAgentTool(deps); + const wait = createWaitAgentsTool({ + sessions: deps.sessions, + fleetRecords: deps.fleetRecords, + }); + + const spawned = await callTool(spawn, { + description: "flaky job", + prompt: "do it", + intent: "explore", + }); + const id = spawned.agent_id as string; + expect(typeof id).toBe("string"); + + const waited = await callTool(wait, { + targets: [id], + timeout_ms: 5000, + mode: "all", + }); + expect(waited.timed_out).toBe(false); + const results = waited.results as Record[]; + expect(results).toHaveLength(1); + expect(results[0]?.status).toBe("failed"); + expect(typeof results[0]?.error).toBe("string"); + // Machine-readable continuable marker plus single-successor guidance. + expect(results[0]?.continuable).toBe(true); + expect(typeof results[0]?.continue_with).toBe("string"); + + // The failure is terminal, never stuck running. + const snap = deps.fleetRecords.peek(id); + expect(snap?.status).toBe("failed"); + expect(isLiveWaitStatus(snap?.status ?? "running")).toBe(false); + + // The parent handle still works: spawn and wait a successor. + const deps2 = makeDeps(async () => ({ report: "successor done" })); + // Share the fleet so the successor is a true sibling lane. + const spawn2 = createSpawnAgentTool({ + ...deps2, + sessions: deps.sessions, + fleetRecords: deps.fleetRecords, + }); + const wait2 = createWaitAgentsTool({ + sessions: deps.sessions, + fleetRecords: deps.fleetRecords, + }); + const spawned2 = await callTool(spawn2, { + description: "successor job", + prompt: "do it again", + intent: "explore", + }); + const id2 = spawned2.agent_id as string; + expect(id2).not.toBe(id); + const waited2 = await callTool(wait2, { + targets: [id2], + timeout_ms: 5000, + mode: "all", + }); + expect(waited2.timed_out).toBe(false); + const results2 = waited2.results as Record[]; + expect(results2[0]?.status).toBe("done"); + }); + + test("credential failure stays failed without a continuable marker", async () => { + const deps = makeDeps(async () => { + throw createResolvedProviderFailureError("test-provider", { + category: "credential_failure", + message: "Authentication failed", + statusCode: 401, + }); + }); + const spawn = createSpawnAgentTool(deps); + const wait = createWaitAgentsTool({ + sessions: deps.sessions, + fleetRecords: deps.fleetRecords, + }); + + const spawned = await callTool(spawn, { + description: "auth job", + prompt: "do it", + intent: "explore", + }); + const waited = await callTool(wait, { + targets: [spawned.agent_id as string], + timeout_ms: 5000, + mode: "all", + }); + const results = waited.results as Record[]; + expect(results[0]?.status).toBe("failed"); + expect(results[0]?.continuable).toBeUndefined(); + }); + + test("failed+recoverable lane is delivered as mailbox mail, and a failed send re-arms instead of dropping the terminal", async () => { + const deps = makeDeps(async () => { + throw retryableAfterToolsFailure(); + }); + const spawn = createSpawnAgentTool(deps); + const spawned = await callTool(spawn, { + description: "flaky job", + prompt: "do it", + intent: "explore", + }); + const id = spawned.agent_id as string; + await waitUntilMailboxTerminal(deps.fleetRecords, deps.sessions, id); + + // Terminal failure is occupancy-yielding, so the parent is driven back + // into a turn instead of sitting silent for the stall bound. + expect(occupancyShouldYieldWait(deps.fleetRecords)).toBe(true); + + const prompts: string[] = []; + let sendShouldFail = true; + const drive = (): boolean | Promise => + driveMailboxMail({ + parentProcessing: false, + mailbox: deps.fleetRecords, + lanes: deps.sessions.list(), + beginSystemContinuation: () => undefined, + send: (prompt: string) => { + prompts.push(prompt); + if (sendShouldFail) throw new Error("send down"); + return { status: "accepted" }; + }, + }); + + // Occupancy send failure leaves the terminal uncollected ... + expect(await drive()).toBe(false); + expect(prompts).toHaveLength(1); + expect(prompts[0]).toContain(id); + expect(deps.fleetRecords.peek(id)?.collected).not.toBe(true); + + // ... and the re-flush delivers it, starting the parent turn. + sendShouldFail = false; + expect(await drive()).toBe(true); + expect(prompts).toHaveLength(2); + expect(deps.fleetRecords.peek(id)?.collected).toBe(true); + // The mailbox path carries the same continuable marker as wait_agents. + expect(prompts[1]).toContain('"continuable":true'); + }); + + test("in-flight work stays live until settled: a timeout is liveness, not failure", async () => { + let resolveRun!: (v: RunSubAgentResult) => void; + const gate = new Promise((res) => { + resolveRun = res; + }); + const deps = makeDeps(() => gate); + const spawn = createSpawnAgentTool(deps); + const wait = createWaitAgentsTool({ + sessions: deps.sessions, + fleetRecords: deps.fleetRecords, + }); + + const spawned = await callTool(spawn, { + description: "slow job", + prompt: "do it", + intent: "explore", + }); + const id = spawned.agent_id as string; + + // Unsettled work (including an in-flight provider retry) projects live. + const snap = deps.fleetRecords.peek(id); + expect(snap?.status).toBe("running"); + expect(isLiveWaitStatus(snap?.status ?? "failed")).toBe(true); + + const waited = await callTool(wait, { + targets: [id], + timeout_ms: 50, + mode: "all", + }); + expect(waited.timed_out).toBe(true); + const results = waited.results as Record[]; + expect(results[0]?.status).toBe("running"); + expect(results[0]?.continuable).toBeUndefined(); + + resolveRun({ report: "slow done" }); + const waited2 = await callTool(wait, { + targets: [id], + timeout_ms: 5000, + mode: "all", + }); + expect(waited2.timed_out).toBe(false); + const results2 = waited2.results as Record[]; + expect(results2[0]?.status).toBe("done"); + }); + + test("wait_agents documents the continuable failed marker", () => { + expect(waitAgentsToolDefinition.description).toContain("continuable"); + }); +});