Repository navigation
fix(core): don't strand a run when the max-deliveries terminal write is throttled (v4) #4422
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,5 @@ | ||
| --- | ||
| '@workflow/core': patch | ||
| --- | ||
|
|
||
| Retry the max-deliveries `run_failed`/`step_failed` write and the step handler's workflow re-queue through queue redelivery when they fail transiently (429, 5xx, transport) instead of acking and leaving the run stuck `running`. |
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -17,6 +17,7 @@ import { | |
| type Step, | ||
| StepInvokePayloadSchema, | ||
| } from '@workflow/world'; | ||
| import { isRetryableWorldError } from '../classify-error.js'; | ||
| import { importKey } from '../encryption.js'; | ||
| import { runtimeLogger, stepLogger } from '../logger.js'; | ||
| import { getStepFunction } from '../private.js'; | ||
|
|
@@ -90,8 +91,9 @@ const stepHandler = createQueueHandler( | |
| // This prevents runaway steps from consuming infinite queue deliveries. | ||
| // At this point, we want to do the minimal amount of work (no fetching | ||
| // of the step details, etc. We simply attempt to mark the step as failed | ||
| // and enqueue the workflow once, and if either of those fails, the message | ||
| // is still consumed but with adequate logging that an error occurred. | ||
| // and enqueue the workflow once. A transient failure of either throws so | ||
| // the queue retries it; a definitive step_failed rejection consumes the | ||
| // message with adequate logging that an error occurred. | ||
| if (metadata.attempt > MAX_QUEUE_DELIVERIES) { | ||
| runtimeLogger.error( | ||
| `Step handler exceeded max deliveries (${metadata.attempt}/${MAX_QUEUE_DELIVERIES})`, | ||
|
|
@@ -102,8 +104,8 @@ const stepHandler = createQueueHandler( | |
| attempt: metadata.attempt, | ||
| } | ||
| ); | ||
| const world = getWorld(); | ||
| try { | ||
| const world = getWorld(); | ||
| await world.events.create( | ||
| workflowRunId, | ||
| { | ||
|
|
@@ -117,36 +119,62 @@ const stepHandler = createQueueHandler( | |
| }, | ||
| { requestId } | ||
| ); | ||
| // Re-queue the workflow to handle the failed step | ||
| await queueMessage( | ||
| world, | ||
| getWorkflowQueueName(workflowName, stepNamespace), | ||
| { | ||
| runId: workflowRunId, | ||
| traceCarrier: await serializeTraceCarrier(), | ||
| requestedAt: new Date(), | ||
| } | ||
| ); | ||
| } catch (err) { | ||
| if (EntityConflictError.is(err) || RunExpiredError.is(err)) { | ||
| if (RunExpiredError.is(err)) { | ||
| return; | ||
| } | ||
| // Can't even mark the step as failed. Consume the message to stop | ||
| // further retries. The run will remain in its current state. | ||
| runtimeLogger.error( | ||
| `Failed to mark step as failed after ${metadata.attempt} delivery attempts. ` + | ||
| `A persistent error is preventing the step from being terminated. ` + | ||
| `The run will remain in its current state until manually resolved. ` + | ||
| `This is most likely due to a persistent outage of the workflow backend ` + | ||
| `or a bug in the workflow runtime and should be reported to the Workflow team.`, | ||
| { | ||
| workflowRunId, | ||
| stepId, | ||
| attempt: metadata.attempt, | ||
| error: err instanceof Error ? err.message : String(err), | ||
| // EntityConflictError: the step is already terminal. That may be this | ||
| // message's own earlier delivery, which wrote step_failed and then | ||
| // failed to re-queue the workflow below, so fall through and re-queue | ||
| // it. An extra wake only costs one replay. | ||
| if (!EntityConflictError.is(err)) { | ||
| // A transient backend failure (429 / 5xx / transport) must not | ||
| // abandon the run: acking here leaves it `running` with no message | ||
| // left to drive it. Throw so the queue redelivers. The redelivery | ||
| // is still past the ceiling, so it only retries this write and | ||
| // never re-runs the step body. | ||
| if (isRetryableWorldError(err)) { | ||
| runtimeLogger.warn( | ||
| 'Transient error marking step as failed after max deliveries, retrying via queue redelivery', | ||
| { | ||
| workflowRunId, | ||
| stepId, | ||
| attempt: metadata.attempt, | ||
| error: err instanceof Error ? err.message : String(err), | ||
| } | ||
| ); | ||
| throw err; | ||
| } | ||
| ); | ||
| // Can't even mark the step as failed. Consume the message to stop | ||
| // further retries. The run will remain in its current state. | ||
| runtimeLogger.error( | ||
| `Failed to mark step as failed after ${metadata.attempt} delivery attempts. ` + | ||
| `A persistent error is preventing the step from being terminated. ` + | ||
| `The run will remain in its current state until manually resolved. ` + | ||
| `This is most likely due to a persistent outage of the workflow backend ` + | ||
| `or a bug in the workflow runtime and should be reported to the Workflow team.`, | ||
| { | ||
| workflowRunId, | ||
| stepId, | ||
| attempt: metadata.attempt, | ||
| error: err instanceof Error ? err.message : String(err), | ||
| } | ||
| ); | ||
| return; | ||
| } | ||
| } | ||
| // Re-queue the workflow to handle the failed step. A failure here | ||
| // throws so the queue redelivers: acking would leave the step failed | ||
| // with no message left to wake the workflow. | ||
| await queueMessage( | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. This is the one spot that throws on any error rather than checking |
||
| world, | ||
| getWorkflowQueueName(workflowName, stepNamespace), | ||
| { | ||
| runId: workflowRunId, | ||
| traceCarrier: await serializeTraceCarrier(), | ||
| requestedAt: new Date(), | ||
| } | ||
| ); | ||
| return; | ||
| } | ||
|
|
||
|
|
||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
Same caveat as #4421, for anyone reading this later: this relies on the queue redelivering after delivery 49. VQS (24 h TTL, no delivery cap) and world-local (256) do. On this branch, world-postgres jobs cap at 3 attempts, so they never reach the ceiling.