From e8f1c1bb078c58a28ab259ca3a2f4db985a0e07d Mon Sep 17 00:00:00 2001 From: Manas Srivastava Date: Thu, 4 Jun 2026 20:48:59 +0530 Subject: [PATCH] fix(deploy): reap stuck 'building' deploys with no provider_id (sweep #5) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A deployments row stuck at status='building' with an EMPTY provider_id is never reaped: the deploy_status_reconcile sweep skips empty-provider_id rows unconditionally (correct for a build in flight), so a deploy whose api goroutine DIED before UpdateDeploymentProviderID (crash mid-runDeploy) keeps the row 'building' forever and permanently consumes the team's deployments_apps tier cap. Fix: in the empty-provider_id branch, if status='building' AND the row is older than a 15m grace window, reap it to 'failed' via a guarded UPDATE (double-guarded on status='building' AND empty provider_id, so a row that raced and acquired a provider_id is a no-op). This frees the tier cap; the idempotent deploy_failure_autopsy job then emits the deploy.failed audit -> failure email. Fresh (<15m) builds are still left alone — 15m is well beyond the normal ~30-90s build. - add createdAt to activeDeployment + created_at to listActiveDeployments SELECT - add stuckBuildingGrace (15m) + stuckBuildingReapMessage consts - add guarded reapStuckBuilding helper (does NOT reuse updateStatus, which doesn't gate on provider_id) - log jobs.deploy_status_reconcile.stuck_building_reaped + reaped counter Coverage block: Symptom: building deploy w/ empty provider_id never reaped; tier cap leak Enumeration: rg 'if d.providerID == ""' / grep listActiveDeployments SELECT callers Sites found: 1 (single sweep loop + its SELECT/scan) Sites touched: 1 Coverage test: TestDeployStatusReconciler_Work_StuckBuildingReaped (+ Fresh/ ReapFailed/StuckDeploying/RowWithProviderIDUnaffected) Live verified: awaiting post-merge auto-deploy + worker /healthz SHA gate (rule 14) 100% patch coverage (Work/listActiveDeployments/reapStuckBuilding all 100.0%); jobs package 97.2% (>=95%). make gate green. Co-Authored-By: Claude Opus 4.8 (1M context) --- .../jobs/deploy_lifecycle_coverage_test.go | 204 +++++++++++++++--- internal/jobs/deploy_status_reconcile.go | 78 ++++++- ...deploy_status_reconcile_job_failed_test.go | 17 +- 3 files changed, 263 insertions(+), 36 deletions(-) diff --git a/internal/jobs/deploy_lifecycle_coverage_test.go b/internal/jobs/deploy_lifecycle_coverage_test.go index 7c68b44..3729155 100644 --- a/internal/jobs/deploy_lifecycle_coverage_test.go +++ b/internal/jobs/deploy_lifecycle_coverage_test.go @@ -237,10 +237,10 @@ func TestDeployNamespaceFromProviderID(t *testing.T) { }{ {"app-abc", "instant-deploy-abc"}, {"app-1234", "instant-deploy-1234"}, - {"app-", ""}, // empty appID after prefix - {"instant-stack-xyz", ""}, // foreign prefix - {"", ""}, // nothing - {"appabc", ""}, // missing hyphen + {"app-", ""}, // empty appID after prefix + {"instant-stack-xyz", ""}, // foreign prefix + {"", ""}, // nothing + {"appabc", ""}, // missing hyphen } for _, tc := range cases { if got := deployNamespaceFromProviderID(tc.in); got != tc.want { @@ -297,7 +297,7 @@ func TestDeployStatusReconciler_Work_NoActiveRows(t *testing.T) { defer db.Close() mock.ExpectQuery(`FROM deployments\s+WHERE status IN`). - WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status"})) + WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status", "created_at"})) w := NewDeployStatusReconciler(db, newFakeDeployStatusK8s()) if err := w.Work(context.Background(), fakeRiverJob[DeployStatusReconcileArgs]()); err != nil { @@ -347,14 +347,18 @@ func TestDeployStatusReconciler_Work_FullSweep(t *testing.T) { idHealthy := uuid.New() idFailed := uuid.New() + // idBlank gets a FRESH created_at so the empty-provider_id row stays + // "skipped" (a build still in flight), not reaped by the stuck-building + // path — that path is exercised in its own dedicated tests. + now := time.Now().UTC() mock.ExpectQuery(`FROM deployments\s+WHERE status IN`). - WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status"}). - AddRow(idBlank, "", "building"). - AddRow(idForeign, "instant-stack-zzz", "building"). - AddRow(idStopped, "app-stopped", "building"). - AddRow(idSame, "app-same", "building"). - AddRow(idHealthy, "app-healthy", "building"). - AddRow(idFailed, "app-failed", "deploying")) + WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status", "created_at"}). + AddRow(idBlank, "", "building", now). + AddRow(idForeign, "instant-stack-zzz", "building", now). + AddRow(idStopped, "app-stopped", "building", now). + AddRow(idSame, "app-same", "building", now). + AddRow(idHealthy, "app-healthy", "building", now). + AddRow(idFailed, "app-failed", "deploying", now)) k8s := newFakeDeployStatusK8s() // app-same: deployment with all-zero status → building @@ -413,12 +417,12 @@ func TestDeployStatusReconciler_Work_AutopsyCapDeferred(t *testing.T) { const n = maxAutopsiesPerTick + 1 // one beyond the cap → at least 1 deferred k8s := newFakeDeployStatusK8s() - rows := sqlmock.NewRows([]string{"id", "provider_id", "status"}) + rows := sqlmock.NewRows([]string{"id", "provider_id", "status", "created_at"}) ids := make([]uuid.UUID, n) for i := 0; i < n; i++ { ids[i] = uuid.New() appID := "appfail" + uuid.NewString()[:8] - rows.AddRow(ids[i], "app-"+appID, "deploying") + rows.AddRow(ids[i], "app-"+appID, "deploying", time.Now().UTC()) k8s.objs["instant-deploy-"+appID+"|app-"+appID] = &appsv1.Deployment{ Status: appsv1.DeploymentStatus{ Conditions: []appsv1.DeploymentCondition{{ @@ -481,8 +485,8 @@ func TestDeployStatusReconciler_Work_K8sGetFailed(t *testing.T) { idA := uuid.New() mock.ExpectQuery(`FROM deployments\s+WHERE status IN`). - WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status"}). - AddRow(idA, "app-broken", "building")) + WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status", "created_at"}). + AddRow(idA, "app-broken", "building", time.Now().UTC())) k8s := newFakeDeployStatusK8s() k8s.errOn["instant-deploy-broken|app-broken"] = errors.New("network blip") @@ -507,8 +511,8 @@ func TestDeployStatusReconciler_Work_UpdateFailed(t *testing.T) { id := uuid.New() mock.ExpectQuery(`FROM deployments\s+WHERE status IN`). - WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status"}). - AddRow(id, "app-borked", "building")) + WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status", "created_at"}). + AddRow(id, "app-borked", "building", time.Now().UTC())) k8s := newFakeDeployStatusK8s() k8s.objs["instant-deploy-borked|app-borked"] = &appsv1.Deployment{ @@ -539,8 +543,8 @@ func TestDeployStatusReconciler_listActiveDeployments_ScanError(t *testing.T) { // Return a row whose id is not a UUID → Scan fails. mock.ExpectQuery(`FROM deployments\s+WHERE status IN`). - WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status"}). - AddRow("not-a-uuid", "app-x", "building")) + WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status", "created_at"}). + AddRow("not-a-uuid", "app-x", "building", time.Now().UTC())) w := NewDeployStatusReconciler(db, newFakeDeployStatusK8s()) if werr := w.Work(context.Background(), fakeRiverJob[DeployStatusReconcileArgs]()); werr == nil { @@ -559,6 +563,154 @@ func TestComputeNewStatus_ForeignProviderID(t *testing.T) { } } +// ─── stuck-building reaper (sweep finding #5) ───────────────────────────────── + +// TestDeployStatusReconciler_Work_StuckBuildingReaped pins the core finding-#5 +// fix: a deployments row stuck at status="building" with an EMPTY provider_id +// for longer than stuckBuildingGrace (api goroutine died before the build was +// created) is reaped to "failed" via the guarded UPDATE — freeing the team's +// deployments_apps tier cap. No k8s Get happens (nothing to poll without a +// provider_id); the only DB write is the guarded reap UPDATE. +func TestDeployStatusReconciler_Work_StuckBuildingReaped(t *testing.T) { + db, mock, err := sqlmock.New(sqlmock.QueryMatcherOption(sqlmock.QueryMatcherRegexp)) + if err != nil { + t.Fatalf("sqlmock.New: %v", err) + } + defer db.Close() + + id := uuid.New() + stale := time.Now().UTC().Add(-stuckBuildingGrace - time.Minute) // > grace + mock.ExpectQuery(`FROM deployments\s+WHERE status IN`). + WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status", "created_at"}). + AddRow(id, "", "building", stale)) + + // The guarded reap UPDATE: status→failed, double-guarded on the prior + // status='building' AND empty provider_id. + mock.ExpectExec(`UPDATE deployments\s+SET status = \$1`). + WithArgs(deployStatusFailed, stuckBuildingReapMessage, id, deployStatusBuilding). + WillReturnResult(sqlmock.NewResult(0, 1)) + + // k8s provider must NOT be consulted for an empty-provider_id row. + k8s := newFakeDeployStatusK8s() + w := NewDeployStatusReconciler(db, k8s) + if werr := w.Work(context.Background(), fakeRiverJob[DeployStatusReconcileArgs]()); werr != nil { + t.Fatalf("Work: %v", werr) + } + if len(k8s.callLog) != 0 { + t.Errorf("empty-provider_id reap must not call GetDeployment, got calls: %v", k8s.callLog) + } + if err := mock.ExpectationsWereMet(); err != nil { + t.Errorf("unmet expectations: %v", err) + } +} + +// TestDeployStatusReconciler_Work_StuckBuildingFreshNotReaped pins the +// safety boundary: a FRESH (< stuckBuildingGrace) "building" row with an empty +// provider_id is a build still in flight — it MUST be left alone (skipped), no +// UPDATE, so a legitimately-in-progress build is never killed early. +func TestDeployStatusReconciler_Work_StuckBuildingFreshNotReaped(t *testing.T) { + db, mock, err := sqlmock.New(sqlmock.QueryMatcherOption(sqlmock.QueryMatcherRegexp)) + if err != nil { + t.Fatalf("sqlmock.New: %v", err) + } + defer db.Close() + + id := uuid.New() + fresh := time.Now().UTC().Add(-30 * time.Second) // well inside grace + mock.ExpectQuery(`FROM deployments\s+WHERE status IN`). + WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status", "created_at"}). + AddRow(id, "", "building", fresh)) + // No ExpectExec — any UPDATE is an unmet/unexpected expectation failure. + + w := NewDeployStatusReconciler(db, newFakeDeployStatusK8s()) + if werr := w.Work(context.Background(), fakeRiverJob[DeployStatusReconcileArgs]()); werr != nil { + t.Fatalf("Work: %v", werr) + } + if err := mock.ExpectationsWereMet(); err != nil { + t.Errorf("a fresh empty-provider_id build must not be reaped: %v", err) + } +} + +// TestDeployStatusReconciler_Work_StuckBuildingReapFailed covers the reap +// UPDATE error branch: the failure is logged and counted but the sweep does +// not abort (fail-open) — Work returns nil. +func TestDeployStatusReconciler_Work_StuckBuildingReapFailed(t *testing.T) { + db, mock, err := sqlmock.New(sqlmock.QueryMatcherOption(sqlmock.QueryMatcherRegexp)) + if err != nil { + t.Fatalf("sqlmock.New: %v", err) + } + defer db.Close() + + id := uuid.New() + stale := time.Now().UTC().Add(-stuckBuildingGrace - time.Minute) + mock.ExpectQuery(`FROM deployments\s+WHERE status IN`). + WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status", "created_at"}). + AddRow(id, "", "building", stale)) + mock.ExpectExec(`UPDATE deployments\s+SET status = \$1`). + WithArgs(deployStatusFailed, stuckBuildingReapMessage, id, deployStatusBuilding). + WillReturnError(errors.New("deadlock")) + + w := NewDeployStatusReconciler(db, newFakeDeployStatusK8s()) + if werr := w.Work(context.Background(), fakeRiverJob[DeployStatusReconcileArgs]()); werr != nil { + t.Fatalf("Work must isolate a reap-UPDATE failure (fail-open), got: %v", werr) + } + if err := mock.ExpectationsWereMet(); err != nil { + t.Errorf("unmet expectations: %v", err) + } +} + +// TestDeployStatusReconciler_Work_StuckDeployingEmptyProviderNotReaped guards +// that the reaper is scoped to status="building" ONLY: an empty-provider_id +// row in "deploying" (an unusual but possible interim state) is NOT reaped even +// when old — the guard is `d.status == deployStatusBuilding`. +func TestDeployStatusReconciler_Work_StuckDeployingEmptyProviderNotReaped(t *testing.T) { + db, mock, err := sqlmock.New(sqlmock.QueryMatcherOption(sqlmock.QueryMatcherRegexp)) + if err != nil { + t.Fatalf("sqlmock.New: %v", err) + } + defer db.Close() + + id := uuid.New() + stale := time.Now().UTC().Add(-stuckBuildingGrace - time.Hour) + mock.ExpectQuery(`FROM deployments\s+WHERE status IN`). + WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status", "created_at"}). + AddRow(id, "", "deploying", stale)) + // No ExpectExec — a non-building status must be skipped, never reaped. + + w := NewDeployStatusReconciler(db, newFakeDeployStatusK8s()) + if werr := w.Work(context.Background(), fakeRiverJob[DeployStatusReconcileArgs]()); werr != nil { + t.Fatalf("Work: %v", werr) + } + if err := mock.ExpectationsWereMet(); err != nil { + t.Errorf("a non-building empty-provider_id row must not be reaped: %v", err) + } +} + +// TestReapStuckBuilding_RowWithProviderIDUnaffected exercises the guarded +// helper directly: the UPDATE is double-guarded on empty provider_id, so a row +// that raced and acquired a provider_id between SELECT and UPDATE matches zero +// rows (RowsAffected 0) and the helper returns nil — a safe no-op. +func TestReapStuckBuilding_RowWithProviderIDUnaffected(t *testing.T) { + db, mock, err := sqlmock.New(sqlmock.QueryMatcherOption(sqlmock.QueryMatcherRegexp)) + if err != nil { + t.Fatalf("sqlmock.New: %v", err) + } + defer db.Close() + + id := uuid.New() + mock.ExpectExec(`UPDATE deployments\s+SET status = \$1`). + WithArgs(deployStatusFailed, stuckBuildingReapMessage, id, deployStatusBuilding). + WillReturnResult(sqlmock.NewResult(0, 0)) // guard matched nothing + + w := NewDeployStatusReconciler(db, newFakeDeployStatusK8s()) + if err := w.reapStuckBuilding(context.Background(), id); err != nil { + t.Fatalf("reapStuckBuilding no-op must not error: %v", err) + } + if err := mock.ExpectationsWereMet(); err != nil { + t.Errorf("unmet expectations: %v", err) + } +} + // ─── deploy_failure_autopsy.go ──────────────────────────────────────────────── // TestExtractPodFailure_TerminatedCurrentState exercises the @@ -1558,7 +1710,7 @@ func TestDeploymentReminderWorker_FullSweep(t *testing.T) { // row B: CAS wins, past expiry → floor branch + audit emit rows := sqlmock.NewRows(reminderCandidateCols). AddRow("deploy-A", teamID, "appA", "https://a.deployment.instanode.dev", - pastExpiry, 0, "auto_24h", "owner@example.com"). + pastExpiry, 0, "auto_24h", "owner@example.com"). AddRow("deploy-B", teamID, "appB", "", // empty app_url → deployURL fallback pastExpiry, 1, "auto_24h", nil) // nil email → audited_no_owner branch mock.ExpectQuery(`SELECT d.id::text, d.team_id::text, d.app_id, d.app_url`). @@ -2289,11 +2441,11 @@ func TestNewRazorpayOrphanCanceler_UnconfiguredReturnsNil(t *testing.T) { // TestRazorpayOrphanCanceler_CancelSubscription_Fake exercises the // production CancelSubscription wrapper logic through the test seam: -// - empty subID is a no-op -// - whitespace subID is a no-op -// - non-empty subID hits the SDK -// - SDK error fragment is mapped to "terminal" success -// - SDK error not in the terminal set is propagated +// - empty subID is a no-op +// - whitespace subID is a no-op +// - non-empty subID hits the SDK +// - SDK error fragment is mapped to "terminal" success +// - SDK error not in the terminal set is propagated func TestRazorpayOrphanCanceler_CancelSubscription_Fake(t *testing.T) { // Empty + whitespace. sdk := &fakeCancelSDKCov{} diff --git a/internal/jobs/deploy_status_reconcile.go b/internal/jobs/deploy_status_reconcile.go index df3146f..a6662cb 100644 --- a/internal/jobs/deploy_status_reconcile.go +++ b/internal/jobs/deploy_status_reconcile.go @@ -142,6 +142,11 @@ const ( deployStatusFailed = "failed" deployStatusStopped = "stopped" + // stuckBuildingReapMessage is stamped onto a reaped row's error_message + // (only when the api hadn't already written one) so the user-facing + // failure surface explains why the build never produced an app. + stuckBuildingReapMessage = "build did not start: api goroutine exited before the build was created (reaped after 15m)" + // providerIDPrefix mirrors api/internal/providers/compute/k8s/client.go's // deploymentName(appID) = "app-" + appID. providerIDPrefix = "app-" @@ -151,6 +156,24 @@ const ( // storing it on the deployments row. deployNamespacePrefix = "instant-deploy-" + // stuckBuildingGrace bounds how long a deployments row may sit at + // status="building" with an EMPTY provider_id before the reconciler reaps + // it to "failed". An empty provider_id normally means runDeploy() on the + // api side hasn't reached UpdateDeploymentProviderID yet (kaniko build in + // flight) — those fresh rows are left alone. But a deploy whose api + // goroutine DIED before that write (pod OOM, ctx kill, crash mid-runDeploy) + // leaves the row "building" with no provider_id FOREVER: the per-row sweep + // has nothing to poll (no namespace derivable without a provider_id), so it + // is skipped every tick and the row permanently consumes the team's + // deployments_apps tier cap (sweep finding #5, P2). + // + // 15m is well beyond the normal ~30-90s build, so the grace window cannot + // catch a legitimately in-flight build — a "building" row with no + // provider_id still present at 15m is genuinely wedged. Reaping it frees + // the tier cap; the separate deploy_failure_autopsy job (idempotent) then + // emits the deploy.failed audit → failure email. + stuckBuildingGrace = 15 * time.Minute + // buildJobNamePrefix mirrors the api's k8s.buildImage() jobName format: // jobName := "build-" + sanitizeName(appID) // (api/internal/providers/compute/k8s/client.go ~L1390 / L1217). @@ -310,6 +333,7 @@ type activeDeployment struct { id uuid.UUID providerID string status string + createdAt time.Time } // Work runs the full sweep. Errors on individual rows are logged and swallowed @@ -351,6 +375,7 @@ func (w *DeployStatusReconciler) Work(ctx context.Context, job *river.Job[Deploy transitions int errors int skipped int + reaped int // autopsiesThisTick counts failure-autopsy captures performed in // this sweep; deferred counts failed rows whose autopsy was // skipped because a per-tick cap was reached (BugBash 2026-05-18 @@ -363,6 +388,26 @@ func (w *DeployStatusReconciler) Work(ctx context.Context, job *river.Job[Deploy for _, d := range deployments { if d.providerID == "" { + // A "building" row with no provider_id whose api goroutine died + // before UpdateDeploymentProviderID (crash mid-runDeploy) sits + // here forever, permanently consuming the team's deployments_apps + // tier cap (sweep finding #5). Reap it once it's past the grace + // window — a fresh build (<15m) is still left alone (the build is + // genuinely in flight, runDeploy just hasn't stamped provider_id). + age := time.Since(d.createdAt) + if d.status == deployStatusBuilding && age > stuckBuildingGrace { + if err := w.reapStuckBuilding(ctx, d.id); err != nil { + slog.Error("jobs.deploy_status_reconcile.stuck_building_reap_failed", + "id", d.id, "age", age.String(), "error", err) + errors++ + continue + } + slog.Warn("jobs.deploy_status_reconcile.stuck_building_reaped", + "id", d.id, "age", age.String(), + "note", "building row with empty provider_id past grace window; api goroutine likely died before the build was created — reaped to failed to free the tier cap") + reaped++ + continue + } // runDeploy() hasn't reached UpdateDeploymentProviderID yet // (kaniko build still in flight on the api side). Nothing to // poll — leave the row alone. @@ -456,6 +501,7 @@ func (w *DeployStatusReconciler) Work(ctx context.Context, job *river.Job[Deploy "transitions", transitions, "errors", errors, "skipped", skipped, + "reaped", reaped, "autopsies", autopsiesThisTick, "autopsies_deferred", autopsiesDeferred, "duration_ms", time.Since(start).Milliseconds(), @@ -662,7 +708,7 @@ func deployNamespaceFromProviderID(providerID string) string { // "deploying" and "building" transitively — both are picked up here. func (w *DeployStatusReconciler) listActiveDeployments(ctx context.Context) ([]activeDeployment, error) { rows, err := w.db.QueryContext(ctx, ` - SELECT id, COALESCE(provider_id, ''), status + SELECT id, COALESCE(provider_id, ''), status, created_at FROM deployments WHERE status IN ($1, $2, $3) ORDER BY updated_at ASC @@ -675,7 +721,7 @@ func (w *DeployStatusReconciler) listActiveDeployments(ctx context.Context) ([]a var out []activeDeployment for rows.Next() { var d activeDeployment - if err := rows.Scan(&d.id, &d.providerID, &d.status); err != nil { + if err := rows.Scan(&d.id, &d.providerID, &d.status, &d.createdAt); err != nil { return nil, fmt.Errorf("listActiveDeployments: scan: %w", err) } out = append(out, d) @@ -704,3 +750,31 @@ func (w *DeployStatusReconciler) updateStatus(ctx context.Context, id uuid.UUID, } return nil } + +// reapStuckBuilding flips a deployments row that has been stuck at +// status="building" with no provider_id (api goroutine died before the build +// was created) to "failed", freeing the team's deployments_apps tier cap +// (sweep finding #5). +// +// We deliberately do NOT reuse updateStatus — that helper gates only on +// status IN (building, deploying, healthy) and would happily flip a row that +// has since acquired a provider_id. This UPDATE is double-guarded on BOTH +// status='building' AND a still-empty provider_id, so it is a no-op if the api +// raced us and stamped the provider_id (the next tick reconciles it normally) +// or already wrote a terminal status. The COALESCE/NULLIF preserves any +// error_message the api may have written before crashing. +func (w *DeployStatusReconciler) reapStuckBuilding(ctx context.Context, id uuid.UUID) error { + _, err := w.db.ExecContext(ctx, ` + UPDATE deployments + SET status = $1, + error_message = COALESCE(NULLIF(error_message, ''), $2), + updated_at = now() + WHERE id = $3 + AND status = $4 + AND (provider_id IS NULL OR provider_id = '') + `, deployStatusFailed, stuckBuildingReapMessage, id, deployStatusBuilding) + if err != nil { + return fmt.Errorf("reapStuckBuilding: %w", err) + } + return nil +} diff --git a/internal/jobs/deploy_status_reconcile_job_failed_test.go b/internal/jobs/deploy_status_reconcile_job_failed_test.go index 1b64856..5dff2c3 100644 --- a/internal/jobs/deploy_status_reconcile_job_failed_test.go +++ b/internal/jobs/deploy_status_reconcile_job_failed_test.go @@ -24,6 +24,7 @@ import ( "context" "errors" "testing" + "time" sqlmock "github.com/DATA-DOG/go-sqlmock" "github.com/google/uuid" @@ -109,8 +110,8 @@ func TestDeployStatusReconcile_JobFailedAfterPodGC(t *testing.T) { id := uuid.New() mock.ExpectQuery(`FROM deployments\s+WHERE status IN`). - WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status"}). - AddRow(id, "app-gced", "building")) + WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status", "created_at"}). + AddRow(id, "app-gced", "building", time.Now().UTC())) k8s := newFakeDeployStatusK8s() // Deployment is missing (build never reached the apply step) — pre-fix @@ -153,8 +154,8 @@ func TestDeployStatusReconcile_JobActiveAndPodMissing_StaysBuilding(t *testing.T id := uuid.New() mock.ExpectQuery(`FROM deployments\s+WHERE status IN`). - WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status"}). - AddRow(id, "app-active", "building")) + WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status", "created_at"}). + AddRow(id, "app-active", "building", time.Now().UTC())) k8s := newFakeDeployStatusK8s() // Deployment missing — apply step hasn't run yet. @@ -186,8 +187,8 @@ func TestDeployStatusReconcile_BothNotFound_StaysStopped(t *testing.T) { id := uuid.New() mock.ExpectQuery(`FROM deployments\s+WHERE status IN`). - WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status"}). - AddRow(id, "app-gone", "building")) + WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status", "created_at"}). + AddRow(id, "app-gone", "building", time.Now().UTC())) // Both Deployment AND Job missing from the fake → NewNotFound errors. k8s := newFakeDeployStatusK8s() @@ -219,8 +220,8 @@ func TestDeployStatusReconcile_JobQueryError_FallsThroughToDeployment(t *testing id := uuid.New() mock.ExpectQuery(`FROM deployments\s+WHERE status IN`). - WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status"}). - AddRow(id, "app-h1", "building")) + WillReturnRows(sqlmock.NewRows([]string{"id", "provider_id", "status", "created_at"}). + AddRow(id, "app-h1", "building", time.Now().UTC())) k8s := newFakeDeployStatusK8s() // Healthy runtime Deployment — Deployment query MUST be authoritative.