@agent-native/core 0.76.0 → 0.76.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/corpus/core/CHANGELOG.md +56 -0
  2. package/corpus/core/package.json +1 -1
  3. package/corpus/core/src/agent/durable-background.ts +58 -18
  4. package/corpus/core/src/agent/production-agent.ts +54 -0
  5. package/corpus/core/src/agent/run-manager.ts +27 -0
  6. package/corpus/core/src/agent/run-store.ts +185 -4
  7. package/corpus/core/src/deploy/build.ts +8 -6
  8. package/corpus/core/src/server/agent-chat-plugin.ts +71 -0
  9. package/corpus/core/src/server/auth.ts +13 -0
  10. package/corpus/templates/analytics/app/components/layout/CommandPalette.tsx +12 -12
  11. package/corpus/templates/analytics/app/components/layout/Sidebar.tsx +23 -23
  12. package/corpus/templates/analytics/app/pages/analyses/AnalysesList.tsx +2 -2
  13. package/corpus/templates/brain/app/routes/knowledge.tsx +6 -6
  14. package/corpus/templates/brain/app/routes/ops.tsx +4 -4
  15. package/corpus/templates/brain/app/routes/search.tsx +4 -4
  16. package/corpus/templates/brain/app/routes/settings.tsx +3 -3
  17. package/corpus/templates/brain/app/routes/sources.tsx +3 -3
  18. package/corpus/templates/dispatch/app/routes/integrations.tsx +3 -3
  19. package/corpus/templates/mail/app/components/email/EmailThread.tsx +10 -10
  20. package/corpus/templates/mail/app/pages/DraftQueuePage.tsx +3 -3
  21. package/corpus/templates/mail/app/pages/InboxPage.tsx +3 -3
  22. package/corpus/templates/mail/app/pages/SettingsPage.tsx +2 -2
  23. package/corpus/templates/plan/app/pages/PlansPage.tsx +3 -3
  24. package/corpus/templates/slides/app/components/deck/DeckCard.tsx +4 -4
  25. package/corpus/templates/slides/app/components/layout/Layout.tsx +1 -1
  26. package/corpus/templates/slides/app/components/layout/Sidebar.tsx +4 -4
  27. package/corpus/templates/slides/app/pages/Index.tsx +3 -3
  28. package/corpus/templates/videos/app/components/layout/NavSidebar.tsx +1 -1
  29. package/corpus/templates/videos/app/pages/ComponentLibraryView.tsx +1 -1
  30. package/corpus/templates/videos/app/pages/DesignSystems.tsx +1 -1
  31. package/dist/agent/durable-background.d.ts +19 -3
  32. package/dist/agent/durable-background.d.ts.map +1 -1
  33. package/dist/agent/durable-background.js +47 -17
  34. package/dist/agent/durable-background.js.map +1 -1
  35. package/dist/agent/production-agent.d.ts.map +1 -1
  36. package/dist/agent/production-agent.js +34 -1
  37. package/dist/agent/production-agent.js.map +1 -1
  38. package/dist/agent/run-manager.d.ts +8 -0
  39. package/dist/agent/run-manager.d.ts.map +1 -1
  40. package/dist/agent/run-manager.js +18 -1
  41. package/dist/agent/run-manager.js.map +1 -1
  42. package/dist/agent/run-store.d.ts +99 -0
  43. package/dist/agent/run-store.d.ts.map +1 -1
  44. package/dist/agent/run-store.js +155 -4
  45. package/dist/agent/run-store.js.map +1 -1
  46. package/dist/deploy/build.d.ts +6 -4
  47. package/dist/deploy/build.d.ts.map +1 -1
  48. package/dist/deploy/build.js +8 -6
  49. package/dist/deploy/build.js.map +1 -1
  50. package/dist/server/agent-chat-plugin.d.ts.map +1 -1
  51. package/dist/server/agent-chat-plugin.js +58 -1
  52. package/dist/server/agent-chat-plugin.js.map +1 -1
  53. package/dist/server/auth.d.ts.map +1 -1
  54. package/dist/server/auth.js +12 -0
  55. package/dist/server/auth.js.map +1 -1
  56. package/package.json +1 -1
@@ -1,5 +1,61 @@
1
1
  # @agent-native/core
2
2
 
3
+ ## 0.76.2
4
+
5
+ ### Patch Changes
6
+
7
+ - 93c06b0: Make durable background agent runs opt-in (default-off) again. Both the runtime
8
+ gate (`isFlagEnabled` in durable-background.ts) and the deploy-time `-background`
9
+ emit gate (`isDurableBackgroundDeployEnabled` in deploy/build.ts) now default to
10
+ OFF when `AGENT_CHAT_DURABLE_BACKGROUND` is unset; an app opts in only with an
11
+ explicit truthy value (`true`/`1`/`yes`/`on`). A premature fleet-wide default-on
12
+ caused real-user incidents (apps hit "Failed to dispatch background run" + chat
13
+ stalls) because the async background-function worker path is not yet proven
14
+ end-to-end and the deploy-time env opt-out is not reliably baked into a given
15
+ deploy. Re-enable default-on only after the 15-min background-function worker is
16
+ verified live in production.
17
+
18
+ ## 0.76.1
19
+
20
+ ### Patch Changes
21
+
22
+ - 50f32ff: Make durable-background agent-chat worker failures diagnosable from the client
23
+ and harden recovery when the background worker never starts.
24
+
25
+ A durable-background run is dispatched into a Netlify `-background` function,
26
+ which acks asynchronously with a 202. If that worker then dies silently (its
27
+ logs are not readable from the build tooling), the run would just time out with
28
+ no clue why, and because dispatch already returned 202 the existing fast-fail
29
+ inline fallback never engaged — so the run errored opaquely.
30
+
31
+ Diagnostics (readable WITHOUT bg-fn logs). The `_process-run` worker pipeline
32
+ now records the last reached stage onto the run row (`agent_runs.diag_stage`, a
33
+ compact JSON `{stage,detail?,at}`) via the new best-effort `recordRunDiagnostic`:
34
+ route entered, HMAC auth pass/fail (recorded onto the run BEFORE the 401/503 is
35
+ returned, including whether `A2A_SECRET` is present in the bg-fn isolate),
36
+ worker entered (with the resolved `runsInBackgroundFunction` value), claim
37
+ win/lose, worker loop started, and any thrown error. `/runs/active?threadId=`
38
+ (and `listRunsForThread`) now surface `dispatchMode` and `diagStage`, so the
39
+ next prod run's death cause is readable straight from the client.
40
+
41
+ Recovery (covers "202 acked but worker never started"). A background-dispatched
42
+ run that is still unclaimed (`dispatch_mode = 'background'`, never flipped to
43
+ `background-processing`) past a tight 25s grace is reaped early and recoverably
44
+ with the new `background_worker_never_started` error code (the wide 90s window
45
+ only exists to protect a CLAIMED, cold-starting worker — an unclaimed run has no
46
+ worker to protect). The `/runs/active` read path attempts this recovery before
47
+ the generic stale reaper, so a silent worker death surfaces as a recoverable
48
+ error the client can re-drive instead of hanging for 90s.
49
+
50
+ Also fixes a latent gate: `/_agent-native/agent-chat/_process-run` now bypasses
51
+ the session-auth guard (mirroring the agent-teams processor). The self-dispatch
52
+ carries only an HMAC Bearer token and no session cookie, so without the bypass
53
+ the worker was 401'd before it could authenticate and claim the run.
54
+
55
+ - 50f32ff: Tighten bundled visual plan wireframe guidance so agents use literal spacing,
56
+ pad root containers, and choose feature-cloud layouts for abundance-style
57
+ marketing sections.
58
+
3
59
  ## 0.76.0
4
60
 
5
61
  ### Minor Changes
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@agent-native/core",
3
- "version": "0.76.0",
3
+ "version": "0.76.2",
4
4
  "type": "module",
5
5
  "engines": {
6
6
  "node": ">=22"
@@ -237,30 +237,34 @@ function isFlagEnabled(): boolean {
237
237
  // can statically verify it against the allowlisted `AGENT_*` prefix. Keep this
238
238
  // in sync with AGENT_CHAT_DURABLE_BACKGROUND_ENV.
239
239
  //
240
- // DEFAULT-ON: durable background runs are the desired behavior for every
241
- // hosted app. So an unset/empty/unknown flag means ON; an app opts OUT only
242
- // with an explicit falsy value. This still composes with the hosted +
243
- // A2A_SECRET gates below, so non-hosted / unconfigured apps stay synchronous.
244
- // Safety net: a failed dispatch degrades to a synchronous inline run (see
245
- // production-agent.ts), so default-on cannot break chat even if the
246
- // self-dispatch can't be delivered on a given app.
240
+ // DEFAULT-OFF (opt-in): durable background runs are still being hardened. A
241
+ // premature fleet-wide default-on caused real-user incidents (Assets/Analytics
242
+ // hit "Failed to dispatch" + stalls, 2026-06-24) because the async background
243
+ // worker path is not yet proven end-to-end and the deploy-time env opt-out is
244
+ // not reliably baked into a given deploy. So an unset/empty/unknown flag means
245
+ // OFF; an app opts IN only with an explicit truthy value
246
+ // (AGENT_CHAT_DURABLE_BACKGROUND=true). This still composes with the hosted +
247
+ // A2A_SECRET gates below. Flip back to default-on only after the 15-min
248
+ // background-function worker is verified live in production (see the
249
+ // project_durable_bg_prod_verified memory).
247
250
  const raw = process.env.AGENT_CHAT_DURABLE_BACKGROUND;
248
- if (raw == null) return true;
251
+ if (raw == null) return false;
249
252
  const normalized = raw.trim().toLowerCase();
250
- return !(
251
- normalized === "0" ||
252
- normalized === "false" ||
253
- normalized === "no" ||
254
- normalized === "off"
253
+ return (
254
+ normalized === "1" ||
255
+ normalized === "true" ||
256
+ normalized === "yes" ||
257
+ normalized === "on"
255
258
  );
256
259
  }
257
260
 
258
261
  /**
259
- * The single gate. True when the flag is not explicitly disabled (default-on)
262
+ * The single gate. True when the flag is explicitly enabled (opt-in/default-off)
260
263
  * AND the runtime is hosted AND A2A_SECRET is configured. False otherwise — and
261
264
  * false means the current synchronous behavior is used, unchanged. So a local /
262
- * non-hosted / unconfigured app stays synchronous even with the flag defaulting
263
- * on; durable only engages where the runtime actually supports it.
265
+ * non-hosted / unconfigured app stays synchronous, and an app that has not opted
266
+ * in stays synchronous; durable only engages where it is explicitly enabled and
267
+ * the runtime actually supports it.
264
268
  */
265
269
  export function isAgentChatDurableBackgroundEnabled(): boolean {
266
270
  return (
@@ -285,8 +289,37 @@ export type ProcessRunPreparation =
285
289
  status: number;
286
290
  /** Error payload. */
287
291
  error: string;
292
+ /**
293
+ * The run id parsed from the body, when present. Carried even on failure
294
+ * so the route can RECORD the auth/validation failure ONTO the run
295
+ * (diag_stage) before returning the error status — otherwise a 401/503 in
296
+ * the unreadable Netlify background function would leave the run to time
297
+ * out with no clue why. Null when no run id could be parsed.
298
+ */
299
+ runId: string | null;
288
300
  };
289
301
 
302
+ /**
303
+ * Parse the run id from a `_process-run` request body without authenticating.
304
+ * Mirrors the precedence in `prepareProcessRunRequest` (marker.runId, then
305
+ * top-level taskId). Returns null when neither is a usable string. Used so the
306
+ * route can attach a diagnostic to the run even on an auth/validation failure.
307
+ */
308
+ export function extractProcessRunId(body: unknown): string | null {
309
+ if (!body || typeof body !== "object") return null;
310
+ const record = body as Record<string, unknown>;
311
+ const marker = record[AGENT_CHAT_BACKGROUND_RUN_FIELD] as
312
+ | { runId?: unknown }
313
+ | undefined;
314
+ if (marker && typeof marker.runId === "string" && marker.runId) {
315
+ return marker.runId;
316
+ }
317
+ if (typeof record.taskId === "string" && record.taskId) {
318
+ return record.taskId;
319
+ }
320
+ return null;
321
+ }
322
+
290
323
  /**
291
324
  * Pure, transport-agnostic core of the `_process-run` route: validate the body,
292
325
  * authenticate the HMAC self-dispatch, and produce the body the re-entered
@@ -308,7 +341,12 @@ export function prepareProcessRunRequest(
308
341
  authHeader: string | undefined,
309
342
  ): ProcessRunPreparation {
310
343
  if (!body || typeof body !== "object") {
311
- return { ok: false, status: 400, error: "Invalid request body" };
344
+ return {
345
+ ok: false,
346
+ status: 400,
347
+ error: "Invalid request body",
348
+ runId: null,
349
+ };
312
350
  }
313
351
  const record = body as Record<string, unknown>;
314
352
  const marker = record[AGENT_CHAT_BACKGROUND_RUN_FIELD] as
@@ -321,7 +359,7 @@ export function prepareProcessRunRequest(
321
359
  ? (record.taskId as string)
322
360
  : "";
323
361
  if (!runId) {
324
- return { ok: false, status: 400, error: "runId required" };
362
+ return { ok: false, status: 400, error: "runId required", runId: null };
325
363
  }
326
364
 
327
365
  if (hasConfiguredA2ASecret()) {
@@ -331,6 +369,7 @@ export function prepareProcessRunRequest(
331
369
  ok: false,
332
370
  status: 401,
333
371
  error: "Invalid or expired processor token",
372
+ runId,
334
373
  };
335
374
  }
336
375
  } else if (isA2AProductionRuntime()) {
@@ -339,6 +378,7 @@ export function prepareProcessRunRequest(
339
378
  status: 503,
340
379
  error:
341
380
  "Agent chat background processor not configured — set A2A_SECRET on this deployment.",
381
+ runId,
342
382
  };
343
383
  }
344
384
 
@@ -109,6 +109,8 @@ import {
109
109
  updateRunHeartbeat,
110
110
  updateRunStatusIfRunning,
111
111
  claimBackgroundRun,
112
+ recordRunDiagnostic,
113
+ RUN_DIAG_STAGE,
112
114
  } from "./run-store.js";
113
115
  import {
114
116
  classifyToolCallJournal,
@@ -4599,6 +4601,30 @@ export function createProductionAgentHandler(
4599
4601
  isBackgroundWorker || baseHandleRunComplete
4600
4602
  ? async (run: ActiveRun) => {
4601
4603
  try {
4604
+ // DIAGNOSTIC: a background worker that completed in an errored
4605
+ // state threw inside the loop. Record it (with the last error
4606
+ // event's message when available) so the failure cause is
4607
+ // readable from the client. Skipped for clean completions and for
4608
+ // recoverable soft-timeout boundaries (those chain a continuation
4609
+ // below, they did not "throw").
4610
+ if (
4611
+ isBackgroundWorker &&
4612
+ run.status === "errored" &&
4613
+ !endsAtInternalContinuationBoundary(run)
4614
+ ) {
4615
+ const errEvent = [...run.events]
4616
+ .reverse()
4617
+ .find((e) => e.event.type === "error")?.event as
4618
+ | { error?: string; errorCode?: string }
4619
+ | undefined;
4620
+ await recordRunDiagnostic(
4621
+ run.runId,
4622
+ RUN_DIAG_STAGE.workerThrew,
4623
+ errEvent?.errorCode || errEvent?.error
4624
+ ? `${errEvent.errorCode ?? ""} ${errEvent.error ?? ""}`.trim()
4625
+ : "run ended in errored state",
4626
+ ).catch(() => {});
4627
+ }
4602
4628
  // Persist the (partial) assistant turn to thread_data FIRST — the
4603
4629
  // server-driven continuation below rebuilds from it, so it must be
4604
4630
  // committed before we re-fire.
@@ -4689,6 +4715,17 @@ export function createProductionAgentHandler(
4689
4715
  // on entry so a slow cold-start doesn't leave the row looking stale to the
4690
4716
  // reaper before startRun's 1.5s heartbeat timer takes over.
4691
4717
  if (isBackgroundWorker) {
4718
+ // DIAGNOSTIC: the re-entered handler recognized itself as the background
4719
+ // worker. Record the runtime regime too — `isInBackgroundFunctionRuntime()`
4720
+ // reads a globalThis marker set by the bg-fn entry, which may NOT be set in
4721
+ // this isolate; recording the ACTUAL resolved value reveals whether the
4722
+ // worker is on the 13-min `-background` budget or the 40s clamp. This is
4723
+ // the proof the worker reached its own code (vs. dying at auth before it).
4724
+ await recordRunDiagnostic(
4725
+ runId,
4726
+ RUN_DIAG_STAGE.workerEntered,
4727
+ `runsInBackgroundFunction=${runsInBackgroundFunction} continuationCount=${backgroundContinuationCount}`,
4728
+ ).catch(() => {});
4692
4729
  // A chained continuation chunk's runId was minted by the prior chunk and
4693
4730
  // never inserted, so insert its background row now (idempotently — a
4694
4731
  // duplicate Netlify delivery that already inserted it just PK-collides and
@@ -4703,8 +4740,16 @@ export function createProductionAgentHandler(
4703
4740
  if (!won) {
4704
4741
  // Already claimed by an earlier delivery — return a benign ack so
4705
4742
  // Netlify doesn't retry a successful handoff.
4743
+ await recordRunDiagnostic(runId, RUN_DIAG_STAGE.workerClaimLost).catch(
4744
+ () => {},
4745
+ );
4706
4746
  return { ok: true, skipped: "already-claimed" };
4707
4747
  }
4748
+ // DIAGNOSTIC: this worker won the claim and now OWNS the run. If a run
4749
+ // ever stalls at this stage it means the loop below failed to start.
4750
+ await recordRunDiagnostic(runId, RUN_DIAG_STAGE.workerClaimed).catch(
4751
+ () => {},
4752
+ );
4708
4753
  await updateRunHeartbeat(runId).catch(() => {});
4709
4754
  }
4710
4755
 
@@ -4719,6 +4764,15 @@ export function createProductionAgentHandler(
4719
4764
 
4720
4765
  send({ type: "activity", label: "Starting agent" });
4721
4766
 
4767
+ // DIAGNOSTIC: the agent loop body actually started running. For a
4768
+ // background worker, a run that is claimed but never reaches this stage
4769
+ // died between claiming and loop start. Best-effort, background only.
4770
+ if (isBackgroundWorker) {
4771
+ await recordRunDiagnostic(runId, RUN_DIAG_STAGE.workerStarted).catch(
4772
+ () => {},
4773
+ );
4774
+ }
4775
+
4722
4776
  // Notify listeners that a run has started (used by agent teams)
4723
4777
  if (options.onRunStart) {
4724
4778
  await options.onRunStart(send, threadId ?? runId);
@@ -15,6 +15,7 @@ import {
15
15
  updateRunHeartbeat,
16
16
  bumpRunProgress,
17
17
  reapIfStale,
18
+ reapUnclaimedBackgroundRun,
18
19
  ensureTerminalRunEvent,
19
20
  setRunError,
20
21
  STALE_RUN_ERROR_EVENT,
@@ -1022,6 +1023,14 @@ export async function getActiveRunForThreadAsync(threadId: string): Promise<{
1022
1023
  status: string;
1023
1024
  heartbeatAt: number;
1024
1025
  lastProgressAt: number | null;
1026
+ /** How the run was dispatched (NULL/foreground, background, background-processing). */
1027
+ dispatchMode?: string | null;
1028
+ /**
1029
+ * Last reached `_process-run` worker stage as a JSON string
1030
+ * `{stage,detail?,at}`. Surfaced so a silent background-worker death is
1031
+ * diagnosable from the client WITHOUT the unreadable bg-fn logs.
1032
+ */
1033
+ diagStage?: string | null;
1025
1034
  } | null> {
1026
1035
  // Check memory first — return both running AND recently-completed runs
1027
1036
  // that still have events in memory. This allows sub-agent tabs to replay
@@ -1055,6 +1064,20 @@ export async function getActiveRunForThreadAsync(threadId: string): Promise<{
1055
1064
  const sqlRun = await getRunByThread(threadId, { includeTerminal: true });
1056
1065
  if (!sqlRun) return null;
1057
1066
  if (sqlRun.status === "running") {
1067
+ // FALLBACK HARDENING: a background-dispatched run that is still UNCLAIMED
1068
+ // (dispatch_mode === 'background', never flipped to 'background-processing')
1069
+ // past the tight grace means the bg-fn worker never started — a silent
1070
+ // async-worker death that the 202-ack inline fallback can't catch. Reap it
1071
+ // early and recoverably (background_worker_never_started) so the run no
1072
+ // longer hangs for the full 90s window and the client's recoverable-error
1073
+ // path can re-drive the turn. Only fires when there is provably no live
1074
+ // worker; a claimed/heartbeating run is left alone by the conditional SQL.
1075
+ if (sqlRun.dispatchMode === "background") {
1076
+ const recovered = await reapUnclaimedBackgroundRun(sqlRun.id).catch(
1077
+ () => false,
1078
+ );
1079
+ if (recovered) return null;
1080
+ }
1058
1081
  // If the producer is dead (no recent heartbeat), reap before the
1059
1082
  // client can see a stale "running" status and enter a reconnect
1060
1083
  // loop it can never exit.
@@ -1067,6 +1090,8 @@ export async function getActiveRunForThreadAsync(threadId: string): Promise<{
1067
1090
  status: sqlRun.status,
1068
1091
  heartbeatAt: sqlRun.heartbeatAt ?? sqlRun.startedAt,
1069
1092
  lastProgressAt: sqlRun.lastProgressAt,
1093
+ dispatchMode: sqlRun.dispatchMode,
1094
+ diagStage: sqlRun.diagStage,
1070
1095
  };
1071
1096
  }
1072
1097
  if (sqlRun.status === "completed" || sqlRun.status === "errored") {
@@ -1092,6 +1117,8 @@ export async function getActiveRunForThreadAsync(threadId: string): Promise<{
1092
1117
  status: sqlRun.status,
1093
1118
  heartbeatAt: sqlRun.heartbeatAt ?? sqlRun.startedAt,
1094
1119
  lastProgressAt: sqlRun.lastProgressAt,
1120
+ dispatchMode: sqlRun.dispatchMode,
1121
+ diagStage: sqlRun.diagStage,
1095
1122
  };
1096
1123
  }
1097
1124
  } catch {
@@ -44,6 +44,40 @@ export const STALE_RUN_ERROR_EVENT = {
44
44
  "The run heartbeat stopped while the run was still marked running. Partial output and tool calls were preserved when available.",
45
45
  } as const;
46
46
 
47
+ /**
48
+ * Terminal error for a background-dispatched run whose worker NEVER claimed it
49
+ * (the foreground fired the self-dispatch, Netlify acked it async with a 202,
50
+ * but the `_process-run` worker never ran far enough to flip
51
+ * `dispatch_mode background → background-processing`). Distinct errorCode so the
52
+ * client (and prod triage) can tell "the worker died silently" apart from "a
53
+ * claimed worker's heartbeat went stale". Recoverable so the client surfaces a
54
+ * retry affordance and re-drives the turn. See `reapUnclaimedBackgroundRun`.
55
+ */
56
+ export const UNCLAIMED_BACKGROUND_RUN_ERROR_EVENT = {
57
+ type: "error",
58
+ error:
59
+ "The agent run was handed off to a background worker that never started. It was recovered so you can try again.",
60
+ errorCode: "background_worker_never_started",
61
+ recoverable: true,
62
+ details:
63
+ "A background-dispatched run was acknowledged (HTTP 202) but its worker never claimed the run, so no progress was produced. The run was reaped early (it had no live worker to protect) so the turn can be retried.",
64
+ } as const;
65
+
66
+ /**
67
+ * Grace period before a never-claimed background run (dispatch_mode still
68
+ * 'background', no worker claim) is treated as a dead handoff and reaped.
69
+ *
70
+ * This is intentionally MUCH tighter than `BACKGROUND_RUN_STALE_MS` (90s). The
71
+ * wide 90s window exists ONLY to protect a CLAIMED worker whose heartbeat lags
72
+ * during a slow cold start. A run that is still `dispatch_mode = 'background'`
73
+ * has, by definition, NO worker — nothing to protect — so once a Netlify
74
+ * background function has had a reasonable cold-start window to claim it and
75
+ * hasn't, the handoff is dead and should surface promptly instead of leaving
76
+ * the user staring at a spinner for 90s. 25s comfortably exceeds a normal
77
+ * Netlify Lambda cold start while still failing fast on a silent worker death.
78
+ */
79
+ export const UNCLAIMED_BACKGROUND_RUN_GRACE_MS = 25_000;
80
+
47
81
  async function ensureRunTables(): Promise<void> {
48
82
  if (!_initPromise) {
49
83
  _initPromise = (async () => {
@@ -61,7 +95,8 @@ async function ensureRunTables(): Promise<void> {
61
95
  turn_id TEXT,
62
96
  error_code TEXT,
63
97
  error_detail TEXT,
64
- dispatch_mode TEXT
98
+ dispatch_mode TEXT,
99
+ diag_stage TEXT
65
100
  )
66
101
  `);
67
102
  // Backfill heartbeat_at on older deployments.
@@ -118,11 +153,16 @@ async function ensureRunTables(): Promise<void> {
118
153
  // normal synchronous path, "background" for a run dispatched into a
119
154
  // Netlify background function. The reaper/claim widen the stale window
120
155
  // for background rows so a slow cold-start isn't falsely reaped.
156
+ // diag_stage records the last reached pipeline stage (+ any error) for a
157
+ // background-dispatched run so a silent worker death is DIAGNOSABLE from
158
+ // the client (/runs/active surfaces it) without reading the unreadable
159
+ // Netlify background-function logs. See recordRunDiagnostic.
121
160
  for (const col of [
122
161
  "turn_id",
123
162
  "error_code",
124
163
  "error_detail",
125
164
  "dispatch_mode",
165
+ "diag_stage",
126
166
  ] as const) {
127
167
  try {
128
168
  if (isPostgres()) {
@@ -416,6 +456,72 @@ export async function setRunError(
416
456
  }
417
457
  }
418
458
 
459
+ /**
460
+ * Diagnostic stage names recorded onto a background run as it moves through the
461
+ * `_process-run` worker pipeline. Each value is the LAST stage successfully
462
+ * reached, so a stuck run's `diag_stage` reveals exactly where it died. Ordered
463
+ * roughly by execution; the literal strings are the client-readable contract.
464
+ */
465
+ export const RUN_DIAG_STAGE = {
466
+ /** The `_process-run` route handler was entered (the request reached Nitro). */
467
+ routeEntered: "route_entered",
468
+ /** HMAC auth + body validation in prepareProcessRunRequest FAILED. */
469
+ authFailed: "auth_failed",
470
+ /** HMAC auth + body validation PASSED; about to invoke the worker handler. */
471
+ authPassed: "auth_passed",
472
+ /** The re-entered agent-chat handler recognized itself as the bg worker. */
473
+ workerEntered: "worker_entered",
474
+ /** The worker won the atomic claim (it owns the run). */
475
+ workerClaimed: "worker_claimed",
476
+ /** The worker LOST the claim (a duplicate delivery already owns the run). */
477
+ workerClaimLost: "worker_claim_lost",
478
+ /** The agent loop started (startRun fired). */
479
+ workerStarted: "worker_started",
480
+ /** The worker threw before/while running the loop (message carried in detail). */
481
+ workerThrew: "worker_threw",
482
+ /** The route handler caught an error from the worker invocation. */
483
+ routeThrew: "route_threw",
484
+ } as const;
485
+
486
+ export type RunDiagStage = (typeof RUN_DIAG_STAGE)[keyof typeof RUN_DIAG_STAGE];
487
+
488
+ /**
489
+ * Record the last reached pipeline stage (+ optional short detail) for a run.
490
+ *
491
+ * PURPOSE: a Netlify background function's logs are not readable from the build
492
+ * tooling, so when its worker dies silently the run just times out with no clue
493
+ * WHY. This writes the failure stage straight onto the `agent_runs` row, which
494
+ * `/runs/active` and `listRunsForThread` surface to the client — so the next
495
+ * prod run's death cause is readable WITHOUT bg-fn logs. Cheap, additive, and
496
+ * best-effort: it must never throw or perturb the run (it is called on the auth
497
+ * path BEFORE a 401 is returned, and around the worker body).
498
+ *
499
+ * The stored value is a compact JSON `{ stage, detail?, at }` capped to 2 KB so
500
+ * a long stack can't bloat the row.
501
+ */
502
+ export async function recordRunDiagnostic(
503
+ runId: string,
504
+ stage: RunDiagStage,
505
+ detail?: string,
506
+ ): Promise<void> {
507
+ if (!runId) return;
508
+ try {
509
+ await ensureRunTables();
510
+ const client = getDbExec();
511
+ const payload = JSON.stringify({
512
+ stage,
513
+ ...(detail ? { detail: detail.slice(0, 1500) } : {}),
514
+ at: Date.now(),
515
+ }).slice(0, 2000);
516
+ await client.execute({
517
+ sql: `UPDATE agent_runs SET diag_stage = ? WHERE id = ?`,
518
+ args: [payload, runId],
519
+ });
520
+ } catch {
521
+ // Diagnostics are best-effort; never let them break the run or the route.
522
+ }
523
+ }
524
+
419
525
  /** Update the run's liveness heartbeat. Called periodically by run-manager. */
420
526
  export async function updateRunHeartbeat(runId: string): Promise<void> {
421
527
  await ensureRunTables();
@@ -490,6 +596,68 @@ export async function reapIfStale(
490
596
  return reaped;
491
597
  }
492
598
 
599
+ /**
600
+ * FALLBACK HARDENING for the "dispatched with 202 but the worker never started"
601
+ * case. A background-dispatched run sits in `dispatch_mode = 'background'` until
602
+ * the worker wins `claimBackgroundRun` (which flips it to
603
+ * `background-processing`). If the worker silently dies (e.g. the bg-fn 401s
604
+ * before it can claim), the row stays `background`, never heartbeats again, and
605
+ * — because dispatch returned 202 — the foreground already returned the SSE
606
+ * stream, so the existing fast-fail inline fallback never engaged. The run would
607
+ * otherwise hang for the full 90s background window and then error opaquely.
608
+ *
609
+ * This reaps such a run EARLY and DISTINCTLY: a row that is still unclaimed
610
+ * (`dispatch_mode = 'background'`) past the tight `UNCLAIMED_BACKGROUND_RUN_GRACE_MS`
611
+ * grace is a dead handoff — there is no live worker to protect with the wide
612
+ * window — so we flip it to `errored` with the recoverable
613
+ * `background_worker_never_started` code. The client's existing recoverable-error
614
+ * path then lets the user (or auto-recovery) re-drive the turn. Idempotent and
615
+ * conditional: only an unclaimed, still-running, grace-exceeded row matches, so a
616
+ * claimed worker, a fresh dispatch, or a terminal row is never touched.
617
+ *
618
+ * Returns true when this call reaped the run.
619
+ */
620
+ export async function reapUnclaimedBackgroundRun(
621
+ runId: string,
622
+ ): Promise<boolean> {
623
+ await ensureRunTables();
624
+ const client = getDbExec();
625
+ const completedAt = Date.now();
626
+ const cutoff = completedAt - UNCLAIMED_BACKGROUND_RUN_GRACE_MS;
627
+ const { rowsAffected } = await client.execute({
628
+ sql: `UPDATE agent_runs
629
+ SET status = 'errored',
630
+ completed_at = ?,
631
+ error_code = ?,
632
+ error_detail = ?
633
+ WHERE id = ?
634
+ AND status = 'running'
635
+ AND dispatch_mode = 'background'
636
+ AND COALESCE(heartbeat_at, started_at) < ?`,
637
+ args: [
638
+ completedAt,
639
+ UNCLAIMED_BACKGROUND_RUN_ERROR_EVENT.errorCode,
640
+ UNCLAIMED_BACKGROUND_RUN_ERROR_EVENT.details,
641
+ runId,
642
+ cutoff,
643
+ ],
644
+ });
645
+ const reaped = (rowsAffected ?? 0) > 0;
646
+ if (reaped) {
647
+ await recordRunDiagnostic(
648
+ runId,
649
+ RUN_DIAG_STAGE.workerThrew,
650
+ "unclaimed background dispatch reaped (worker never claimed the run)",
651
+ );
652
+ await safeAppendTerminalRunEvent(
653
+ runId,
654
+ UNCLAIMED_BACKGROUND_RUN_ERROR_EVENT,
655
+ "reap-unclaimed-background",
656
+ );
657
+ }
658
+ return reaped;
659
+ }
660
+
493
661
  export async function updateRunStatus(
494
662
  runId: string,
495
663
  status: "completed" | "errored" | "aborted",
@@ -644,12 +812,14 @@ export async function getRunByThread(
644
812
  heartbeatAt: number | null;
645
813
  completedAt: number | null;
646
814
  lastProgressAt: number | null;
815
+ dispatchMode: string | null;
816
+ diagStage: string | null;
647
817
  } | null> {
648
818
  await ensureRunTables();
649
819
  const client = getDbExec();
650
820
  const sql = options?.includeTerminal
651
- ? `SELECT id, thread_id, turn_id, status, started_at, heartbeat_at, completed_at, last_progress_at FROM agent_runs WHERE thread_id = ? ORDER BY started_at DESC LIMIT 1`
652
- : `SELECT id, thread_id, turn_id, status, started_at, heartbeat_at, completed_at, last_progress_at FROM agent_runs WHERE thread_id = ? AND status = 'running' ORDER BY started_at DESC LIMIT 1`;
821
+ ? `SELECT id, thread_id, turn_id, status, started_at, heartbeat_at, completed_at, last_progress_at, dispatch_mode, diag_stage FROM agent_runs WHERE thread_id = ? ORDER BY started_at DESC LIMIT 1`
822
+ : `SELECT id, thread_id, turn_id, status, started_at, heartbeat_at, completed_at, last_progress_at, dispatch_mode, diag_stage FROM agent_runs WHERE thread_id = ? AND status = 'running' ORDER BY started_at DESC LIMIT 1`;
653
823
  const { rows } = await client.execute({ sql, args: [threadId] });
654
824
  if (rows.length === 0) return null;
655
825
  const r = rows[0] as {
@@ -661,6 +831,8 @@ export async function getRunByThread(
661
831
  heartbeat_at: number | string | null;
662
832
  completed_at: number | string | null;
663
833
  last_progress_at: number | string | null;
834
+ dispatch_mode?: string | null;
835
+ diag_stage?: string | null;
664
836
  };
665
837
  return {
666
838
  id: r.id,
@@ -672,6 +844,8 @@ export async function getRunByThread(
672
844
  completedAt: r.completed_at == null ? null : Number(r.completed_at),
673
845
  lastProgressAt:
674
846
  r.last_progress_at == null ? null : Number(r.last_progress_at),
847
+ dispatchMode: r.dispatch_mode ?? null,
848
+ diagStage: r.diag_stage ?? null,
675
849
  };
676
850
  }
677
851
 
@@ -686,6 +860,9 @@ export interface AgentRunSummary {
686
860
  lastProgressAt: number | null;
687
861
  errorCode: string | null;
688
862
  abortReason: string | null;
863
+ dispatchMode: string | null;
864
+ /** Last reached `_process-run` worker stage (JSON `{stage,detail?,at}`). */
865
+ diagStage: string | null;
689
866
  }
690
867
 
691
868
  export async function listRunsForThread(
@@ -696,7 +873,7 @@ export async function listRunsForThread(
696
873
  const limit = Math.min(Math.max(options.limit ?? 10, 1), 50);
697
874
  const client = getDbExec();
698
875
  const { rows } = await client.execute({
699
- sql: `SELECT id, thread_id, turn_id, status, started_at, heartbeat_at, completed_at, last_progress_at, error_code, abort_reason
876
+ sql: `SELECT id, thread_id, turn_id, status, started_at, heartbeat_at, completed_at, last_progress_at, error_code, abort_reason, dispatch_mode, diag_stage
700
877
  FROM agent_runs
701
878
  WHERE thread_id = ?
702
879
  ORDER BY started_at DESC
@@ -715,6 +892,8 @@ export async function listRunsForThread(
715
892
  last_progress_at?: number | string | null;
716
893
  error_code?: string | null;
717
894
  abort_reason?: string | null;
895
+ dispatch_mode?: string | null;
896
+ diag_stage?: string | null;
718
897
  };
719
898
  return {
720
899
  id: row.id,
@@ -728,6 +907,8 @@ export async function listRunsForThread(
728
907
  row.last_progress_at == null ? null : Number(row.last_progress_at),
729
908
  errorCode: row.error_code ?? null,
730
909
  abortReason: row.abort_reason ?? null,
910
+ dispatchMode: row.dispatch_mode ?? null,
911
+ diagStage: row.diag_stage ?? null,
731
912
  };
732
913
  });
733
914
  }
@@ -1485,10 +1485,12 @@ export function findInstalledResvgPackages(
1485
1485
  * Reads the same env flag the runtime gate uses
1486
1486
  * (`AGENT_CHAT_DURABLE_BACKGROUND`).
1487
1487
  *
1488
- * DEFAULT-ON, matching the runtime gate (`isFlagEnabled` in
1489
- * durable-background.ts): unset/empty/unknown means enabled; an app opts OUT
1490
- * only with an explicit falsy value (`false`/`0`/`no`/`off`). This is what
1491
- * actually emits the 15-min `-background` function so the `_process-run`
1488
+ * DEFAULT-OFF (opt-in), matching the runtime gate (`isFlagEnabled` in
1489
+ * durable-background.ts): unset/empty/unknown means DISABLED; an app opts IN
1490
+ * only with an explicit truthy value (`true`/`1`/`yes`/`on`). A premature
1491
+ * fleet-wide default-on caused real-user incidents (2026-06-24) before the
1492
+ * async worker path was proven, so durable is opt-in until verified live. This
1493
+ * gate is what emits the 15-min `-background` function so the `_process-run`
1492
1494
  * dispatch lands on it (async 202 → the worker runs with the real 15-min budget
1493
1495
  * → its ~13-min soft-timeout fits). The deploy gate and runtime gate MUST agree:
1494
1496
  * if the deploy emitted no `-background` function but the runtime still routed
@@ -1499,9 +1501,9 @@ export function findInstalledResvgPackages(
1499
1501
  */
1500
1502
  export function isDurableBackgroundDeployEnabled(): boolean {
1501
1503
  const raw = process.env.AGENT_CHAT_DURABLE_BACKGROUND;
1502
- if (raw == null) return true;
1504
+ if (raw == null) return false;
1503
1505
  const v = raw.trim().toLowerCase();
1504
- return !(v === "0" || v === "false" || v === "no" || v === "off");
1506
+ return v === "1" || v === "true" || v === "yes" || v === "on";
1505
1507
  }
1506
1508
 
1507
1509
  /**