@intentface/latch-core 0.9.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +45 -0
  3. package/dist/agent.d.ts +200 -0
  4. package/dist/agent.d.ts.map +1 -0
  5. package/dist/agent.js +9 -0
  6. package/dist/agent.js.map +1 -0
  7. package/dist/compaction.d.ts +33 -0
  8. package/dist/compaction.d.ts.map +1 -0
  9. package/dist/compaction.js +104 -0
  10. package/dist/compaction.js.map +1 -0
  11. package/dist/connections.d.ts +16 -0
  12. package/dist/connections.d.ts.map +1 -0
  13. package/dist/connections.js +41 -0
  14. package/dist/connections.js.map +1 -0
  15. package/dist/context.d.ts +41 -0
  16. package/dist/context.d.ts.map +1 -0
  17. package/dist/context.js +25 -0
  18. package/dist/context.js.map +1 -0
  19. package/dist/current-date.d.ts +12 -0
  20. package/dist/current-date.d.ts.map +1 -0
  21. package/dist/current-date.js +25 -0
  22. package/dist/current-date.js.map +1 -0
  23. package/dist/extensions.d.ts +333 -0
  24. package/dist/extensions.d.ts.map +1 -0
  25. package/dist/extensions.js +569 -0
  26. package/dist/extensions.js.map +1 -0
  27. package/dist/harness/index.d.ts +17 -0
  28. package/dist/harness/index.d.ts.map +1 -0
  29. package/dist/harness/index.js +15 -0
  30. package/dist/harness/index.js.map +1 -0
  31. package/dist/harness/tools.d.ts +88 -0
  32. package/dist/harness/tools.d.ts.map +1 -0
  33. package/dist/harness/tools.js +296 -0
  34. package/dist/harness/tools.js.map +1 -0
  35. package/dist/harness/web-fetch.d.ts +47 -0
  36. package/dist/harness/web-fetch.d.ts.map +1 -0
  37. package/dist/harness/web-fetch.js +247 -0
  38. package/dist/harness/web-fetch.js.map +1 -0
  39. package/dist/index.d.ts +25 -0
  40. package/dist/index.d.ts.map +1 -0
  41. package/dist/index.js +25 -0
  42. package/dist/index.js.map +1 -0
  43. package/dist/limits.d.ts +152 -0
  44. package/dist/limits.d.ts.map +1 -0
  45. package/dist/limits.js +97 -0
  46. package/dist/limits.js.map +1 -0
  47. package/dist/memory.d.ts +93 -0
  48. package/dist/memory.d.ts.map +1 -0
  49. package/dist/memory.js +13 -0
  50. package/dist/memory.js.map +1 -0
  51. package/dist/message.d.ts +46 -0
  52. package/dist/message.d.ts.map +1 -0
  53. package/dist/message.js +2 -0
  54. package/dist/message.js.map +1 -0
  55. package/dist/models/catalog.d.ts +56 -0
  56. package/dist/models/catalog.d.ts.map +1 -0
  57. package/dist/models/catalog.js +211 -0
  58. package/dist/models/catalog.js.map +1 -0
  59. package/dist/models/defaults.d.ts +23 -0
  60. package/dist/models/defaults.d.ts.map +1 -0
  61. package/dist/models/defaults.js +19 -0
  62. package/dist/models/defaults.js.map +1 -0
  63. package/dist/models/index.d.ts +23 -0
  64. package/dist/models/index.d.ts.map +1 -0
  65. package/dist/models/index.js +19 -0
  66. package/dist/models/index.js.map +1 -0
  67. package/dist/models/prompt-caching.d.ts +19 -0
  68. package/dist/models/prompt-caching.d.ts.map +1 -0
  69. package/dist/models/prompt-caching.js +18 -0
  70. package/dist/models/prompt-caching.js.map +1 -0
  71. package/dist/models/provider.d.ts +19 -0
  72. package/dist/models/provider.d.ts.map +1 -0
  73. package/dist/models/provider.js +22 -0
  74. package/dist/models/provider.js.map +1 -0
  75. package/dist/models/reasoning.d.ts +21 -0
  76. package/dist/models/reasoning.d.ts.map +1 -0
  77. package/dist/models/reasoning.js +59 -0
  78. package/dist/models/reasoning.js.map +1 -0
  79. package/dist/pricing.d.ts +52 -0
  80. package/dist/pricing.d.ts.map +1 -0
  81. package/dist/pricing.js +37 -0
  82. package/dist/pricing.js.map +1 -0
  83. package/dist/principal.d.ts +37 -0
  84. package/dist/principal.d.ts.map +1 -0
  85. package/dist/principal.js +30 -0
  86. package/dist/principal.js.map +1 -0
  87. package/dist/projections.d.ts +37 -0
  88. package/dist/projections.d.ts.map +1 -0
  89. package/dist/projections.js +128 -0
  90. package/dist/projections.js.map +1 -0
  91. package/dist/prompt-caching.d.ts +106 -0
  92. package/dist/prompt-caching.d.ts.map +1 -0
  93. package/dist/prompt-caching.js +165 -0
  94. package/dist/prompt-caching.js.map +1 -0
  95. package/dist/runtime.d.ts +670 -0
  96. package/dist/runtime.d.ts.map +1 -0
  97. package/dist/runtime.js +2425 -0
  98. package/dist/runtime.js.map +1 -0
  99. package/dist/scheduler.d.ts +31 -0
  100. package/dist/scheduler.d.ts.map +1 -0
  101. package/dist/scheduler.js +43 -0
  102. package/dist/scheduler.js.map +1 -0
  103. package/dist/storage.d.ts +389 -0
  104. package/dist/storage.d.ts.map +1 -0
  105. package/dist/storage.js +38 -0
  106. package/dist/storage.js.map +1 -0
  107. package/dist/telemetry.d.ts +155 -0
  108. package/dist/telemetry.d.ts.map +1 -0
  109. package/dist/telemetry.js +2 -0
  110. package/dist/telemetry.js.map +1 -0
  111. package/dist/vault-node.d.ts +20 -0
  112. package/dist/vault-node.d.ts.map +1 -0
  113. package/dist/vault-node.js +30 -0
  114. package/dist/vault-node.js.map +1 -0
  115. package/dist/vault.d.ts +62 -0
  116. package/dist/vault.d.ts.map +1 -0
  117. package/dist/vault.js +88 -0
  118. package/dist/vault.js.map +1 -0
  119. package/package.json +95 -0
@@ -0,0 +1,2425 @@
1
+ import { ToolLoopAgent, consumeStream, createIdGenerator, createUIMessageStream, createUIMessageStreamResponse, jsonSchema, readUIMessageStream, smoothStream, stepCountIs, tool, toUIMessageStream, } from "ai";
2
+ import { currentDateLine } from "./current-date.js";
3
+ import { COMPACTION_PART_TYPE, sliceAtCompaction, toClientMessages, toModelMessages, } from "./projections.js";
4
+ import { summarizeForCompaction } from "./compaction.js";
5
+ import { computeCost } from "./pricing.js";
6
+ import { defaultPromptCachingPlan, markLastFunctionTool, mergeProviderOptions, toolsetHash, } from "./prompt-caching.js";
7
+ import { LimitExceededError, exceededPolicy, remainingTokens, windowRef, } from "./limits.js";
8
+ /** Ids for assistant response messages — `msg_<random>`, matching storage ids. */
9
+ const generateMessageId = createIdGenerator({ prefix: "msg", separator: "_" });
10
+ /** Sub-thread chat ids start with `sub_` so the host can keep them out of the
11
+ * main chat list (they're nested under the parent's spawn_agent tool calls). */
12
+ const generateSubchatId = createIdGenerator({ prefix: "sub", separator: "_" });
13
+ /** Parent-level id for a subagent approval bubbled up via `spawn_agent`. */
14
+ const generateSubApprovalId = createIdGenerator({ prefix: "sapv", separator: "_" });
15
+ /** Recursion ceiling for subagent delegation. */
16
+ const MAX_SUBAGENT_DEPTH = 4;
17
+ /**
18
+ * Below this many messages in the live window, compaction isn't worth a model
19
+ * call: one exchange (user + assistant) plus the message asking for it summarizes
20
+ * to roughly what it replaces.
21
+ */
22
+ const MIN_COMPACTABLE_MESSAGES = 4;
23
+ const newInstanceId = createIdGenerator({ prefix: "inst", separator: "_" });
24
+ /** One-line summary of a gated tool call, for the approval dialog. */
25
+ function summarizeApproval(toolName, input) {
26
+ let args = "";
27
+ try {
28
+ args = JSON.stringify(input ?? {});
29
+ }
30
+ catch {
31
+ args = "";
32
+ }
33
+ if (args.length > 200)
34
+ args = `${args.slice(0, 200)}…`;
35
+ return `${toolName}(${args})`;
36
+ }
37
+ /** The undecided (`approved === undefined`) approvals in a thread's messages. */
38
+ function collectPendingApprovals(messages) {
39
+ const out = [];
40
+ for (const m of messages) {
41
+ for (const p of (m.parts ?? [])) {
42
+ if (p.state === "approval-requested" &&
43
+ p.approval &&
44
+ p.approval.approved === undefined) {
45
+ const toolName = p.type.replace(/^tool-/, "");
46
+ out.push({
47
+ approvalId: p.approval.id,
48
+ toolName,
49
+ summary: summarizeApproval(toolName, p.input),
50
+ });
51
+ }
52
+ }
53
+ }
54
+ return out;
55
+ }
56
+ /** The concatenated text of a message's text parts (for a chat title). */
57
+ function firstText(message) {
58
+ const text = (message.parts ?? [])
59
+ .filter((p) => p.type === "text")
60
+ .map((p) => p.text)
61
+ .join(" ")
62
+ .trim();
63
+ return text.length > 0 ? text : undefined;
64
+ }
65
+ /** The text of the turn's incoming user message (for telemetry). */
66
+ /** The runtime's own user-role message: a compaction summary + boundary marker, not a turn. */
67
+ function isCompactionMarker(m) {
68
+ return (m.parts ?? []).some((p) => p.type === COMPACTION_PART_TYPE);
69
+ }
70
+ /**
71
+ * The model that ran THIS run, off the assistant message it produced (each one
72
+ * carries its run id). Falls back to the tail when nothing is correlated —
73
+ * the legacy shape — and to nothing when the run never produced a message.
74
+ */
75
+ function ranAs(messages, runId) {
76
+ for (let i = messages.length - 1; i >= 0; i--) {
77
+ const m = messages[i];
78
+ if (m.role !== "assistant")
79
+ continue;
80
+ const owner = m.metadata?.runId;
81
+ if (owner === runId || owner === undefined)
82
+ return m.metadata?.model;
83
+ }
84
+ return undefined;
85
+ }
86
+ function lastUserText(messages) {
87
+ for (let i = messages.length - 1; i >= 0; i--) {
88
+ const m = messages[i];
89
+ if (m?.role === "user")
90
+ return firstText(m);
91
+ }
92
+ return undefined;
93
+ }
94
+ /**
95
+ * Thrown when work is attempted on a chat whose last turn is still paused
96
+ * awaiting a human — by `compactChat`, where the pause can also be a parked
97
+ * client tool call and there's no sensible way to summarize around it. Carries
98
+ * a stable `code` so the handler can map it to a 409 and the client can prompt
99
+ * the user to resolve it first.
100
+ *
101
+ * `handleChat` deliberately does NOT throw this: a new message supersedes the
102
+ * pause (see `declinePendingApprovals`).
103
+ */
104
+ export class PendingApprovalError extends Error {
105
+ code = "pending_approval";
106
+ constructor(message) {
107
+ super(message);
108
+ this.name = "PendingApprovalError";
109
+ }
110
+ }
111
+ /**
112
+ * Is the chat parked on an undecided approval RIGHT NOW?
113
+ *
114
+ * Only the last assistant message counts. A paused turn's gated calls all live
115
+ * on the assistant message it paused on (a resume starts a new one), so that
116
+ * message is the live pause point. Undecided parts in OLDER messages are
117
+ * abandoned turns — `closeDanglingToolCalls` already fabricates a terminal
118
+ * result for them in the model projection, and stored history is never
119
+ * rewritten. Scanning all of history instead wedged chats permanently in both
120
+ * directions: `handleChat` refused every new message with PendingApprovalError,
121
+ * AND `applyApproval` returned an empty 200 without resuming, so no button
122
+ * press could clear it either.
123
+ */
124
+ function hasPendingApproval(messages) {
125
+ const last = messages[messages.length - 1];
126
+ // Strictly the FINAL message: a later user message means the turn moved on.
127
+ if (last?.role !== "assistant")
128
+ return false;
129
+ for (const p of (last.parts ?? [])) {
130
+ if (p.state === "approval-requested" && p.approval?.approved === undefined) {
131
+ return true;
132
+ }
133
+ // A bubbled-up subagent approval still awaiting a decision.
134
+ if (p.type === "data-subagent-approval" && p.data && p.data.approved === undefined) {
135
+ return true;
136
+ }
137
+ }
138
+ return false;
139
+ }
140
+ /** Why an approval was closed without the user ever deciding it. */
141
+ const SUPERSEDED_REASON = "Superseded — the user sent a new message instead of responding to this request.";
142
+ /**
143
+ * Decline every undecided approval on the paused turn, because the user moved
144
+ * on: they typed a new message instead of pressing a button.
145
+ *
146
+ * Without this a chat could wedge for good. A gate the user CAN'T satisfy — a
147
+ * misconfigured `connect_<name>` whose OAuth never completes is the reported
148
+ * case — left `handleChat` refusing every message with `PendingApprovalError`,
149
+ * so the only escape was a decision UI that (on the web) had no deny button.
150
+ * Declining is always the safe direction: no gated tool runs, and the model
151
+ * sees a denial plus the new instruction, so the user can simply redirect it.
152
+ *
153
+ * Bubbled-up subagent approvals are marked `resolved` too, so
154
+ * `collectDecidedSubagentApprovals` won't later try to resume a sub thread the
155
+ * user has walked away from. Mutates in place; returns the changed messages.
156
+ */
157
+ function declinePendingApprovals(messages) {
158
+ const last = messages[messages.length - 1];
159
+ if (last?.role !== "assistant")
160
+ return [];
161
+ const changed = new Set();
162
+ for (const p of (last.parts ?? [])) {
163
+ if (p.state === "approval-requested" && p.approval && p.approval.approved === undefined) {
164
+ p.state = "approval-responded";
165
+ p.approval = { ...p.approval, approved: false, reason: SUPERSEDED_REASON };
166
+ changed.add(last);
167
+ }
168
+ if (p.type === "data-subagent-approval" && p.data && p.data.approved === undefined) {
169
+ const { spawnToolCallId, subChatId } = p.data;
170
+ p.data = { ...p.data, approved: false, reason: SUPERSEDED_REASON, resolved: true };
171
+ changed.add(last);
172
+ // Close the parent's `spawn_agent` call too. Its stored result is the
173
+ // `awaiting_approval` payload, whose note tells the model to stop and
174
+ // wait for a decision that is never coming — and the model projection
175
+ // drops data parts, so that note is ALL the next turn would see of this.
176
+ if (spawnToolCallId) {
177
+ for (const m of overwriteToolResult(messages, spawnToolCallId, {
178
+ status: "declined",
179
+ threadId: subChatId,
180
+ note: SUPERSEDED_REASON,
181
+ })) {
182
+ changed.add(m);
183
+ }
184
+ }
185
+ }
186
+ }
187
+ return [...changed];
188
+ }
189
+ /**
190
+ * Record decisions onto matching `approval-requested` parts (→ `approval-responded`).
191
+ * Mutates in place and returns the messages that changed (to re-persist).
192
+ */
193
+ function applyDecisions(messages, decisions) {
194
+ const byId = new Map(decisions.map((d) => [d.approvalId, d]));
195
+ const changed = [];
196
+ for (const m of messages) {
197
+ let touched = false;
198
+ for (const p of (m.parts ?? [])) {
199
+ if (p.state === "approval-requested" && p.approval && byId.has(p.approval.id)) {
200
+ const d = byId.get(p.approval.id);
201
+ p.state = "approval-responded";
202
+ p.approval = { ...p.approval, approved: d.approved, reason: d.reason };
203
+ touched = true;
204
+ }
205
+ // A bubbled-up subagent approval — record the decision on the data part;
206
+ // applyApproval routes it to the sub thread.
207
+ if (p.type === "data-subagent-approval" &&
208
+ p.data &&
209
+ byId.has(p.data.approvalId)) {
210
+ const d = byId.get(p.data.approvalId);
211
+ p.data = { ...p.data, approved: d.approved, reason: d.reason };
212
+ touched = true;
213
+ }
214
+ }
215
+ if (touched)
216
+ changed.push(m);
217
+ }
218
+ return changed;
219
+ }
220
+ /** Overwrite a tool call's result (regardless of current state) — used to fold
221
+ * a resumed subagent's answer back into the `spawn_agent` call on the parent. */
222
+ function overwriteToolResult(messages, toolCallId, output) {
223
+ const changed = [];
224
+ for (const m of messages) {
225
+ let touched = false;
226
+ for (const p of (m.parts ?? [])) {
227
+ if (p.toolCallId === toolCallId &&
228
+ (p.state === "output-available" || p.state === "input-available")) {
229
+ p.state = "output-available";
230
+ p.output = output;
231
+ touched = true;
232
+ }
233
+ }
234
+ if (touched)
235
+ changed.push(m);
236
+ }
237
+ return changed;
238
+ }
239
+ /** Decided-but-unresolved subagent approvals to route to their sub threads. */
240
+ function collectDecidedSubagentApprovals(messages) {
241
+ const out = [];
242
+ for (const m of messages) {
243
+ for (const p of (m.parts ?? [])) {
244
+ if (p.type === "data-subagent-approval" &&
245
+ p.data &&
246
+ p.data.approved !== undefined &&
247
+ !p.data.resolved) {
248
+ out.push(p.data);
249
+ }
250
+ }
251
+ }
252
+ return out;
253
+ }
254
+ /**
255
+ * Re-bubble: append a fresh `data-subagent-approval` part next to a resolved
256
+ * one (same parent assistant message), for a resumed subagent that paused on
257
+ * ANOTHER gated tool. The undecided part keeps the parent `awaiting_input`;
258
+ * deciding it round-trips through `applyApproval` again, so bubbling recurses
259
+ * for as many rounds as the sub needs.
260
+ */
261
+ function appendSubagentApprovalPart(messages, afterApprovalId, data) {
262
+ for (const m of messages) {
263
+ const parts = (m.parts ?? []);
264
+ if (parts.some((p) => p.type === "data-subagent-approval" &&
265
+ p.data?.approvalId === afterApprovalId)) {
266
+ parts.push({
267
+ type: "data-subagent-approval",
268
+ id: data.approvalId,
269
+ data,
270
+ });
271
+ m.parts = parts;
272
+ return [m];
273
+ }
274
+ }
275
+ return [];
276
+ }
277
+ /** Mark a routed subagent approval resolved so it isn't re-processed. */
278
+ function markSubagentApprovalResolved(messages, approvalId) {
279
+ const changed = [];
280
+ for (const m of messages) {
281
+ let touched = false;
282
+ for (const p of (m.parts ?? [])) {
283
+ if (p.type === "data-subagent-approval" &&
284
+ p.data &&
285
+ p.data.approvalId === approvalId) {
286
+ p.data = { ...p.data, resolved: true };
287
+ touched = true;
288
+ }
289
+ }
290
+ if (touched)
291
+ changed.push(m);
292
+ }
293
+ return changed;
294
+ }
295
+ /**
296
+ * Has this chat opened another turn since `run` started? Then `run` was
297
+ * superseded — the conversation moved on while it sat reaped — and it must
298
+ * settle rather than run: re-running would generate against a conversation that
299
+ * has moved on and, because the message id is reused, append the result to the
300
+ * later turn's message.
301
+ *
302
+ * Read off the RUNS, not the messages. Every user message is persisted together
303
+ * with its run (`openTurn`), so a later run in the chat is the one fact that
304
+ * proves a later turn exists. Messages cannot tell us this: user messages carry
305
+ * no run id, and when a turn crashes before its first checkpoint the previous
306
+ * turn's assistant message is legitimately the last assistant in history —
307
+ * reading either as "someone else's" marks a live run superseded and drops the
308
+ * user's request without a model call, which is the one failure worse than the
309
+ * bug this all fixes.
310
+ *
311
+ * Compaction runs summarise the conversation so far; they are not turns.
312
+ */
313
+ export function supersededBy(run, runs) {
314
+ return runs.some((r) => r.chatId === run.chatId && r.id !== run.id && r.kind !== "compaction" && r.startedAt > run.startedAt);
315
+ }
316
+ /**
317
+ * What a reclaimed run's OWN history says it needs. Pure: whether the chat has
318
+ * since moved on is `supersededBy`'s question, answered from the runs table.
319
+ */
320
+ export function resumeShapeOf(messages) {
321
+ // Compaction markers are not turns, so the tail is read past them: a summary
322
+ // appended after a reaped run's finished message must not make that message
323
+ // look like it was followed by a user turn (which would say "continue" and
324
+ // regenerate against the summary). `compactChat` refuses while a gate or
325
+ // client tool is pending, so a marker can never sit after an undecided one.
326
+ let end = messages.length;
327
+ while (end > 0 && isCompactionMarker(messages[end - 1]))
328
+ end--;
329
+ const turns = end === messages.length ? messages : messages.slice(0, end);
330
+ // Checked next: a pending gate outranks the tail's shape, and its own tail
331
+ // (narration after the gated call) would otherwise read as `loop-finished`.
332
+ if (hasPendingApproval(turns))
333
+ return "decision-pending";
334
+ const last = turns[turns.length - 1];
335
+ // Nothing of this turn was checkpointed (or a user message came last) — the
336
+ // model has not spoken yet, so run it.
337
+ if (last?.role !== "assistant")
338
+ return "continue";
339
+ const parts = (last.parts ?? []);
340
+ let stepStart = -1;
341
+ for (let i = parts.length - 1; i >= 0; i--) {
342
+ if (parts[i].type === "step-start") {
343
+ stepStart = i;
344
+ break;
345
+ }
346
+ }
347
+ const finalStep = parts.slice(stepStart + 1);
348
+ // A step that opened and produced nothing: the crash landed inside it, so
349
+ // there is unfinished work regardless of what earlier steps did.
350
+ if (finalStep.length === 0)
351
+ return "continue";
352
+ const toolParts = finalStep.filter((p) => p.type.startsWith("tool-") || p.type === "dynamic-tool");
353
+ // A CLIENT-handled tool (no `execute`, e.g. askUser) is the one tool part a
354
+ // checkpoint can hold with no result: the step's other tools ran, this one is
355
+ // for the client to answer, and the loop stops because not every call has an
356
+ // output. `onFinish` records that turn as `completed` and the answer arrives
357
+ // later through `applyToolResult`, which opens its own run — so that is what
358
+ // recovery must record too. Re-running instead would make the projection
359
+ // close the call as "interrupted" and the model ask its question twice.
360
+ // (A provider-executed tool still pending is the loop continuing, not a
361
+ // client wait, so it must not match.)
362
+ const awaitingClient = toolParts.some((p) => p.state === "input-available" && typeof p.toolCallId === "string" && !p.providerExecuted);
363
+ if (awaitingClient)
364
+ return "loop-finished";
365
+ return toolParts.length > 0 ? "continue" : "loop-finished";
366
+ }
367
+ /**
368
+ * The resume paths ({@link applyApproval} / {@link applyToolResult}) re-run the
369
+ * model from history whose LAST message is the assistant turn that paused — no
370
+ * fresh user message is appended. If that turn streamed trailing user-facing
371
+ * text before pausing (e.g. "…ready to create the page but needs your approval
372
+ * first"), the model projection ends on an assistant message. Some providers
373
+ * reject that ("This model does not support assistant message prefill. The
374
+ * conversation must end with a user message.").
375
+ *
376
+ * Drop the trailing text/reasoning that follows the final TOOL part of the last
377
+ * assistant message, so the projection ends on the tool result (a user turn) and
378
+ * the model generates a fresh continuation. Returns a shallow copy touching only
379
+ * the last message.
380
+ *
381
+ * **The anchor must be a part the model projection keeps.** `toModelMessages`
382
+ * drops every `data-*` part, so anchoring on a bubbled `data-subagent-approval`
383
+ * kept the text BEFORE it — leaving the projection ending on assistant text,
384
+ * i.e. exactly the prefill this exists to prevent. That is the shape the
385
+ * agent-builder produces on every approval round: `… tool-spawn_agent,
386
+ * data-subagent-progress, data-subagent-approval, step-start, text,
387
+ * data-subagent-approval`. Only `tool-*` / `dynamic-tool` anchor.
388
+ *
389
+ * `data-*` parts after the anchor are KEPT: the resumed turn's `onFinish`
390
+ * re-persists this message, so removing them would erase the approval markers
391
+ * the UI renders and `hasPendingApproval` reads. Trailing `step-start` is
392
+ * DROPPED along with the text: it is invisible content, but it instructs
393
+ * `convertToModelMessages` to open a NEW assistant message — and with the text
394
+ * gone, that message is empty, so the projection would still end on the
395
+ * assistant (`{ role: "assistant", content: [] }`), trading the prefill
396
+ * rejection for an empty-content one. The only thing in that step was the
397
+ * narration being removed, so history loses nothing but an empty divider.
398
+ * The trailing text is dropped from stored history too — it is provisional
399
+ * commentary about a decision that has now been made; the resumed turn
400
+ * regenerates the real continuation.
401
+ */
402
+ export function trimTrailingAssistantPrefill(messages, opts) {
403
+ if (messages.length === 0)
404
+ return messages;
405
+ const lastIndex = messages.length - 1;
406
+ const last = messages[lastIndex];
407
+ if (last.role !== "assistant")
408
+ return messages;
409
+ const parts = (last.parts ?? []);
410
+ // Last part that survives into the model projection as a turn boundary.
411
+ let anchor = -1;
412
+ for (let i = parts.length - 1; i >= 0; i--) {
413
+ const t = parts[i].type;
414
+ if (t.startsWith("tool-") || t === "dynamic-tool") {
415
+ anchor = i;
416
+ break;
417
+ }
418
+ }
419
+ // No tool anchor at all. On the DECISION paths that shape is not theirs to
420
+ // judge, so leave it (a bubbled `data-subagent-approval` message carries its
421
+ // narration and no `tool-*` part; trimming there would silently lose it).
422
+ //
423
+ // `unanchored` is belt and braces for CRASH RECOVERY, which cannot afford to
424
+ // hand the model a prefill under any tail shape: with no anchor every prefill
425
+ // part is dropped, which empties the message out of the model projection
426
+ // entirely. Today `resumeShapeOf` settles the one anchorless tail a
427
+ // checkpoint can actually produce (`[step-start, text]` — a tool-free turn,
428
+ // which is `loop-finished`) before the trim is reached, so this branch is a
429
+ // guard against a future classifier change, not a live path.
430
+ if (anchor === -1 && !opts?.unanchored)
431
+ return messages;
432
+ // step-start joins text/reasoning: see the doc comment — keeping it ends the
433
+ // projection on an EMPTY assistant message, which providers also reject.
434
+ const isPrefill = (t) => t === "text" || t === "reasoning" || t === "step-start";
435
+ const tail = parts.slice(anchor + 1);
436
+ if (!tail.some((p) => isPrefill(p.type)))
437
+ return messages;
438
+ const kept = [...parts.slice(0, anchor + 1), ...tail.filter((p) => !isPrefill(p.type))];
439
+ const trimmed = { ...last, parts: kept };
440
+ return [...messages.slice(0, lastIndex), trimmed];
441
+ }
442
+ /**
443
+ * True if any tool call is still awaiting a (client-supplied) result.
444
+ *
445
+ * Callers must pass only the messages of the step being resolved (see
446
+ * `applyToolResult`) — NOT full history. A client tool call abandoned turns ago
447
+ * stays `input-available` in stored history forever (`closeDanglingToolCalls`
448
+ * repairs the model projection, never the store), so an all-history scan would
449
+ * make every later resume in that chat return "still waiting" and never re-run.
450
+ */
451
+ function hasPendingToolResult(messages) {
452
+ for (const m of messages) {
453
+ for (const p of (m.parts ?? [])) {
454
+ if (p.state === "input-available" && typeof p.toolCallId === "string")
455
+ return true;
456
+ }
457
+ }
458
+ return false;
459
+ }
460
+ /**
461
+ * Record a tool `output` onto the matching tool call (input-available →
462
+ * output-available). Mutates in place and returns the messages that changed.
463
+ */
464
+ function recordToolResult(messages, toolCallId, output) {
465
+ // Only the live pause point may be answered. A parked client tool call always
466
+ // sits on the assistant message the turn paused on (a resume starts a new
467
+ // one), so anything older was abandoned and buried by a later user turn.
468
+ // Recording onto it would mutate dead history AND resume an obsolete turn —
469
+ // a stale button, or a redelivered submit for a long-dead question, would
470
+ // trigger a spurious agent run instead of doing nothing.
471
+ const last = messages[messages.length - 1];
472
+ if (last?.role !== "assistant")
473
+ return [];
474
+ let touched = false;
475
+ for (const p of (last.parts ?? [])) {
476
+ if (p.toolCallId === toolCallId && p.state === "input-available") {
477
+ p.state = "output-available";
478
+ p.output = output;
479
+ touched = true;
480
+ }
481
+ }
482
+ return touched ? [last] : [];
483
+ }
484
+ /** Run a hook fire-and-forget — a thrown/rejected hook is swallowed so
485
+ * telemetry never breaks a turn. */
486
+ function fireAndForget(fn) {
487
+ if (!fn)
488
+ return;
489
+ try {
490
+ void Promise.resolve(fn()).catch(() => { });
491
+ }
492
+ catch {
493
+ // synchronous throw from the hook — ignore
494
+ }
495
+ }
496
+ /**
497
+ * The one place an error becomes a client-visible string. Latch deliberately
498
+ * forwards `Error` messages — they are the debugging surface for sandbox and
499
+ * tool failures, and the audience is the org's own authenticated users — but
500
+ * every stream/tool error path must route through here so the exposure policy
501
+ * can be tightened in a single place. Non-`Error` throws and unbounded
502
+ * payloads never pass through verbatim.
503
+ */
504
+ /** Keys whose values are masked when a non-`Error` throw is serialized. */
505
+ const SECRETISH_KEY = /token|secret|password|authorization|cookie|api[-_]?key|credential/i;
506
+ export function clientErrorMessage(error) {
507
+ const message = error instanceof Error ? error.message.trim() : "";
508
+ if (message)
509
+ return message.length > 500 ? `${message.slice(0, 499)}…` : message;
510
+ // A non-Error throw or an empty-message Error still carries signal — a raw
511
+ // HTTP body, a JSON-RPC error object, a string. Describe it instead of
512
+ // masking: the bare mask turned an MCP server's `404 {"error":"Session not
513
+ // found"}` into an undebuggable generic string. Serialization masks
514
+ // secret-shaped fields (a thrown response object can embed headers or
515
+ // tokens) and stays bounded; JSON.stringify can throw on cycles, hence the
516
+ // try. Exported so hosts render connection failures through the SAME
517
+ // exposure policy instead of reinventing (String(err) → "[object Object]").
518
+ if (error instanceof Error) {
519
+ const name = error.name?.trim();
520
+ if (name && name !== "Error")
521
+ return `${name.slice(0, 100)} (no message)`;
522
+ }
523
+ try {
524
+ const desc = typeof error === "string"
525
+ ? error
526
+ : JSON.stringify(error, (key, value) => key && SECRETISH_KEY.test(key) ? "[redacted]" : value);
527
+ if (desc && desc !== "{}" && desc !== "null" && desc !== '""' && desc !== "undefined") {
528
+ return `An error occurred: ${desc.length > 300 ? `${desc.slice(0, 299)}…` : desc}`;
529
+ }
530
+ }
531
+ catch {
532
+ // circular or non-serializable — fall through to the mask
533
+ }
534
+ return "An error occurred.";
535
+ }
536
+ export function createRuntime(config) {
537
+ const { storage } = config;
538
+ const reconstructPrincipal = config.reconstructPrincipal ?? ((identity) => identity);
539
+ const dur = {
540
+ enabled: config.durability?.enabled ?? false,
541
+ leaseTtlMs: config.durability?.leaseTtlMs ?? 30_000,
542
+ heartbeatMs: config.durability?.heartbeatMs ?? 10_000,
543
+ instanceId: config.durability?.instanceId ?? newInstanceId(),
544
+ };
545
+ if (dur.enabled && !storage.openTurn) {
546
+ // Without `openTurn` a turn opens as two writes (run, then message), and a
547
+ // crash between them leaves a run whose own message never landed. Recovery
548
+ // then reads the PREVIOUS turn's tail as this run's history. Run
549
+ // correlation catches the common shape, but the contract is explicit that
550
+ // a durable adapter should be atomic here — every shipped adapter is.
551
+ console.warn("[latch] durability is enabled but the storage adapter has no openTurn(): " +
552
+ "turn creation is not atomic, so a crash can leave a run without its message. " +
553
+ "Implement openTurn — see the StorageAdapter contract.");
554
+ }
555
+ /** Resolve an agent's per-request config + a display model id. */
556
+ async function resolveAgentConfig(agentName, principal, request, turnContext) {
557
+ const runtimeCtx = await config.context.build({ principal, request, turnContext });
558
+ // Code-declared agents first; otherwise a dynamic (e.g. DB-stored) agent.
559
+ const factory = config.agents[agentName];
560
+ const cfg = factory
561
+ ? await factory({ context: runtimeCtx, principal, turnContext })
562
+ : await config.dynamicAgents?.resolve(agentName, principal);
563
+ if (!cfg)
564
+ throw new Error(`Unknown agent: ${agentName}`);
565
+ const modelId = typeof cfg.model === "string"
566
+ ? cfg.model
567
+ : (cfg.model.modelId ?? "unknown");
568
+ return { cfg, modelId, runtimeCtx };
569
+ }
570
+ /**
571
+ * Build (and start) one turn's stream — shared by a fresh chat turn and a
572
+ * resume. With durability on, it heartbeats the lease and aborts if it's
573
+ * lost; onFinish writes are fenced by `lease`. The caller decides whether to
574
+ * hand the stream to a client (handleChat) or drive it to completion
575
+ * server-side (resume).
576
+ */
577
+ async function buildTurn(args) {
578
+ const { resolvedChatId, principal, modelId, cfg, run, lease, uiMessages, runtimeCtx } = args;
579
+ const admitted = args.admitted;
580
+ const autoApprove = args.autoApprove ?? false;
581
+ const blockGated = args.blockGated ?? false;
582
+ const depth = args.depth ?? 0;
583
+ const trigger = args.trigger ?? "chat";
584
+ // The top-level conversation id — this turn's own chat unless a spawn
585
+ // threaded a root down. Subagent turns spawned below inherit it, so a whole
586
+ // delegation tree shares one telemetry session (see telemetryMeta).
587
+ const rootChatId = args.rootChatId ?? resolvedChatId;
588
+ const abort = new AbortController();
589
+ let heartbeat;
590
+ const stopHeartbeat = () => {
591
+ if (heartbeat !== undefined) {
592
+ clearInterval(heartbeat);
593
+ heartbeat = undefined;
594
+ }
595
+ };
596
+ if (dur.enabled && lease) {
597
+ // A THROWN heartbeat write means "we don't know if we still hold the
598
+ // lease" — very different from `held === false` ("someone else owns this
599
+ // run"). Swallowing it silently let a run whose beats were all failing
600
+ // keep streaming at full token cost while the reaper reclaimed it and a
601
+ // second worker resumed the same turn. So: retry transient failures, but
602
+ // once consecutive failures span a full TTL the lease has provably
603
+ // expired — abort, exactly as if `held` had come back false.
604
+ let missedBeats = 0;
605
+ let beatInFlight = false;
606
+ const maxMissedBeats = Math.max(1, Math.ceil(dur.leaseTtlMs / dur.heartbeatMs));
607
+ const missBeat = (why, e) => {
608
+ missedBeats++;
609
+ console.warn(`[durability] heartbeat ${why} for run ${run.id} (${missedBeats}/${maxMissedBeats})`, e ?? "");
610
+ if (missedBeats >= maxMissedBeats) {
611
+ stopHeartbeat();
612
+ abort.abort(new Error("lease lost: heartbeats failing for a full TTL"));
613
+ }
614
+ };
615
+ heartbeat = setInterval(() => {
616
+ // A write that never settles is as blind as one that throws — count it
617
+ // against the TTL budget instead of stacking another write on top.
618
+ if (beatInFlight) {
619
+ missBeat("write still unsettled from the previous interval");
620
+ return;
621
+ }
622
+ beatInFlight = true;
623
+ void storage
624
+ .heartbeatRun(run.id, lease.owner, dur.leaseTtlMs, lease.fencingToken)
625
+ .then((held) => {
626
+ missedBeats = 0;
627
+ if (!held) {
628
+ stopHeartbeat();
629
+ abort.abort(new Error("lease lost"));
630
+ }
631
+ })
632
+ .catch((e) => missBeat("write failed", e))
633
+ .finally(() => {
634
+ beatInFlight = false;
635
+ });
636
+ }, dur.heartbeatMs);
637
+ // Don't let the heartbeat alone keep a Node process alive.
638
+ heartbeat.unref?.();
639
+ }
640
+ // Open dynamic tool sources (e.g. MCP) for this tenant, then merge their
641
+ // tools with the static ones. They're closed when the turn ends. If opening
642
+ // one fails, close any already opened so we don't leak connections.
643
+ // Effective tool sources: the agent's own, plus a registry-resolved source
644
+ // for its declared `connections` (its MCP servers / integrations).
645
+ const sources = [...(cfg.toolSources ?? [])];
646
+ if (config.connections && cfg.connections && cfg.connections.length > 0) {
647
+ sources.push(config.connections.hostFor(cfg.connections));
648
+ }
649
+ const opened = [];
650
+ try {
651
+ for (const source of sources) {
652
+ opened.push(await source.open(principal));
653
+ }
654
+ }
655
+ catch (error) {
656
+ await Promise.all(opened.map((o) => o.close().catch(() => { })));
657
+ stopHeartbeat();
658
+ throw error;
659
+ }
660
+ const closeSources = () => Promise.all(opened.map((o) => o.close().catch(() => { }))).then(() => undefined);
661
+ const tools = Object.assign({}, cfg.tools, ...opened.map((o) => o.tools));
662
+ // Default harness tools (eve-mirrored), merged per the agent's flags. The
663
+ // platform supplies them (core carries no sandbox/QuickJS dep): `sandbox`
664
+ // tools (bash/read_file/write_file/glob/grep) operate on the resolved
665
+ // Experimental_SandboxSession; `app` tools (web_fetch/todo/ask_question) run
666
+ // in the app process. `disableTools` opts any of them back out.
667
+ // Names that must stay callable even when `enabledTools` restricts the set
668
+ // (it only restricts *connection* tools): the agent's own declared tools,
669
+ // the harness tools, and `spawn_agent` are capabilities, not connection picks.
670
+ const capabilityTools = Object.keys(cfg.tools ?? {});
671
+ const harness = config.harnessTools ?? {};
672
+ // Gate defaults for the harness tools this agent actually gets (a policy
673
+ // naming an unattached tool would be dead weight in the merge below).
674
+ const harnessApproval = {};
675
+ const attachHarness = (name, t) => {
676
+ tools[name] = t;
677
+ capabilityTools.push(name);
678
+ const gate = harness.approval?.[name];
679
+ if (gate)
680
+ harnessApproval[name] = gate;
681
+ };
682
+ const wantSandbox = !!cfg.sandbox && !!config.sandbox;
683
+ if (wantSandbox && harness.sandbox) {
684
+ for (const [name, t] of Object.entries(harness.sandbox))
685
+ attachHarness(name, t);
686
+ }
687
+ if (cfg.defaultTools && harness.app) {
688
+ let appTools;
689
+ if (typeof harness.app === "function") {
690
+ try {
691
+ appTools = await harness.app({ principal, chatId: resolvedChatId, agent: run.agent });
692
+ }
693
+ catch (error) {
694
+ // Same teardown as a failing tool-source open above: the sources are
695
+ // already open and the lease heartbeat is running — release both
696
+ // before surfacing the factory error.
697
+ await closeSources();
698
+ stopHeartbeat();
699
+ throw error;
700
+ }
701
+ }
702
+ else {
703
+ appTools = harness.app;
704
+ }
705
+ const only = Array.isArray(cfg.defaultTools) ? new Set(cfg.defaultTools) : null;
706
+ for (const [name, t] of Object.entries(appTools)) {
707
+ if (!only || only.has(name))
708
+ attachHarness(name, t);
709
+ }
710
+ }
711
+ // Render tools — a separate capability (like sandbox), opted into via
712
+ // `renderTools: true`, never granted by `defaultTools`.
713
+ if (cfg.renderTools && harness.render) {
714
+ for (const [name, t] of Object.entries(harness.render))
715
+ attachHarness(name, t);
716
+ }
717
+ // Provider-native web tools (server-side search/fetch), if the agent opts in
718
+ // and the platform supplies a resolver for this model's provider.
719
+ if (cfg.providerWebTools && config.resolveProviderTools) {
720
+ for (const [name, t] of Object.entries(config.resolveProviderTools(modelId))) {
721
+ tools[name] = t;
722
+ capabilityTools.push(name);
723
+ }
724
+ }
725
+ // Memory seam: when the agent declares `memory` and the runtime has a
726
+ // provider, merge the scope-bound memory tools (capabilities, not
727
+ // connection picks — they must survive an `enabledTools` restriction) and
728
+ // append the provider's compiled-index block to the instructions. A
729
+ // provider failure degrades to a turn without memory, never a dead turn.
730
+ let instructions = cfg.instructions;
731
+ if (cfg.memory && config.memory) {
732
+ const memArgs = { principal, agent: run.agent, memory: cfg.memory };
733
+ try {
734
+ // Resolve both hooks BEFORE mutating the toolset, so a failure in
735
+ // either leaves the turn exactly as it was without memory.
736
+ const memoryTools = Object.entries(config.memory.toolsFor(memArgs));
737
+ const block = await config.memory.instructionsFor(memArgs);
738
+ for (const [name, t] of memoryTools) {
739
+ tools[name] = t;
740
+ capabilityTools.push(name);
741
+ }
742
+ if (block)
743
+ instructions = [instructions, block].filter(Boolean).join("\n\n");
744
+ }
745
+ catch {
746
+ // Memory is optional and degradable — continue without it.
747
+ }
748
+ }
749
+ for (const name of cfg.disableTools ?? [])
750
+ delete tools[name];
751
+ // Per-agent tool description overrides — swap the resolved tool's
752
+ // description so the model sees the agent's custom guidance.
753
+ if (cfg.toolDescriptions) {
754
+ for (const [name, description] of Object.entries(cfg.toolDescriptions)) {
755
+ const t = tools[name];
756
+ if (t && description)
757
+ tools[name] = { ...t, description };
758
+ }
759
+ }
760
+ // Subagent delegation: one `spawn_agent` tool that runs another agent and
761
+ // returns its final answer. Skipped past the recursion ceiling.
762
+ if (cfg.subagents?.length && depth < MAX_SUBAGENT_DEPTH) {
763
+ const allowed = new Set(cfg.subagents);
764
+ const infos = await agentInfos(principal);
765
+ const lines = cfg.subagents
766
+ .map((n) => {
767
+ const i = infos.find((a) => a.name === n);
768
+ return `- ${n}: ${i?.description ?? i?.title ?? "(no description)"}`;
769
+ })
770
+ .join("\n");
771
+ tools.spawn_agent = tool({
772
+ description: "Delegate a sub-task to one of your agents and get its final answer.\n" +
773
+ `Available agents:\n${lines}\n\n` +
774
+ "Omit threadId to start a new conversation; pass the threadId returned by a " +
775
+ "previous call to continue with the SAME agent (it keeps full context). " +
776
+ "Returns { threadId, answer }.",
777
+ inputSchema: jsonSchema({
778
+ type: "object",
779
+ properties: {
780
+ agent: { type: "string", description: "One of the available agent names above." },
781
+ prompt: { type: "string", description: "The task / message for the subagent." },
782
+ threadId: { type: "string", description: "Continue an existing subagent thread." },
783
+ },
784
+ required: ["agent", "prompt"],
785
+ additionalProperties: false,
786
+ }),
787
+ execute: async ({ agent, prompt, threadId }, options) => {
788
+ if (!allowed.has(agent))
789
+ return { error: `Not an allowed subagent: ${agent}` };
790
+ // Stream the delegated conversation live into the parent turn (one
791
+ // data part, reconciled in place by id). Pre-generate the sub-thread
792
+ // id so the very first write already carries it.
793
+ const writer = options.context?.writer;
794
+ // `||` (not `??`): strict-schema models fill optional params with "" — an
795
+ // empty threadId must mean "new thread", never become a real chat id.
796
+ const subChatId = threadId?.trim() || generateSubchatId();
797
+ let onProgress;
798
+ if (writer) {
799
+ const write = (messages) => {
800
+ writer.write({
801
+ type: "data-subagent-progress",
802
+ id: `subagent_${options.toolCallId ?? subChatId}`,
803
+ data: {
804
+ toolCallId: options.toolCallId,
805
+ agent,
806
+ threadId: subChatId,
807
+ messages,
808
+ },
809
+ });
810
+ };
811
+ write([]);
812
+ // Each write re-sends the whole snapshot — throttle the re-sends.
813
+ // The trailing tokens are covered by the tool RESULT (the client
814
+ // switches to the persisted thread once output lands).
815
+ let lastWrite = 0;
816
+ onProgress = (messages) => {
817
+ const now = Date.now();
818
+ if (now - lastWrite < 250)
819
+ return;
820
+ lastWrite = now;
821
+ write(messages);
822
+ };
823
+ }
824
+ // Propagate the PARENT turn's execution mode AS A SET: interactive →
825
+ // the subagent gates its tools (a pending write bubbles up here);
826
+ // autonomous → it auto-approves; a TEST run (blockGated) stays a test
827
+ // run all the way down, or a stubbed parent could delegate to a child
828
+ // whose gated tools run for real.
829
+ const result = await runToCompletion({
830
+ principal,
831
+ agent,
832
+ message: userMessageOf(prompt),
833
+ chatId: subChatId,
834
+ depth: depth + 1,
835
+ autoApprove,
836
+ blockGated,
837
+ onProgress,
838
+ // Telemetry linkage: this turn is the parent; carry the root chat
839
+ // down so the subagent's trace nests under this conversation.
840
+ parentRunId: run.id,
841
+ rootChatId,
842
+ // A subagent inherits what set the tree off, so a scheduled fire's
843
+ // delegate is still attributable to the schedule (see TurnTrigger).
844
+ trigger,
845
+ });
846
+ if (result.pending?.length) {
847
+ // The subagent paused on a gated tool. Surface it on THIS turn as a
848
+ // `data-subagent-approval` part (tagged with the sub thread + this
849
+ // spawn_agent call) so the turn ends `awaiting_input` and
850
+ // `applyApproval` can route the decision back → resume the sub →
851
+ // resume here. Raw sub output stays in the sub thread (isolation).
852
+ const approvalId = generateSubApprovalId();
853
+ writer?.write({
854
+ type: "data-subagent-approval",
855
+ id: approvalId,
856
+ data: {
857
+ approvalId,
858
+ subChatId: result.chatId,
859
+ subApprovalIds: result.pending.map((p) => p.approvalId),
860
+ spawnToolCallId: options.toolCallId,
861
+ summaries: result.pending.map((p) => p.summary),
862
+ // Persist the sub-turn's telemetry linkage so its resumed turn
863
+ // (after approval) stays nested under this conversation/run.
864
+ parentRunId: run.id,
865
+ rootChatId,
866
+ },
867
+ });
868
+ return {
869
+ status: "awaiting_approval",
870
+ threadId: result.chatId,
871
+ pending: result.pending.map((p) => p.summary),
872
+ note: "Awaiting the user's approval — do not retry; stop here, the user will decide and you'll continue automatically.",
873
+ };
874
+ }
875
+ return { threadId: result.chatId, answer: result.answer };
876
+ },
877
+ });
878
+ capabilityTools.push("spawn_agent");
879
+ }
880
+ // Subagent runs are autonomous — there's no human to authorize, so waive
881
+ // approvals and drop the OAuth `connect_` gates (they'd pause forever).
882
+ if (autoApprove) {
883
+ for (const k of Object.keys(tools))
884
+ if (k.startsWith("connect_"))
885
+ delete tools[k];
886
+ // Test runs: stub the approval-gated (mutating) tools so a trial has no
887
+ // real side effects. Gated set = connection defaults + the agent's policy.
888
+ if (blockGated) {
889
+ const gated = new Set();
890
+ for (const [k, v] of Object.entries(harnessApproval))
891
+ if (v)
892
+ gated.add(k);
893
+ for (const o of opened) {
894
+ for (const [k, v] of Object.entries(o.toolApproval ?? {}))
895
+ if (v)
896
+ gated.add(k);
897
+ }
898
+ if (cfg.toolApproval && typeof cfg.toolApproval !== "function") {
899
+ for (const [k, v] of Object.entries(cfg.toolApproval)) {
900
+ if (v)
901
+ gated.add(k);
902
+ else
903
+ gated.delete(k);
904
+ }
905
+ }
906
+ else if (typeof cfg.toolApproval === "function") {
907
+ // A function policy can't be enumerated — it may gate ANY tool. To keep
908
+ // the test side-effect-free, conservatively block every executable tool
909
+ // (connection, authored, and harness — `connect_` is already dropped).
910
+ for (const k of Object.keys(tools))
911
+ gated.add(k);
912
+ }
913
+ for (const name of gated) {
914
+ const t = tools[name];
915
+ if (t && typeof t.execute === "function") {
916
+ tools[name] = {
917
+ ...t,
918
+ execute: async (input) => ({
919
+ __blocked: true,
920
+ tool: name,
921
+ input,
922
+ reason: "approval-gated tool blocked during test run (no real side effects)",
923
+ }),
924
+ };
925
+ }
926
+ }
927
+ }
928
+ }
929
+ // Merge the default approval policies — harness-tool gates and the
930
+ // source-contributed ones (connection defaults + synthetic `connect_<name>`
931
+ // gates) — with the agent's. The AGENT WINS for tools it names, so an agent
932
+ // can require approval on a GET or waive it on a mutating op; tools it
933
+ // doesn't mention keep the default. Only for the object form (a function
934
+ // policy is left as-is).
935
+ const sourceApproval = Object.assign({}, harnessApproval, ...opened.map((o) => o.toolApproval ?? {}));
936
+ const toolApproval = autoApprove
937
+ ? undefined
938
+ : cfg.toolApproval && typeof cfg.toolApproval !== "function"
939
+ ? { ...sourceApproval, ...cfg.toolApproval }
940
+ : Object.keys(sourceApproval).length > 0 && !cfg.toolApproval
941
+ ? sourceApproval
942
+ : cfg.toolApproval;
943
+ // Which tools the model may call. `enabledTools` restricts only CONNECTION
944
+ // tools, so union in each source's `alwaysActive` (e.g. connect_<name>) AND
945
+ // the capability tools (the agent's own tools, harness web_fetch/sandbox,
946
+ // spawn_agent) — otherwise picking specific connection tools would hide the
947
+ // harness. Omitted → all tools active.
948
+ const alwaysActive = opened.flatMap((o) => o.alwaysActive ?? []);
949
+ const activeTools = cfg.enabledTools
950
+ ? // Exclude capability names that `disableTools` removed from `tools` (they
951
+ // were pushed to capabilityTools before the delete) — never list a tool
952
+ // in activeTools that isn't in the resolved toolset.
953
+ Array.from(new Set([...cfg.enabledTools, ...alwaysActive, ...capabilityTools])).filter((n) => n in tools)
954
+ : undefined;
955
+ // Tool-loop ceiling: a number caps it; "unlimited" runs until the model
956
+ // stops on its own; omitted → SDK default (stepCountIs(20)).
957
+ const stepCap = cfg.maxSteps === "unlimited"
958
+ ? stepCountIs(Number.MAX_SAFE_INTEGER)
959
+ : typeof cfg.maxSteps === "number"
960
+ ? stepCountIs(cfg.maxSteps)
961
+ : undefined;
962
+ // Mid-turn limit: stop the tool loop once THIS turn's tokens reach the
963
+ // tightest remaining budget it was admitted with. The model's last text is
964
+ // kept; no further step runs. (Counters were read at admission — concurrent
965
+ // turns can overshoot by at most one turn each, which settlement records.)
966
+ const tokenBudget = admitted?.remainingTokens;
967
+ const budgetStop = tokenBudget !== undefined
968
+ ? ({ steps }) => steps.reduce((n, st) => n + (st.usage?.inputTokens ?? 0) + (st.usage?.outputTokens ?? 0), 0) >=
969
+ tokenBudget
970
+ : undefined;
971
+ const stopWhen = budgetStop ? [stepCap ?? stepCountIs(20), budgetStop] : stepCap;
972
+ // Reasoning/thinking effort → provider-specific options (the platform maps
973
+ // it per provider + gates to reasoning-capable models). Called even without
974
+ // an effort so the hook can set safe defaults (e.g. lift a provider's tiny
975
+ // default max_tokens for models that think by default); an absent/unknown
976
+ // effort must be a thinking-config no-op in the hook.
977
+ const reasoning = config.resolveReasoningOptions
978
+ ? config.resolveReasoningOptions(modelId, cfg.effort ?? "")
979
+ : undefined;
980
+ // Prompt-caching plan for this model's provider (platform hook — see
981
+ // `prompt-caching.ts` for the breakpoint layout). `promptCaching: false`
982
+ // (e.g. the control arm of an A/B comparison) skips it for this turn only.
983
+ const caching = args.promptCaching === false
984
+ ? undefined
985
+ : config.resolvePromptCaching
986
+ ? config.resolvePromptCaching(modelId, {
987
+ agent: run.agent,
988
+ memoryScoped: !!(cfg.memory && config.memory),
989
+ principal,
990
+ })
991
+ : // No hook → caching is ON by default for first-party Anthropic /
992
+ // OpenAI models (derived from the model object's `provider`).
993
+ // Measured at ~6x unit cost on MCP-heavy agents — too expensive to
994
+ // leave off for hosts that never found the hook. Opt out with
995
+ // `resolvePromptCaching: () => undefined`.
996
+ defaultPromptCachingPlan(cfg.model, run.agent);
997
+ if (caching?.toolProviderOptions) {
998
+ markLastFunctionTool(tools, caching.toolProviderOptions, activeTools);
999
+ }
1000
+ // Fingerprint the toolset the model will actually see — persisted on the
1001
+ // run so churn across a chat's turns (a cache invalidator) is queryable.
1002
+ const toolsHash = toolsetHash(activeTools ?? Object.keys(tools));
1003
+ const providerOptions = mergeProviderOptions(reasoning?.providerOptions, caching?.requestProviderOptions);
1004
+ // Telemetry seam: per-run AI SDK telemetry options + trace correlation
1005
+ // identity, resolved once per turn (undefined → nothing traced).
1006
+ const telemetryMeta = {
1007
+ runId: run.id,
1008
+ chatId: resolvedChatId,
1009
+ agent: run.agent,
1010
+ modelId,
1011
+ principal,
1012
+ startedAt: run.startedAt,
1013
+ depth,
1014
+ trigger,
1015
+ parentRunId: args.parentRunId,
1016
+ // Top-level turns are their own root; subagent turns inherit the root the
1017
+ // spawn threaded down, so a whole delegation tree shares one session id.
1018
+ rootChatId,
1019
+ };
1020
+ const telemetryOptions = config.telemetry?.telemetryFor?.(telemetryMeta);
1021
+ // A telemetry impl that defines `telemetryFor` but returns undefined is
1022
+ // opting THIS run out — so the span wrapper and finish/flush hooks must go
1023
+ // silent too, not just the AI SDK settings. If `telemetryFor` is absent
1024
+ // entirely, the impl simply isn't customizing settings, so tracing stays on.
1025
+ const telemetryOn = !!config.telemetry &&
1026
+ !(config.telemetry.telemetryFor && telemetryOptions === undefined);
1027
+ const agent = new ToolLoopAgent({
1028
+ model: cfg.model,
1029
+ // Every agent gets a trailing "today's date" line — after the memory
1030
+ // block, so it stays the tail. Day-granularity by design — see
1031
+ // `currentDateLine` for the prompt-cache reasoning. With a caching plan,
1032
+ // the instructions become a MARKED system message and the date line a
1033
+ // separate unmarked one, so the daily flip lands after the breakpoint
1034
+ // instead of invalidating it at midnight UTC.
1035
+ instructions: caching?.systemProviderOptions
1036
+ ? [
1037
+ ...(instructions
1038
+ ? [
1039
+ {
1040
+ role: "system",
1041
+ content: instructions,
1042
+ providerOptions: caching.systemProviderOptions,
1043
+ },
1044
+ ]
1045
+ : []),
1046
+ { role: "system", content: currentDateLine() },
1047
+ ]
1048
+ : [instructions, currentDateLine()].filter(Boolean).join("\n\n"),
1049
+ tools,
1050
+ toolApproval,
1051
+ stopWhen,
1052
+ activeTools,
1053
+ ...(providerOptions ? { providerOptions } : {}),
1054
+ ...(reasoning?.maxOutputTokens ? { maxOutputTokens: reasoning.maxOutputTokens } : {}),
1055
+ ...(telemetryOptions ? { telemetry: telemetryOptions } : {}),
1056
+ });
1057
+ let inputTokens = 0;
1058
+ let outputTokens = 0;
1059
+ let cacheReadTokens = 0;
1060
+ let cacheWriteTokens = 0;
1061
+ // Set by the FIRST onError, consumed by onFinish (which the SDK still runs
1062
+ // after a stream error, and may precede with more than one onError). It is
1063
+ // what keeps the turn's terminal state single and ordered: the run row is
1064
+ // settled as errored FIRST, and exactly one errored report follows — only
1065
+ // from the writer whose fenced update applied.
1066
+ let failed;
1067
+ // Step checkpoints commit IN STEP ORDER. Each write is still fire-and-forget
1068
+ // for the stream, but chained behind the previous one, so a slow early
1069
+ // write can't commit after a later one and roll the row back to an older
1070
+ // step — and `onFinish` awaits the chain before its final write, so the
1071
+ // finished message is strictly the LAST write of the turn. Without that the
1072
+ // final step's checkpoint races `onFinish` for the same row: the parts are
1073
+ // equal by then, but a late checkpoint clobbers the usage/cost metadata
1074
+ // only `onFinish` stamps. The chain orders, it does not batch: each write
1075
+ // starts the moment the one before it commits, so a crash at step N loses
1076
+ // at most the write in flight and the one queued behind it. A failed write
1077
+ // logs and releases the chain — the next checkpoint still runs.
1078
+ let checkpoints = Promise.resolve();
1079
+ // Post-turn telemetry: fired after persistence on every terminal path
1080
+ // (completed / awaiting_input / errored). Must never fail the turn.
1081
+ const reportRunFinished = async (status, messages, error) => {
1082
+ const t = config.telemetry;
1083
+ if (!t || !telemetryOn)
1084
+ return;
1085
+ try {
1086
+ await t.onRunFinished?.(telemetryMeta, {
1087
+ status,
1088
+ ...(error !== undefined ? { error } : {}),
1089
+ inputText: lastUserText(uiMessages),
1090
+ messages,
1091
+ usage: {
1092
+ inputTokens,
1093
+ outputTokens,
1094
+ ...(cacheReadTokens ? { cacheReadTokens } : {}),
1095
+ ...(cacheWriteTokens ? { cacheWriteTokens } : {}),
1096
+ },
1097
+ cost: computeCost(config.pricing?.(modelId), {
1098
+ inputTokens,
1099
+ outputTokens,
1100
+ cacheReadTokens,
1101
+ cacheWriteTokens,
1102
+ }),
1103
+ endedAt: Date.now(),
1104
+ });
1105
+ }
1106
+ catch (e) {
1107
+ console.warn("latch telemetry onRunFinished failed:", e);
1108
+ }
1109
+ finally {
1110
+ // Always flush, even when the finish hook threw — otherwise a failed
1111
+ // judge/score pass would drop the turn's spans in serverless.
1112
+ try {
1113
+ await t.flush?.();
1114
+ }
1115
+ catch (e) {
1116
+ console.warn("latch telemetry flush failed:", e);
1117
+ }
1118
+ }
1119
+ };
1120
+ // Lower-level composition (vs `createAgentUIStreamResponse`) so we get the
1121
+ // UI-stream `onStepEnd` (per-step persistence) and the `writer`. The model
1122
+ // call runs INSIDE `execute` so the writer is live during the turn — code
1123
+ // here can write data parts (e.g. a compaction marker) before the model's
1124
+ // output, and tools can write via `toolsContext`.
1125
+ const stream = createUIMessageStream({
1126
+ originalMessages: uiMessages,
1127
+ // Persistence mode: one stable assistant-message id per turn (else every
1128
+ // turn would upsert onto the same row — see the dup-id regression test).
1129
+ generateId: generateMessageId,
1130
+ execute: async ({ writer }) => {
1131
+ // (in-turn writer writes — e.g. a "compacting…" / compaction marker —
1132
+ // would go here, before the model output is merged in)
1133
+ // Model projection: drops `sendToModel:false` messages (and data-* parts).
1134
+ const modelMessages = await toModelMessages(uiMessages, { tools });
1135
+ // Per-session sandbox (one per chat) for agents that declared `sandbox`.
1136
+ // Exposed to sandbox tools as `options.experimental_sandbox`.
1137
+ const sandbox = wantSandbox
1138
+ ? await config.sandbox({
1139
+ principal,
1140
+ chatId: resolvedChatId,
1141
+ agent: run.agent,
1142
+ provider: cfg.sandboxProvider,
1143
+ })
1144
+ : undefined;
1145
+ // Start the model call inside the telemetry run span (when provided)
1146
+ // so the impl can root every AI SDK span in one seeded trace.
1147
+ const startStream = () => agent.stream({
1148
+ prompt: modelMessages,
1149
+ experimental_sandbox: sandbox,
1150
+ abortSignal: dur.enabled ? abort.signal : undefined,
1151
+ // Optional: smooth the visible token cadence before deltas stream out.
1152
+ experimental_transform: cfg.smoothStream
1153
+ ? smoothStream(cfg.smoothStream === true ? undefined : cfg.smoothStream)
1154
+ : undefined,
1155
+ // Shared context for hooks / prepareStep / approval policies.
1156
+ runtimeContext: runtimeCtx,
1157
+ // `toolsContext` is keyed BY TOOL NAME (the SDK does
1158
+ // `toolsContext[toolName]`), so point every tool at the same context:
1159
+ // the project's runtime context + who it's for + the run + the writer
1160
+ // + the runtime itself (so a harness tool can schedule, list chats or
1161
+ // spawn a run without reaching for a host-global). A tool reads it as
1162
+ // `options.context`.
1163
+ toolsContext: Object.fromEntries(Object.keys(tools).map((name) => [
1164
+ name,
1165
+ {
1166
+ ...runtimeCtx,
1167
+ principal,
1168
+ runtime: self,
1169
+ // `agent` is the ref the turn resolved (scope-pinned as stored
1170
+ // on the run), so a tool that acts on "the agent I'm running
1171
+ // as" names the same record the turn is using.
1172
+ run: { id: run.id, chatId: resolvedChatId, agent: run.agent },
1173
+ writer,
1174
+ },
1175
+ ])),
1176
+ // generate-text step callback: accumulate usage across steps,
1177
+ // including the cached-prompt breakdown so cost reflects cache reads.
1178
+ onStepEnd: (step) => {
1179
+ inputTokens += step.usage?.inputTokens ?? 0;
1180
+ outputTokens += step.usage?.outputTokens ?? 0;
1181
+ cacheReadTokens += step.usage?.inputTokenDetails?.cacheReadTokens ?? 0;
1182
+ cacheWriteTokens += step.usage?.inputTokenDetails?.cacheWriteTokens ?? 0;
1183
+ if (telemetryOn && config.telemetry?.onStep) {
1184
+ const onStep = config.telemetry.onStep.bind(config.telemetry);
1185
+ fireAndForget(() => onStep(telemetryMeta, {
1186
+ text: step.text,
1187
+ finishReason: step.finishReason,
1188
+ usage: {
1189
+ inputTokens: step.usage?.inputTokens,
1190
+ outputTokens: step.usage?.outputTokens,
1191
+ cacheReadTokens: step.usage?.inputTokenDetails?.cacheReadTokens,
1192
+ cacheWriteTokens: step.usage?.inputTokenDetails?.cacheWriteTokens,
1193
+ },
1194
+ toolCalls: step.toolCalls,
1195
+ toolResults: step.toolResults,
1196
+ }));
1197
+ }
1198
+ },
1199
+ });
1200
+ const result = telemetryOn && config.telemetry?.withRunSpan
1201
+ ? await config.telemetry.withRunSpan(telemetryMeta, startStream)
1202
+ : await startStream();
1203
+ // Fold the model's own output into the stream.
1204
+ writer.merge(toUIMessageStream({
1205
+ stream: result.stream,
1206
+ // Without this the AI SDK collapses every tool `execute` throw to
1207
+ // the literal "An error occurred." — surface the real message so
1208
+ // failures (sandbox init, credential lookups, …) are debuggable
1209
+ // from the chat itself.
1210
+ onError: clientErrorMessage,
1211
+ messageMetadata: ({ part }) => part.type === "start"
1212
+ ? {
1213
+ visibility: "user",
1214
+ model: modelId,
1215
+ createdAt: Date.now(),
1216
+ // Correlates the assistant message to its run (and trace)
1217
+ // — e.g. per-message feedback posts against this id.
1218
+ runId: run.id,
1219
+ }
1220
+ : undefined,
1221
+ }));
1222
+ },
1223
+ // UI-stream per-step hook: checkpoint the in-progress assistant message
1224
+ // (with any completed tool results) so a crash mid-turn doesn't repeat
1225
+ // them on resume. Durable turns only — keeps T1 write counts unchanged.
1226
+ //
1227
+ // Guarded on the turn's abort signal: `appendMessages` has no lease
1228
+ // fence (only `updateRun` does), so once this worker has lost the lease
1229
+ // a reclaimed copy of the run may be replaying into this same chat —
1230
+ // an un-guarded checkpoint from the zombie would interleave with it,
1231
+ // and every later resume replays the polluted history.
1232
+ //
1233
+ // Best-effort, not a fence: the lease can be lost between this check and
1234
+ // the write committing. Closing that window needs a lease/fencing-token
1235
+ // check inside the storage write itself — an adapter interface change,
1236
+ // tracked as a follow-up. With the heartbeat now aborting within one TTL
1237
+ // of losing contact, the exposure is bounded to ~leaseTtlMs.
1238
+ onStepEnd: dur.enabled
1239
+ ? ({ responseMessage }) => {
1240
+ if (abort.signal.aborted)
1241
+ return;
1242
+ checkpoints = checkpoints
1243
+ .then(() => {
1244
+ // Re-checked at write time: the lease may have gone while this
1245
+ // checkpoint waited behind the previous one.
1246
+ if (abort.signal.aborted)
1247
+ return;
1248
+ return storage.appendMessages(resolvedChatId, [responseMessage]);
1249
+ })
1250
+ .catch((e) => console.warn(`[durability] step checkpoint failed for run ${run.id}:`, e));
1251
+ }
1252
+ : undefined,
1253
+ onFinish: async ({ messages, isAborted, }) => {
1254
+ stopHeartbeat();
1255
+ await closeSources();
1256
+ // Cost for this turn (0 if no pricing configured). Stamp usage + cost
1257
+ // onto the assistant message — the message is complete; this is metadata.
1258
+ const cost = computeCost(config.pricing?.(modelId), {
1259
+ inputTokens,
1260
+ outputTokens,
1261
+ cacheReadTokens,
1262
+ cacheWriteTokens,
1263
+ });
1264
+ const last = messages[messages.length - 1];
1265
+ if (last?.role === "assistant") {
1266
+ last.metadata = {
1267
+ ...last.metadata,
1268
+ usage: {
1269
+ inputTokens,
1270
+ outputTokens,
1271
+ ...(cacheReadTokens ? { cacheReadTokens } : {}),
1272
+ ...(cacheWriteTokens ? { cacheWriteTokens } : {}),
1273
+ },
1274
+ cost,
1275
+ };
1276
+ }
1277
+ // Errored (the stream failed — onError already reported it) or aborted
1278
+ // (lease lost): don't claim completion, and DON'T persist the partial
1279
+ // messages/usage — an aborted run may already be reaped and re-claimed
1280
+ // by another node whose writes must win. The guarded update no-ops if
1281
+ // the run was reassigned. Exactly one terminal report either way.
1282
+ if (failed !== undefined || isAborted) {
1283
+ const settled = await storage.updateRun(run.id, { status: "errored", error: failed ?? "aborted", endedAt: Date.now() }, lease);
1284
+ // Only the winning writer reports (same rule as the completion
1285
+ // path): a stale worker whose fenced update did not apply is not
1286
+ // the one that ended this run. The tokens WERE spent, though — a
1287
+ // failing turn settles against its budgets like any other, or a
1288
+ // token-only cap would never bound a run that keeps erroring.
1289
+ if (settled) {
1290
+ await settleLimits(admitted, { tokens: inputTokens + outputTokens, cost });
1291
+ await reportRunFinished("errored", messages, failed ?? "aborted");
1292
+ }
1293
+ return;
1294
+ }
1295
+ // A gated tool ended the turn awaiting a decision — pause, don't
1296
+ // complete. The decision arrives via applyApproval (message state).
1297
+ const paused = hasPendingApproval(messages);
1298
+ // Every step checkpoint has committed (or failed and been logged)
1299
+ // BEFORE the row is settled. Recovery never revisits a completed run,
1300
+ // so the moment the row says `completed` the stored message must
1301
+ // already hold the final step — otherwise a crash between the settle
1302
+ // and the final write below loses the answer text, not just metadata.
1303
+ // Awaiting here also makes the final write the row's last one.
1304
+ await checkpoints;
1305
+ // Fenced finish: settle the run FIRST. If the guarded update doesn't
1306
+ // apply (our lease was reaped + the run re-claimed elsewhere), we're
1307
+ // a stale writer — skip the final message/usage persistence so we
1308
+ // don't overwrite the new owner's state.
1309
+ const applied = await storage.updateRun(run.id, {
1310
+ status: paused ? "awaiting_input" : "completed",
1311
+ usageInputTokens: inputTokens,
1312
+ usageOutputTokens: outputTokens,
1313
+ usageCacheReadTokens: cacheReadTokens || undefined,
1314
+ usageCacheWriteTokens: cacheWriteTokens || undefined,
1315
+ costTotal: cost,
1316
+ toolsHash,
1317
+ endedAt: paused ? undefined : Date.now(),
1318
+ }, lease);
1319
+ if (applied) {
1320
+ await storage.appendMessages(resolvedChatId, messages);
1321
+ await storage.recordUsage(principal, {
1322
+ runId: run.id,
1323
+ model: modelId,
1324
+ inputTokens,
1325
+ outputTokens,
1326
+ ...(cacheReadTokens ? { cacheReadTokens } : {}),
1327
+ ...(cacheWriteTokens ? { cacheWriteTokens } : {}),
1328
+ cost,
1329
+ });
1330
+ await settleLimits(admitted, { tokens: inputTokens + outputTokens, cost });
1331
+ }
1332
+ // Only the winning writer reports terminal telemetry. If our lease was
1333
+ // reaped and the run re-claimed elsewhere (`!applied`), that node will
1334
+ // finish it — a stale writer must not re-run the judge or emit a
1335
+ // duplicate "completed"/"awaiting_input" trace.
1336
+ if (applied)
1337
+ await reportRunFinished(paused ? "awaiting_input" : "completed", messages);
1338
+ },
1339
+ onError: (error) => {
1340
+ stopHeartbeat();
1341
+ void closeSources();
1342
+ const message = clientErrorMessage(error);
1343
+ // Record only — the first error wins; later onError calls for the same
1344
+ // stream are noise. The SDK's onError must return the client string
1345
+ // synchronously and still runs onFinish afterwards, so that is where
1346
+ // the run row is settled as errored and the single terminal report is
1347
+ // made (after, and only if, the fenced update applied).
1348
+ if (failed === undefined)
1349
+ failed = message;
1350
+ return message;
1351
+ },
1352
+ });
1353
+ // Drain the stream server-side regardless of the client. If the browser
1354
+ // closes mid-turn, the response stops being pulled — without this the
1355
+ // generation stalls/cuts short and a refresh shows a partial answer. With
1356
+ // it, the agent runs to completion in the background and onFinish persists
1357
+ // the full message (the model's abort is tied to the lease, not the
1358
+ // request, so a disconnect doesn't cancel generation).
1359
+ let out = stream;
1360
+ if (args.observeStream) {
1361
+ const [main, observed] = out.tee();
1362
+ out = main;
1363
+ args.observeStream(observed);
1364
+ }
1365
+ return createUIMessageStreamResponse({
1366
+ stream: out,
1367
+ consumeSseStream: consumeStream,
1368
+ });
1369
+ }
1370
+ async function reapTick(opts) {
1371
+ const reaped = await storage.reapExpiredRuns(opts?.now ?? Date.now(), {
1372
+ error: opts?.error,
1373
+ });
1374
+ return { reaped };
1375
+ }
1376
+ /**
1377
+ * The admission gate. A FRESH turn (chat, schedule, subagent spawn, runAgent)
1378
+ * is RESERVED: one atomic increment of every policy's turn counter, which
1379
+ * returns the new counts; the first policy now over its cap refuses the turn
1380
+ * (`LimitExceededError`) and the reservation is released. Increment-then-
1381
+ * check is what makes a one-turn cap admit exactly one of N concurrent
1382
+ * requests — a read-then-write check would let them all through.
1383
+ *
1384
+ * A CONTINUATION is the second half of an already-admitted turn: a crash
1385
+ * resume, an approval or tool-result decision, a subagent's decided gate. It
1386
+ * passes `{ countTurn: false, refuse: false }`: never counted twice, never
1387
+ * refused (the persisted decision would strand until the window turned
1388
+ * over), but its tokens still settle and the mid-turn stop still applies —
1389
+ * with an exhausted budget it gets one step and stops. Undefined when no
1390
+ * limits are configured or none apply.
1391
+ */
1392
+ async function admitTurn(principal, agent, trigger, opts = { countTurn: true, refuse: true }) {
1393
+ const limits = config.limits;
1394
+ if (!limits)
1395
+ return undefined;
1396
+ const policies = await limits.limitsFor({ principal, agent, trigger });
1397
+ if (policies.length === 0)
1398
+ return undefined;
1399
+ const now = Date.now();
1400
+ const refs = policies.map((p) => windowRef(p, now));
1401
+ // Reserve (fresh turn) or just look (continuation). Either way `usage` is
1402
+ // the state this turn is judged against, reservation included.
1403
+ const usage = opts.countTurn
1404
+ ? await limits.store.add(refs.map((ref) => ({ ...ref, usage: { turns: 1 } })))
1405
+ : await limits.store.read(refs);
1406
+ if (opts.refuse) {
1407
+ const over = exceededPolicy(policies, usage, now, { nextTurn: false });
1408
+ if (over) {
1409
+ // Release the reservation: the window should count admitted turns, not
1410
+ // refused attempts. Best-effort — a failed release only over-counts.
1411
+ if (opts.countTurn) {
1412
+ await limits.store
1413
+ .add(refs.map((ref) => ({ ...ref, usage: { turns: -1 } })))
1414
+ .catch((e) => console.error("[limits] reservation release failed:", e));
1415
+ }
1416
+ // A host hook must never turn a refusal into a 500: isolate sync throws too.
1417
+ try {
1418
+ await limits.onRefused?.({ principal, agent, trigger, ...over });
1419
+ }
1420
+ catch (e) {
1421
+ console.error("[limits] onRefused failed:", e);
1422
+ }
1423
+ throw new LimitExceededError(over.policy, over.usage, over.retryAfterMs);
1424
+ }
1425
+ }
1426
+ return { policies, refs, remainingTokens: remainingTokens(policies, usage, now), counted: opts.countTurn };
1427
+ }
1428
+ /**
1429
+ * Give a reserved turn back when the turn never opened (busy chat, agent
1430
+ * resolution or run creation failed): the window counts turns that RAN, or a
1431
+ * user retrying into a busy chat would exhaust their own turn cap.
1432
+ */
1433
+ async function releaseReservation(admitted) {
1434
+ if (!admitted?.counted || !config.limits)
1435
+ return;
1436
+ await config.limits.store
1437
+ .add(admitted.refs.map((ref) => ({ ...ref, usage: { turns: -1 } })))
1438
+ .catch((e) => console.error("[limits] reservation release failed:", e));
1439
+ }
1440
+ /** Run `open` for a freshly admitted turn; on failure the reservation is released and the error rethrown. */
1441
+ async function withReservation(admitted, open) {
1442
+ try {
1443
+ return await open();
1444
+ }
1445
+ catch (e) {
1446
+ await releaseReservation(admitted);
1447
+ throw e;
1448
+ }
1449
+ }
1450
+ /** `admitTurn` for continuations (see above). */
1451
+ const admitContinuation = (principal, agent, trigger) => admitTurn(principal, agent, trigger, { countTurn: false, refuse: false });
1452
+ /** Settlement: add this turn's tokens and cost to every policy it was admitted under. */
1453
+ async function settleLimits(admitted, usage) {
1454
+ if (!admitted || !config.limits)
1455
+ return;
1456
+ const now = Date.now();
1457
+ await config.limits.store
1458
+ .add(admitted.policies.map((p) => ({ ...windowRef(p, now), usage })))
1459
+ .catch((e) => console.error("[limits] settlement failed:", e));
1460
+ }
1461
+ /**
1462
+ * Open a turn: the run (one active per chat — `ChatBusyError` if not ours to
1463
+ * run) and the user message that starts it. Atomic when the adapter offers
1464
+ * `openTurn`; otherwise the run first, then the message, so a refusal still
1465
+ * persists nothing (the crash window between the two writes is the trade-off
1466
+ * an adapter without `openTurn` accepts).
1467
+ */
1468
+ async function openTurn(principal, chatId, agentName, messages) {
1469
+ if (storage.openTurn) {
1470
+ const run = await storage.openTurn(principal, {
1471
+ chatId,
1472
+ agent: agentName,
1473
+ messages,
1474
+ ...(dur.enabled ? { lease: { owner: dur.instanceId, ttlMs: dur.leaseTtlMs } } : {}),
1475
+ });
1476
+ return {
1477
+ run,
1478
+ lease: dur.enabled ? { owner: dur.instanceId, fencingToken: run.fencingToken ?? 1 } : undefined,
1479
+ };
1480
+ }
1481
+ const opened = await openRun(principal, chatId, agentName);
1482
+ if (messages.length > 0)
1483
+ await storage.appendMessages(chatId, messages);
1484
+ return opened;
1485
+ }
1486
+ async function openRun(principal, chatId, agentName) {
1487
+ if (dur.enabled) {
1488
+ const run = await storage.claimRun({
1489
+ principal,
1490
+ chatId,
1491
+ agent: agentName,
1492
+ owner: dur.instanceId,
1493
+ ttlMs: dur.leaseTtlMs,
1494
+ });
1495
+ return { run, lease: { owner: dur.instanceId, fencingToken: run.fencingToken ?? 1 } };
1496
+ }
1497
+ return {
1498
+ run: await storage.createRun(principal, { chatId, agent: agentName, kind: "turn" }),
1499
+ lease: undefined,
1500
+ };
1501
+ }
1502
+ async function handleChat({ agent: agentName, chatId, principal, message, request, turnContext, }) {
1503
+ // chat: scoped create-or-continue (others' ids resolve to "not found")
1504
+ let chat = chatId ? await storage.getChat(principal, chatId) : null;
1505
+ if (!chat) {
1506
+ chat = await storage.createChat(principal, {
1507
+ id: chatId, // use the provided id (create-or-continue); generated if omitted
1508
+ agent: agentName,
1509
+ // Title from the first user message — for the chat list.
1510
+ title: firstText(message)?.slice(0, 80),
1511
+ });
1512
+ }
1513
+ const resolvedChatId = chat.id;
1514
+ const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(agentName, principal, request, turnContext);
1515
+ const prior = await storage.loadMessages(principal, resolvedChatId);
1516
+ // A new message while the last turn is parked on an approval is itself the
1517
+ // decision: the user declined to answer and wants to steer elsewhere. Close
1518
+ // the gate as declined (persisted, so the card stops rendering as live) and
1519
+ // carry on — the gated tool never runs, and the model sees the denial
1520
+ // followed by the new instruction. Refusing the message instead is what
1521
+ // wedged chats whose gate could not be satisfied at all.
1522
+ if (hasPendingApproval(prior)) {
1523
+ const declined = declinePendingApprovals(prior);
1524
+ if (declined.length > 0)
1525
+ await storage.appendMessages(resolvedChatId, declined);
1526
+ }
1527
+ // Limits gate BEFORE anything is persisted: a refused turn leaves no
1528
+ // trace (the client gets 429 and can retry when the window turns over).
1529
+ const admitted = await admitTurn(principal, agentName, "chat");
1530
+ // Run + message open together (atomically where the adapter can): a chat
1531
+ // with a live turn refuses (`ChatBusyError` → 409) and the message never
1532
+ // lands in history — the client retries it (with its turn reservation released).
1533
+ const { run, lease } = await withReservation(admitted, () => openTurn(principal, resolvedChatId, agentName, [message]));
1534
+ return buildTurn({
1535
+ resolvedChatId,
1536
+ principal,
1537
+ modelId,
1538
+ cfg,
1539
+ run,
1540
+ lease,
1541
+ uiMessages: [...prior, message],
1542
+ runtimeCtx,
1543
+ admitted,
1544
+ });
1545
+ }
1546
+ async function resumeRun(runId) {
1547
+ if (!dur.enabled)
1548
+ throw new Error("resume requires durability.enabled");
1549
+ // Exactly-once re-claim: only one node wins; a concurrent resumer / an
1550
+ // already-completed run yields null.
1551
+ const run = await storage.reclaimRun(runId, dur.instanceId, dur.leaseTtlMs);
1552
+ if (!run)
1553
+ return null;
1554
+ // System op (no request): rebuild the principal from the run's identity blob.
1555
+ const principal = await reconstructPrincipal(run.identity);
1556
+ const uiMessages = await storage.loadMessages(principal, run.chatId);
1557
+ const lease = {
1558
+ owner: dur.instanceId,
1559
+ fencingToken: run.fencingToken ?? 1,
1560
+ };
1561
+ // What does this run actually need? A crash between the last step's
1562
+ // checkpoint and `onFinish`'s `updateRun` leaves a FINISHED turn on an
1563
+ // `active` row, and re-running the model there would pay twice for an answer
1564
+ // already in history — and append the second copy to the same message. So
1565
+ // reconcile the row instead, and only re-run when work is genuinely left.
1566
+ // Did the chat move on while this run sat reaped? Asked of the runs table,
1567
+ // never inferred from message shapes — see `supersededBy` for why. The list
1568
+ // is owner-scoped and most-recent first; a later sibling is more recent than
1569
+ // we are, so it is present unless more than `limit` runs happened owner-wide
1570
+ // after IT. A miss errs toward re-running (today's behaviour), never toward
1571
+ // settling a live run.
1572
+ const siblings = await storage.listRuns(principal, { limit: 500 });
1573
+ const shape = supersededBy(run, siblings) ? "superseded" : resumeShapeOf(uiMessages);
1574
+ if (shape !== "continue") {
1575
+ const status = shape === "superseded" ? "errored" : shape === "decision-pending" ? "awaiting_input" : "completed";
1576
+ // Same outcome `openTurn` records when a fresh message supersedes a stale
1577
+ // run, so a superseded run reads the same however it got there.
1578
+ const error = shape === "superseded" ? "superseded: a later turn moved the conversation on" : null;
1579
+ // Fenced like every other terminal write: if our lease was reaped again
1580
+ // mid-flight, the new owner decides this run's end state, not us.
1581
+ const applied = await storage.updateRun(run.id, {
1582
+ status,
1583
+ // The reaper stamped BOTH an error and an `endedAt` on the way in
1584
+ // (see `reapExpiredRuns`). A finished turn did not fail, so its error
1585
+ // has to go or the row reads as failed forever — and a run settling to
1586
+ // `awaiting_input` is paused, not over, so its `endedAt` has to go
1587
+ // too. `onFinish` leaves it unset for a paused turn; a crash-settled
1588
+ // one must match, or inter-run gap stats read a run that ended before
1589
+ // its decision arrived. `null` (not `undefined`) is what clears a
1590
+ // column: the adapters drop undefined from the patch.
1591
+ error,
1592
+ endedAt: status === "awaiting_input" ? null : Date.now(),
1593
+ }, lease);
1594
+ // The process that spent this turn's tokens died before recording them,
1595
+ // so usage is unknowable here — the row and this report agree on zero
1596
+ // rather than inventing a number. The notification still fires: a host
1597
+ // watching `onRunFinished` for "the turn ended" must not miss a turn just
1598
+ // because recovery was a row write instead of a model call.
1599
+ const meta = {
1600
+ runId: run.id,
1601
+ chatId: run.chatId,
1602
+ agent: run.agent,
1603
+ // The model that ACTUALLY ran, off the message it produced — not a
1604
+ // freshly resolved config. Reconciling a finished turn must not need
1605
+ // the agent to still exist (see above), and this is the truer answer
1606
+ // anyway: the agent's model may have been changed since.
1607
+ modelId: ranAs(uiMessages, run.id) ?? "unknown",
1608
+ principal,
1609
+ startedAt: run.startedAt,
1610
+ depth: 0,
1611
+ trigger: "resume",
1612
+ rootChatId: run.chatId,
1613
+ };
1614
+ // Same opt-out rule as a streamed turn: a `telemetryFor` that returns
1615
+ // undefined silences this run's hooks too (see buildTurn).
1616
+ const telemetryOn = !!config.telemetry &&
1617
+ !(config.telemetry.telemetryFor && config.telemetry.telemetryFor(meta) === undefined);
1618
+ if (applied && telemetryOn) {
1619
+ try {
1620
+ await config.telemetry?.onRunFinished?.(meta, {
1621
+ status,
1622
+ ...(error !== null ? { error } : {}),
1623
+ inputText: lastUserText(uiMessages),
1624
+ messages: uiMessages,
1625
+ usage: { inputTokens: 0, outputTokens: 0 },
1626
+ cost: 0,
1627
+ endedAt: Date.now(),
1628
+ });
1629
+ }
1630
+ catch (e) {
1631
+ console.warn("latch telemetry onRunFinished failed:", e);
1632
+ }
1633
+ try {
1634
+ await config.telemetry?.flush?.();
1635
+ }
1636
+ catch (e) {
1637
+ console.warn("latch telemetry flush failed:", e);
1638
+ }
1639
+ }
1640
+ const settled = await storage.getRun(runId);
1641
+ return {
1642
+ runId,
1643
+ status: settled?.status ?? status,
1644
+ attempt: settled?.attempt ?? run.attempt ?? null,
1645
+ };
1646
+ }
1647
+ // Only now resolve the agent — `resolveAgentConfig` throws on an agent that
1648
+ // no longer resolves, and a run whose answer is already stored must be
1649
+ // reconcilable even if its (dynamic, DB-stored) agent was since deleted.
1650
+ // Before this ordering that throw also aborted `sweep`'s loop, so one such
1651
+ // run held up recovery for every other run until its attempts ran out.
1652
+ const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(run.agent, principal);
1653
+ // A resume was admitted when the turn first ran — never refuse it (the
1654
+ // half-done work would strand), but its tokens still settle against the
1655
+ // same counters, and the mid-turn stop still applies.
1656
+ const admitted = await admitContinuation(principal, run.agent, "resume");
1657
+ // Re-run from persisted history and drive to completion server-side
1658
+ // (there's no client to consume the stream).
1659
+ const res = await buildTurn({
1660
+ resolvedChatId: run.chatId,
1661
+ principal,
1662
+ modelId,
1663
+ cfg,
1664
+ run,
1665
+ lease,
1666
+ // Same prefill guard the decision paths use: a tail ending in the turn's
1667
+ // own half-written sentence would reach the model as an assistant
1668
+ // prefill, which models that reject one refuse on the request SHAPE — so
1669
+ // every identical retry fails until `maxAttempts` is spent. `unanchored`
1670
+ // extends it to a tail with no tool part to anchor on; no checkpoint
1671
+ // produces such a tail on this path today (see the note there), it is
1672
+ // here so a classifier change cannot reintroduce the prefill.
1673
+ uiMessages: trimTrailingAssistantPrefill(uiMessages, { unanchored: true }),
1674
+ runtimeCtx,
1675
+ admitted,
1676
+ // Crash recovery, driven server-side: nobody is watching this stream.
1677
+ trigger: "resume",
1678
+ });
1679
+ await res.text();
1680
+ const final = await storage.getRun(runId);
1681
+ return {
1682
+ runId,
1683
+ status: final?.status ?? "errored",
1684
+ attempt: final?.attempt ?? run.attempt ?? null,
1685
+ };
1686
+ }
1687
+ async function sweep(opts) {
1688
+ if (!dur.enabled)
1689
+ throw new Error("sweep requires durability.enabled");
1690
+ const maxAttempts = opts?.maxAttempts ?? 5;
1691
+ const { reaped } = await reapTick({ now: opts?.now, error: opts?.error });
1692
+ const resumed = [];
1693
+ for (const run of reaped) {
1694
+ // Give up on poison runs that keep dying — don't loop forever.
1695
+ if ((run.attempt ?? 1) >= maxAttempts)
1696
+ continue;
1697
+ const outcome = await resumeRun(run.id);
1698
+ if (outcome)
1699
+ resumed.push(outcome);
1700
+ }
1701
+ return { reaped, resumed };
1702
+ }
1703
+ async function cron(opts) {
1704
+ const errors = [];
1705
+ const describe = (e) => (e instanceof Error ? e.message : String(e));
1706
+ let due = { fired: 0, errors: 0, skipped: 0, parked: 0 };
1707
+ try {
1708
+ due = await runDue({ now: opts?.now });
1709
+ }
1710
+ catch (e) {
1711
+ errors.push(`runDue: ${describe(e)}`);
1712
+ }
1713
+ let reaped = [];
1714
+ let resumed = [];
1715
+ // Without durability there are no leases to reap — nothing to sweep.
1716
+ if (dur.enabled) {
1717
+ try {
1718
+ ({ reaped, resumed } = await sweep({ now: opts?.now, maxAttempts: opts?.maxAttempts }));
1719
+ }
1720
+ catch (e) {
1721
+ errors.push(`sweep: ${describe(e)}`);
1722
+ }
1723
+ }
1724
+ // Limit counters: windows that ended more than a day ago are dead weight.
1725
+ if (config.limits?.store.prune) {
1726
+ try {
1727
+ await config.limits.store.prune((opts?.now ?? Date.now()) - 24 * 3_600_000);
1728
+ }
1729
+ catch (e) {
1730
+ errors.push(`limits.prune: ${describe(e)}`);
1731
+ }
1732
+ }
1733
+ return { due, reaped, resumed, errors };
1734
+ }
1735
+ async function applyApproval({ chatId, principal, decisions, request, turnContext, }) {
1736
+ const chat = await storage.getChat(principal, chatId);
1737
+ if (!chat)
1738
+ throw new Error(`Unknown chat: ${chatId}`);
1739
+ // Merge the decisions onto the persisted assistant message, then re-persist
1740
+ // only what changed. The agent continues from this updated history.
1741
+ const messages = await storage.loadMessages(principal, chatId);
1742
+ const changed = applyDecisions(messages, decisions);
1743
+ if (changed.length > 0)
1744
+ await storage.appendMessages(chatId, changed);
1745
+ // Route any just-decided SUBAGENT approvals (bubbled up via spawn_agent):
1746
+ // resume the sub thread with the decision, then fold its answer back into
1747
+ // the parent's spawn_agent result so this turn continues with the digest.
1748
+ const subApprovals = collectDecidedSubagentApprovals(messages);
1749
+ if (subApprovals.length > 0) {
1750
+ for (const sa of subApprovals) {
1751
+ const resumed = await resumeSubagentApproval(principal, sa.subChatId, sa.subApprovalIds.map((id) => ({
1752
+ approvalId: id,
1753
+ approved: sa.approved,
1754
+ reason: sa.reason,
1755
+ })), { parentRunId: sa.parentRunId, rootChatId: sa.rootChatId });
1756
+ if (resumed.pending?.length) {
1757
+ // The resumed sub paused on ANOTHER gated tool — re-bubble it as a
1758
+ // fresh data-subagent-approval part on the same parent message. The
1759
+ // undecided part keeps this chat awaiting_input (hasPendingApproval
1760
+ // below), and the next decision routes through here again, so
1761
+ // bubbling recurses per round.
1762
+ const nextApprovalId = generateSubApprovalId();
1763
+ appendSubagentApprovalPart(messages, sa.approvalId, {
1764
+ approvalId: nextApprovalId,
1765
+ subChatId: sa.subChatId,
1766
+ subApprovalIds: resumed.pending.map((p) => p.approvalId),
1767
+ spawnToolCallId: sa.spawnToolCallId,
1768
+ summaries: resumed.pending.map((p) => p.summary),
1769
+ // Carry the linkage forward so each further resume stays nested.
1770
+ parentRunId: sa.parentRunId,
1771
+ rootChatId: sa.rootChatId,
1772
+ });
1773
+ if (sa.spawnToolCallId) {
1774
+ overwriteToolResult(messages, sa.spawnToolCallId, {
1775
+ status: "awaiting_approval",
1776
+ threadId: sa.subChatId,
1777
+ pending: resumed.pending.map((p) => p.summary),
1778
+ note: "Awaiting the user's approval — do not retry; stop here, the user will decide and you'll continue automatically.",
1779
+ });
1780
+ }
1781
+ }
1782
+ else if (sa.spawnToolCallId) {
1783
+ overwriteToolResult(messages, sa.spawnToolCallId, {
1784
+ threadId: sa.subChatId,
1785
+ answer: resumed.answer,
1786
+ });
1787
+ }
1788
+ markSubagentApprovalResolved(messages, sa.approvalId);
1789
+ }
1790
+ // Persist the folded spawn_agent result (or re-bubbled approval part)
1791
+ // + resolved markers.
1792
+ await storage.appendMessages(chatId, messages);
1793
+ }
1794
+ // If the step had several gated calls and some are still undecided, DON'T
1795
+ // re-run yet — a tool call without a result makes convertToModelMessages
1796
+ // throw MissingToolResultsError. Persist the decision, stay awaiting_input,
1797
+ // and let the client submit the rest. Only continue once none are pending.
1798
+ if (hasPendingApproval(messages)) {
1799
+ return new Response(null, { status: 200 });
1800
+ }
1801
+ const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(chat.agent, principal, request, turnContext);
1802
+ const admitted = await admitContinuation(principal, chat.agent, "decision");
1803
+ const { run, lease } = await openRun(principal, chatId, chat.agent);
1804
+ return buildTurn({
1805
+ resolvedChatId: chatId,
1806
+ principal,
1807
+ modelId,
1808
+ cfg,
1809
+ run,
1810
+ lease,
1811
+ // Resume: strip the paused turn's trailing text so the model call ends on
1812
+ // the tool result (a user turn), not an assistant prefill.
1813
+ uiMessages: trimTrailingAssistantPrefill(messages),
1814
+ runtimeCtx,
1815
+ // A human just decided something, so a human is present — even if this
1816
+ // turn parks again on the NEXT gate (see TurnTrigger).
1817
+ trigger: "decision",
1818
+ admitted,
1819
+ });
1820
+ }
1821
+ /**
1822
+ * Resume a turn paused on a client-handled tool (one with no `execute`, e.g.
1823
+ * `askUser`). Records the supplied `output` onto the matching tool call in the
1824
+ * persisted assistant message (input-available → output-available), then
1825
+ * re-runs the agent from history. Mirrors `applyApproval`: the result
1826
+ * round-trips as message state, not a client-message ingest (the handler only
1827
+ * ever accepts a fresh user message otherwise).
1828
+ */
1829
+ async function applyToolResult({ chatId, principal, toolCallId, output, request, turnContext, }) {
1830
+ const chat = await storage.getChat(principal, chatId);
1831
+ if (!chat)
1832
+ throw new Error(`Unknown chat: ${chatId}`);
1833
+ const messages = await storage.loadMessages(principal, chatId);
1834
+ const changed = recordToolResult(messages, toolCallId, output);
1835
+ // Unknown or already-answered call: nothing to resume. Makes a duplicate
1836
+ // submit (double-tapped button, redelivered webhook) a no-op instead of a
1837
+ // spurious extra turn.
1838
+ if (changed.length === 0)
1839
+ return new Response(null, { status: 200 });
1840
+ // Another client tool call in the SAME step may still be awaiting a result;
1841
+ // re-running now would throw MissingToolResultsError. Record this one and
1842
+ // wait for the rest. Scope this to the messages we just touched: a call
1843
+ // abandoned turns ago is still `input-available` in stored history, and
1844
+ // counting it would park every future resume in this chat forever.
1845
+ if (hasPendingToolResult(changed)) {
1846
+ await storage.appendMessages(chatId, changed);
1847
+ return new Response(null, { status: 200 });
1848
+ }
1849
+ const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(chat.agent, principal, request, turnContext);
1850
+ const admitted = await admitContinuation(principal, chat.agent, "decision");
1851
+ // The result is persisted WITH the run (atomically where the adapter can):
1852
+ // a busy chat refuses before anything is recorded, so the retry finds the
1853
+ // call still unanswered and resumes it — persisting first would make the
1854
+ // retry a "duplicate submit" no-op and lose the answer.
1855
+ const { run, lease } = await openTurn(principal, chatId, chat.agent, changed);
1856
+ return buildTurn({
1857
+ resolvedChatId: chatId,
1858
+ principal,
1859
+ modelId,
1860
+ cfg,
1861
+ run,
1862
+ lease,
1863
+ // Resume: same prefill guard as applyApproval (see trimTrailingAssistantPrefill).
1864
+ uiMessages: trimTrailingAssistantPrefill(messages),
1865
+ runtimeCtx,
1866
+ // A client-handled tool result is a human answering (see TurnTrigger).
1867
+ trigger: "decision",
1868
+ admitted,
1869
+ });
1870
+ }
1871
+ /**
1872
+ * Compact a conversation: summarize the live context window and append the
1873
+ * summary as a boundary marker, so later turns send the model the summary
1874
+ * instead of the turns it replaces (see `sliceAtCompaction`). Stored history is
1875
+ * never rewritten — nothing is lost, and a chat can be compacted repeatedly.
1876
+ *
1877
+ * Returns null when there's too little to be worth compacting (or the chat
1878
+ * isn't the caller's). Throws `PendingApprovalError` when the turn is paused:
1879
+ * the parked `tool_use` would fall behind the boundary and vanish from the
1880
+ * model view while stored history still gates every new message.
1881
+ */
1882
+ async function compactChat({ chatId, principal, focus, signal, request, turnContext, }) {
1883
+ const chat = await storage.getChat(principal, chatId);
1884
+ if (!chat)
1885
+ return null;
1886
+ const stored = await storage.loadMessages(principal, chatId);
1887
+ // Only the live window is summarized: everything before the previous
1888
+ // boundary is already represented by that boundary's summary.
1889
+ const live = sliceAtCompaction(stored);
1890
+ if (live.length < MIN_COMPACTABLE_MESSAGES)
1891
+ return null;
1892
+ if (hasPendingApproval(live) || hasPendingToolResult([live[live.length - 1]])) {
1893
+ throw new PendingApprovalError("This chat is awaiting a decision — resolve the pending request before compacting.");
1894
+ }
1895
+ const { cfg, modelId } = await resolveAgentConfig(chat.agent, principal, request, turnContext);
1896
+ // Tool definitions only affect how a recorded tool RESULT is rendered for the
1897
+ // model (`toModelOutput`); an unknown tool falls back to its raw JSON rather
1898
+ // than failing. So take every definition available without I/O — the agent's
1899
+ // own plus the harness ones — and deliberately skip opening dynamic sources
1900
+ // (MCP / connections): summarizing must not depend on a remote server being
1901
+ // reachable, and raw JSON is a fine thing to summarize.
1902
+ const harness = config.harnessTools ?? {};
1903
+ const tools = {
1904
+ ...cfg.tools,
1905
+ ...harness.app,
1906
+ ...harness.sandbox,
1907
+ };
1908
+ // Summarizing needs no reasoning block, and on a model that thinks by
1909
+ // DEFAULT that block spends the same output budget as the summary — leaving
1910
+ // the call truncated with nothing usable. `"minimal"` is the effort every
1911
+ // provider mapping turns into "thinking off", so reuse the host's hook rather
1912
+ // than hardcoding provider shapes here.
1913
+ const noThinking = config.resolveReasoningOptions?.(modelId, "minimal")?.providerOptions;
1914
+ const run = await storage.createRun(principal, {
1915
+ chatId,
1916
+ agent: chat.agent,
1917
+ // Not a conversational turn — stats/rollups filter on kind so a
1918
+ // compaction's cache-less usage never depresses an agent's hit rate.
1919
+ kind: "compaction",
1920
+ });
1921
+ try {
1922
+ const { text, truncated, usage } = await summarizeForCompaction({
1923
+ model: cfg.model,
1924
+ messages: await toModelMessages(live, { tools }),
1925
+ ...(focus ? { focus } : {}),
1926
+ ...(noThinking ? { providerOptions: noThinking } : {}),
1927
+ ...(signal ? { signal } : {}),
1928
+ });
1929
+ if (!text)
1930
+ throw new Error("The model returned an empty summary.");
1931
+ // Still truncated after the shortened retry (see summarizeForCompaction).
1932
+ // A summary cut off mid-thought must not be stored — it would become the
1933
+ // permanent head of every later prompt. History is untouched, so retrying
1934
+ // is safe.
1935
+ if (truncated) {
1936
+ throw new Error("Couldn't fit a summary of this conversation — nothing was changed. Try /compact again, or /new to start fresh.");
1937
+ }
1938
+ const cost = computeCost(config.pricing?.(modelId), usage);
1939
+ // `internal` so the client projection never shows a "user" message the user
1940
+ // didn't write; `sendToModel` stays default-true so the model DOES see it.
1941
+ // Role `user` because after slicing this is the conversation's first
1942
+ // message, and a model prompt may not open with an assistant turn.
1943
+ await storage.appendMessages(chatId, [
1944
+ {
1945
+ id: generateMessageId(),
1946
+ role: "user",
1947
+ parts: [
1948
+ { type: "text", text },
1949
+ {
1950
+ type: COMPACTION_PART_TYPE,
1951
+ data: { compacted: live.length, at: Date.now() },
1952
+ },
1953
+ ],
1954
+ metadata: {
1955
+ visibility: "internal",
1956
+ createdAt: Date.now(),
1957
+ model: modelId,
1958
+ usage: {
1959
+ inputTokens: usage.inputTokens,
1960
+ outputTokens: usage.outputTokens,
1961
+ ...(usage.cacheReadTokens ? { cacheReadTokens: usage.cacheReadTokens } : {}),
1962
+ ...(usage.cacheWriteTokens ? { cacheWriteTokens: usage.cacheWriteTokens } : {}),
1963
+ },
1964
+ cost,
1965
+ },
1966
+ },
1967
+ ]);
1968
+ await storage.updateRun(run.id, {
1969
+ status: "completed",
1970
+ usageInputTokens: usage.inputTokens,
1971
+ usageOutputTokens: usage.outputTokens,
1972
+ usageCacheReadTokens: usage.cacheReadTokens || undefined,
1973
+ usageCacheWriteTokens: usage.cacheWriteTokens || undefined,
1974
+ costTotal: cost,
1975
+ endedAt: Date.now(),
1976
+ });
1977
+ await storage.recordUsage(principal, {
1978
+ runId: run.id,
1979
+ model: modelId,
1980
+ inputTokens: usage.inputTokens,
1981
+ outputTokens: usage.outputTokens,
1982
+ ...(usage.cacheReadTokens ? { cacheReadTokens: usage.cacheReadTokens } : {}),
1983
+ ...(usage.cacheWriteTokens ? { cacheWriteTokens: usage.cacheWriteTokens } : {}),
1984
+ cost,
1985
+ });
1986
+ return { summary: text, compacted: live.length };
1987
+ }
1988
+ catch (e) {
1989
+ await storage
1990
+ .updateRun(run.id, {
1991
+ status: "errored",
1992
+ error: e instanceof Error ? e.message : String(e),
1993
+ endedAt: Date.now(),
1994
+ })
1995
+ .catch(() => { });
1996
+ throw e;
1997
+ }
1998
+ }
1999
+ async function loadHistory({ chatId, principal, }) {
2000
+ const messages = await storage.loadMessages(principal, chatId);
2001
+ return toClientMessages(messages);
2002
+ }
2003
+ async function listChats({ principal, limit, kind, }) {
2004
+ // No default kind filter — passing nothing returns ALL chats (unchanged
2005
+ // behavior). Callers that want only user-facing chats pass kind: "user".
2006
+ return storage.listChats(principal, { limit, kind });
2007
+ }
2008
+ async function listRuns({ principal, limit, }) {
2009
+ return storage.listRuns(principal, { limit });
2010
+ }
2011
+ async function getChatById({ principal, id, }) {
2012
+ return storage.getChat(principal, id);
2013
+ }
2014
+ async function getRun({ principal, runId, }) {
2015
+ const run = await storage.getRun(runId);
2016
+ if (!run)
2017
+ return null;
2018
+ // Ownership: the run is the caller's only if its chat is (getChat is
2019
+ // principal-scoped). Guards cross-tenant reads/writes keyed by runId.
2020
+ const chat = await storage.getChat(principal, run.chatId);
2021
+ return chat ? run : null;
2022
+ }
2023
+ /** A fresh user message from plain text (for cron-fired turns). */
2024
+ function userMessageOf(text) {
2025
+ return {
2026
+ id: generateMessageId(),
2027
+ role: "user",
2028
+ parts: [{ type: "text", text }],
2029
+ metadata: { visibility: "user", createdAt: Date.now() },
2030
+ };
2031
+ }
2032
+ /** Merged agent directory (code-declared + dynamic) — name/title/description. */
2033
+ async function agentInfos(principal) {
2034
+ const stat = Object.entries(config.agents).map(([name, factory]) => ({
2035
+ name,
2036
+ title: factory.meta?.title,
2037
+ description: factory.meta?.description,
2038
+ hidden: factory.meta?.hidden,
2039
+ }));
2040
+ const dyn = (await config.dynamicAgents?.list(principal)) ?? [];
2041
+ // Dedup by name, static first — mirrors resolve-time precedence (a
2042
+ // code-declared agent shadows a same-named dynamic one).
2043
+ const seen = new Set(stat.map((a) => a.name));
2044
+ return [...stat, ...dyn.filter((a) => !seen.has(a.name) && seen.add(a.name))];
2045
+ }
2046
+ /** Concatenated text of the last assistant message in a chat (subagent answer). */
2047
+ function lastAssistantText(messages) {
2048
+ const last = [...messages].reverse().find((m) => m.role === "assistant");
2049
+ if (!last)
2050
+ return "";
2051
+ return (last.parts ?? [])
2052
+ .filter((p) => p.type === "text")
2053
+ .map((p) => p.text)
2054
+ .join("");
2055
+ }
2056
+ /**
2057
+ * Run one agent turn to completion, server-side, and report how it ended —
2058
+ * THE single non-streaming path. `runAgent` (programmatic / test runs),
2059
+ * `spawn_agent` (subagent delegation) and `runDue` (scheduled fires) all
2060
+ * funnel here, so approval semantics, telemetry linkage and the "did it
2061
+ * park?" verdict cannot drift between them.
2062
+ *
2063
+ * `chatId` absent (or "") → a fresh internal `sub_…` thread; present → that
2064
+ * chat continues with its full history (created as an internal thread when
2065
+ * it doesn't exist yet). `autoApprove` is REQUIRED, not defaulted: autonomous
2066
+ * runs (subagents, tests) waive HITL gates, while a scheduled fire keeps them
2067
+ * — and parks on them, which is the whole point of `parked`.
2068
+ *
2069
+ * `parked` = the turn ended waiting on a human (an undecided approval or
2070
+ * client tool result) rather than with an answer; the stream ending is not
2071
+ * the same as an answer, and a parked unattended turn has spent its tokens
2072
+ * and produced nothing. `pending` lifts an interactive subagent's undecided
2073
+ * approvals so `spawn_agent` can bubble them up to the parent turn. Both are
2074
+ * read from the fold over the persisted log — the same truth every
2075
+ * interactive path trusts — not the run row.
2076
+ */
2077
+ async function runToCompletion(o) {
2078
+ // `||` (not `??`): an empty-string id must mint a fresh one, or the chat is
2079
+ // created with id "" and the UI can never open it.
2080
+ const chatId = o.chatId?.trim() || generateSubchatId();
2081
+ let chat = await storage.getChat(o.principal, chatId);
2082
+ if (!chat) {
2083
+ chat = await storage.createChat(o.principal, {
2084
+ id: chatId,
2085
+ agent: o.agent,
2086
+ title: firstText(o.message)?.slice(0, 80),
2087
+ // Runtime-created thread: stamp it internal so a host filtering its
2088
+ // chat list on kind never shows it (see ChatRecord.kind).
2089
+ kind: "internal",
2090
+ });
2091
+ }
2092
+ const prior = await storage.loadMessages(o.principal, chatId);
2093
+ // Same rule as handleChat: a new message while the thread is parked on an
2094
+ // approval IS the decision — close the stale gate as declined (persisted)
2095
+ // and carry on. Otherwise a continued thread (`runAgent({ threadId })`,
2096
+ // `spawn_agent({ threadId })`) would run with an undecided gate in history
2097
+ // and this turn's verdict below would re-report that stale approval.
2098
+ if (hasPendingApproval(prior)) {
2099
+ const declined = declinePendingApprovals(prior);
2100
+ if (declined.length > 0)
2101
+ await storage.appendMessages(chatId, declined);
2102
+ }
2103
+ // Gate before the message persists — a refused schedule/subagent turn
2104
+ // leaves nothing behind, same as handleChat. Same for a busy chat: open
2105
+ // the run first, persist the message only once it is ours to run.
2106
+ const admitted = await admitTurn(o.principal, o.agent, o.trigger);
2107
+ const { cfg, modelId, runtimeCtx, run, lease } = await withReservation(admitted, async () => {
2108
+ const resolved = await resolveAgentConfig(o.agent, o.principal, undefined, o.turnContext);
2109
+ const opened = await openTurn(o.principal, chatId, o.agent, [o.message]);
2110
+ return { ...resolved, ...opened };
2111
+ });
2112
+ const onProgress = o.onProgress;
2113
+ const res = await buildTurn({
2114
+ resolvedChatId: chatId,
2115
+ principal: o.principal,
2116
+ modelId,
2117
+ cfg,
2118
+ run,
2119
+ lease,
2120
+ uiMessages: [...prior, o.message],
2121
+ runtimeCtx,
2122
+ admitted,
2123
+ autoApprove: o.autoApprove,
2124
+ blockGated: o.blockGated,
2125
+ promptCaching: o.promptCaching,
2126
+ depth: o.depth ?? 0,
2127
+ parentRunId: o.parentRunId,
2128
+ rootChatId: o.rootChatId,
2129
+ trigger: o.trigger,
2130
+ // Forward progressive UI-message snapshots to the caller (spawn_agent
2131
+ // streams them into the parent turn). Observation must never break the
2132
+ // run — the primary stream is drained independently below.
2133
+ observeStream: onProgress
2134
+ ? (observed) => {
2135
+ void (async () => {
2136
+ try {
2137
+ const byId = new Map();
2138
+ for await (const m of readUIMessageStream({
2139
+ stream: observed,
2140
+ })) {
2141
+ const msg = m;
2142
+ byId.set(msg.id, msg);
2143
+ onProgress([o.message, ...byId.values()]);
2144
+ }
2145
+ }
2146
+ catch {
2147
+ // ignore — live progress is best-effort
2148
+ }
2149
+ })();
2150
+ }
2151
+ : undefined,
2152
+ });
2153
+ await res.text();
2154
+ const after = await storage.loadMessages(o.principal, chatId);
2155
+ // The verdict is about THIS turn, so read the FINAL message only (the same
2156
+ // rule hasPendingApproval applies): an undecided gate in an older message
2157
+ // is an abandoned turn, not this one's pause point.
2158
+ const last = after[after.length - 1];
2159
+ const pending = last?.role === "assistant" ? collectPendingApprovals([last]) : [];
2160
+ const parked = pending.length > 0 ||
2161
+ hasPendingApproval(after) ||
2162
+ (!!last && hasPendingToolResult([last]));
2163
+ return {
2164
+ chatId,
2165
+ answer: pending.length > 0 ? "" : lastAssistantText(after),
2166
+ parked,
2167
+ ...(pending.length > 0 ? { pending } : {}),
2168
+ };
2169
+ }
2170
+ /**
2171
+ * Resume a subagent thread after its bubbled-up approval was decided: apply
2172
+ * the decision to the sub's gated tool, re-run the sub from history (the
2173
+ * approved tool executes, or returns denied), and return the sub's digest.
2174
+ *
2175
+ * If the resumed sub pauses AGAIN on a new gated tool, its pending
2176
+ * approval(s) are returned so the caller (`applyApproval`) can re-bubble
2177
+ * them to the parent as a fresh `data-subagent-approval` part — bubbling
2178
+ * recurses one round per decision round-trip.
2179
+ */
2180
+ async function resumeSubagentApproval(principal, subChatId, decisions,
2181
+ /** Telemetry linkage of the original sub-turn, so the trace stays nested. */
2182
+ linkage) {
2183
+ const subChat = await storage.getChat(principal, subChatId);
2184
+ if (!subChat)
2185
+ return { answer: "" };
2186
+ const subMessages = await storage.loadMessages(principal, subChatId);
2187
+ const changed = applyDecisions(subMessages, decisions);
2188
+ if (changed.length > 0)
2189
+ await storage.appendMessages(subChatId, changed);
2190
+ // Another gated call in the same sub step still undecided → can't finish.
2191
+ if (hasPendingApproval(subMessages))
2192
+ return { answer: "" };
2193
+ const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(subChat.agent, principal);
2194
+ const admitted = await admitContinuation(principal, subChat.agent, "decision");
2195
+ const { run, lease } = await openRun(principal, subChatId, subChat.agent);
2196
+ const res = await buildTurn({
2197
+ resolvedChatId: subChatId,
2198
+ principal,
2199
+ modelId,
2200
+ cfg,
2201
+ run,
2202
+ admitted,
2203
+ lease,
2204
+ // Resume: same prefill guard as applyApproval — the paused sub turn may
2205
+ // have narrated after its gated call, and that text must not reach the
2206
+ // model as an assistant prefill.
2207
+ uiMessages: trimTrailingAssistantPrefill(subMessages),
2208
+ runtimeCtx,
2209
+ autoApprove: false, // keep gating any further writes in the sub
2210
+ depth: 1,
2211
+ // Keep the resumed turn nested under the original trace/session.
2212
+ parentRunId: linkage?.parentRunId,
2213
+ rootChatId: linkage?.rootChatId,
2214
+ // Reached only by a human deciding the bubbled-up approval.
2215
+ trigger: "decision",
2216
+ });
2217
+ await res.text();
2218
+ const after = await storage.loadMessages(principal, subChatId);
2219
+ // Paused on a SECOND gated tool → lift it so the caller re-bubbles.
2220
+ const pending = collectPendingApprovals(after);
2221
+ if (pending.length > 0)
2222
+ return { answer: "", pending };
2223
+ return { answer: lastAssistantText(after) };
2224
+ }
2225
+ async function schedule({ principal, agent, cron, timezone, prompt, delivery, onParked, }) {
2226
+ // Accept code-declared or dynamic (UI-created) agents.
2227
+ const known = !!config.agents[agent] || !!(await config.dynamicAgents?.resolve(agent, principal));
2228
+ if (!known)
2229
+ throw new Error(`Unknown agent: ${agent}`);
2230
+ const now = Date.now();
2231
+ let nextRunAt = now; // one-shot → fire on the next runDue
2232
+ if (cron) {
2233
+ const next = config.cron?.(cron, now, timezone);
2234
+ if (next == null) {
2235
+ throw new Error("Recurring schedules need a `cron` evaluator in the runtime config (or the expression never fires)");
2236
+ }
2237
+ nextRunAt = next;
2238
+ }
2239
+ return storage.createSchedule(principal, {
2240
+ agent,
2241
+ cron,
2242
+ timezone,
2243
+ prompt,
2244
+ nextRunAt,
2245
+ delivery,
2246
+ onParked,
2247
+ });
2248
+ }
2249
+ async function runDue(opts) {
2250
+ const now = opts?.now ?? Date.now();
2251
+ const claimed = await storage.claimDueSchedules(now, dur.leaseTtlMs, dur.instanceId);
2252
+ let fired = 0;
2253
+ let errors = 0;
2254
+ let skipped = 0;
2255
+ let parked = 0;
2256
+ for (const s of claimed) {
2257
+ const next = s.cron ? (config.cron?.(s.cron, now, s.timezone) ?? null) : null;
2258
+ // Consume the occurrence BEFORE firing, not after. The claim lease is
2259
+ // stamped once (never heartbeated), so while a fire is in flight the row
2260
+ // still reads `enabled AND nextRunAt <= now` — and once the lease TTL
2261
+ // (default 30s) lapses, ANY claimer re-fires the same occurrence: a
2262
+ // second node, a deploy overlap, the POST /cron/run backstop, even this
2263
+ // same instance's next tick. A scheduled agent turn routinely outlives
2264
+ // the TTL, and each duplicate is a full paid turn. Advancing first makes
2265
+ // the occurrence at-most-once: after this write the row is no longer due
2266
+ // (or no longer enabled, for one-shots), so it cannot be claimed again
2267
+ // regardless of lease state. Run-level durability — not an occurrence
2268
+ // retry — is what recovers a fire that dies mid-turn: the run row
2269
+ // survives and `sweep()` resumes it. For work that is billed per
2270
+ // attempt, "ran twice" is strictly worse than "ran once, resumed".
2271
+ try {
2272
+ await storage.completeSchedule(s.id, { nextRunAt: next ?? now, lastRunAt: now, enabled: next != null }, dur.instanceId);
2273
+ }
2274
+ catch (e) {
2275
+ // Cannot consume the occurrence → do NOT fire. Firing anyway would
2276
+ // leave the row due, and every subsequent tick would fire it again —
2277
+ // the swallowed version of this write was itself a repeat-fire bug.
2278
+ console.error(`[schedule] failed to advance ${s.id}; skipping this fire:`, e);
2279
+ errors++;
2280
+ continue;
2281
+ }
2282
+ try {
2283
+ const principal = await reconstructPrincipal(s.identity);
2284
+ // Parked-predecessor guard: when the LAST fire's chat is still waiting
2285
+ // on a human (an undecided approval or tool result), firing again just
2286
+ // mints another parked chat and pays for the tokens up to the gate —
2287
+ // on every tick, forever. `onParked: "skip"` (the default) consumes
2288
+ // the occurrence without running instead; the answer is one approval
2289
+ // away in the existing chat, and the next occurrence re-checks. The
2290
+ // truth is the message-log fold, not run rows — same predicates the
2291
+ // interactive paths trust. Best-effort: a failed check must never
2292
+ // block a fire.
2293
+ if ((s.onParked ?? "skip") === "skip" && s.lastChatId) {
2294
+ try {
2295
+ const prior = await storage.loadMessages(principal, s.lastChatId);
2296
+ const last = prior[prior.length - 1];
2297
+ if (hasPendingApproval(prior) || (last && hasPendingToolResult([last]))) {
2298
+ console.warn(`[schedule] ${s.id} (${s.agent}): previous fire's chat ${s.lastChatId} ` +
2299
+ `is still waiting on a human — skipping this occurrence (onParked: skip)`);
2300
+ skipped++;
2301
+ continue;
2302
+ }
2303
+ }
2304
+ catch (e) {
2305
+ console.error(`[schedule] ${s.id}: parked check failed; firing anyway:`, e);
2306
+ }
2307
+ }
2308
+ // Give the host first refusal (e.g. run it in the user's channel thread
2309
+ // and stream the answer there). A hook that throws is treated as "not
2310
+ // handled": the fire still happens the default way, so a broken
2311
+ // delivery target degrades to a normal chat run instead of losing the
2312
+ // occurrence entirely. Hosts: only throw BEFORE doing paid work — a
2313
+ // hook that already ran the agent and then throws (say, on delivery)
2314
+ // makes the fallback a second full turn for the same occurrence.
2315
+ let handled = false;
2316
+ if (config.runSchedule) {
2317
+ try {
2318
+ const outcome = await config.runSchedule({ schedule: s, principal });
2319
+ handled = typeof outcome === "boolean" ? outcome : outcome.handled;
2320
+ // The host owns the chat a channel fire ran in, so it is the only
2321
+ // one that can report it. Same accounting as the default path when
2322
+ // it does: stamp the chat for the next `onParked` check, and count
2323
+ // a park that nobody can answer.
2324
+ if (handled && typeof outcome === "object") {
2325
+ if (outcome.chatId) {
2326
+ await storage.noteScheduleFire?.(s.id, {
2327
+ lastChatId: outcome.chatId,
2328
+ firedAt: now,
2329
+ });
2330
+ }
2331
+ if (outcome.parked) {
2332
+ parked++;
2333
+ console.warn(`[schedule] ${s.id} (${s.agent}): host-delivered fire parked waiting on a ` +
2334
+ `human${outcome.chatId ? ` in chat ${outcome.chatId}` : ""} — no answer was delivered`);
2335
+ }
2336
+ }
2337
+ }
2338
+ catch (e) {
2339
+ console.error(`[schedule] runSchedule hook failed for ${s.id}:`, e);
2340
+ }
2341
+ }
2342
+ if (!handled) {
2343
+ const chat = await storage.createChat(principal, {
2344
+ agent: s.agent,
2345
+ title: s.prompt.slice(0, 80),
2346
+ });
2347
+ // Remember where this fire ran so the NEXT occurrence can see a
2348
+ // parked predecessor. Stamped before the turn (a parked turn still
2349
+ // "completes" the await); optional — see StorageAdapter. `firedAt`
2350
+ // is the lastRunAt this fire's completeSchedule wrote above, so a
2351
+ // delayed stamp can't clobber a newer occurrence's.
2352
+ await storage.noteScheduleFire?.(s.id, { lastChatId: chat.id, firedAt: now });
2353
+ const outcome = await runToCompletion({
2354
+ principal,
2355
+ agent: s.agent,
2356
+ message: userMessageOf(s.prompt),
2357
+ chatId: chat.id,
2358
+ // A cron fire keeps HITL gates: nobody is here to waive them, and a
2359
+ // gated tool parks the turn (reported below) instead of running.
2360
+ autoApprove: false,
2361
+ trigger: "schedule",
2362
+ });
2363
+ if (outcome.parked) {
2364
+ // Say it on THIS tick, not the next one: a one-shot has no next
2365
+ // occurrence, and even a cron's operator shouldn't have to wait
2366
+ // for the skip message to learn that a paid fire produced nothing.
2367
+ parked++;
2368
+ console.warn(`[schedule] ${s.id} (${s.agent}): fire parked waiting on a human in ` +
2369
+ `chat ${chat.id} — no answer was delivered`);
2370
+ }
2371
+ }
2372
+ fired++;
2373
+ }
2374
+ catch (e) {
2375
+ // The occurrence is already consumed (above), so this cannot repeat-
2376
+ // fire — but a silent `errors++` left hosts unable to see WHY a
2377
+ // schedule produced nothing.
2378
+ console.error(`[schedule] fire failed for ${s.id} (${s.agent}):`, e);
2379
+ errors++;
2380
+ }
2381
+ }
2382
+ return { fired, errors, skipped, parked };
2383
+ }
2384
+ const self = {
2385
+ listAgents: agentInfos,
2386
+ handleChat,
2387
+ runAgent: async ({ principal, agent, prompt, threadId, blockGated = true, turnContext, promptCaching,
2388
+ // No client stream: assume unattended unless the host says otherwise, so
2389
+ // a park here can't be silently filtered out as "someone's watching".
2390
+ trigger = "programmatic", }) => {
2391
+ const r = await runToCompletion({
2392
+ principal,
2393
+ agent,
2394
+ message: userMessageOf(prompt),
2395
+ chatId: threadId,
2396
+ depth: 1,
2397
+ // Autonomous: no human to authorize (blockGated stubs the gated tools).
2398
+ autoApprove: true,
2399
+ blockGated,
2400
+ turnContext,
2401
+ promptCaching,
2402
+ trigger,
2403
+ });
2404
+ return { threadId: r.chatId, answer: r.answer };
2405
+ },
2406
+ compactChat,
2407
+ loadHistory,
2408
+ listChats,
2409
+ getChat: getChatById,
2410
+ listRuns,
2411
+ getRun,
2412
+ resume: resumeRun,
2413
+ sweep,
2414
+ cron,
2415
+ applyApproval,
2416
+ applyToolResult,
2417
+ schedule,
2418
+ listSchedules: ({ principal }) => storage.listSchedules(principal),
2419
+ unschedule: ({ principal, id }) => storage.deleteSchedule(principal, id),
2420
+ runScheduleNow: ({ principal, id }) => storage.bumpSchedule(principal, id),
2421
+ runDue,
2422
+ };
2423
+ return self;
2424
+ }
2425
+ //# sourceMappingURL=runtime.js.map