@intentface/latch-core 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +45 -0
- package/dist/agent.d.ts +200 -0
- package/dist/agent.d.ts.map +1 -0
- package/dist/agent.js +9 -0
- package/dist/agent.js.map +1 -0
- package/dist/compaction.d.ts +33 -0
- package/dist/compaction.d.ts.map +1 -0
- package/dist/compaction.js +104 -0
- package/dist/compaction.js.map +1 -0
- package/dist/connections.d.ts +16 -0
- package/dist/connections.d.ts.map +1 -0
- package/dist/connections.js +41 -0
- package/dist/connections.js.map +1 -0
- package/dist/context.d.ts +41 -0
- package/dist/context.d.ts.map +1 -0
- package/dist/context.js +25 -0
- package/dist/context.js.map +1 -0
- package/dist/current-date.d.ts +12 -0
- package/dist/current-date.d.ts.map +1 -0
- package/dist/current-date.js +25 -0
- package/dist/current-date.js.map +1 -0
- package/dist/extensions.d.ts +333 -0
- package/dist/extensions.d.ts.map +1 -0
- package/dist/extensions.js +569 -0
- package/dist/extensions.js.map +1 -0
- package/dist/harness/index.d.ts +17 -0
- package/dist/harness/index.d.ts.map +1 -0
- package/dist/harness/index.js +15 -0
- package/dist/harness/index.js.map +1 -0
- package/dist/harness/tools.d.ts +88 -0
- package/dist/harness/tools.d.ts.map +1 -0
- package/dist/harness/tools.js +296 -0
- package/dist/harness/tools.js.map +1 -0
- package/dist/harness/web-fetch.d.ts +47 -0
- package/dist/harness/web-fetch.d.ts.map +1 -0
- package/dist/harness/web-fetch.js +247 -0
- package/dist/harness/web-fetch.js.map +1 -0
- package/dist/index.d.ts +25 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +25 -0
- package/dist/index.js.map +1 -0
- package/dist/limits.d.ts +152 -0
- package/dist/limits.d.ts.map +1 -0
- package/dist/limits.js +97 -0
- package/dist/limits.js.map +1 -0
- package/dist/memory.d.ts +93 -0
- package/dist/memory.d.ts.map +1 -0
- package/dist/memory.js +13 -0
- package/dist/memory.js.map +1 -0
- package/dist/message.d.ts +46 -0
- package/dist/message.d.ts.map +1 -0
- package/dist/message.js +2 -0
- package/dist/message.js.map +1 -0
- package/dist/models/catalog.d.ts +56 -0
- package/dist/models/catalog.d.ts.map +1 -0
- package/dist/models/catalog.js +211 -0
- package/dist/models/catalog.js.map +1 -0
- package/dist/models/defaults.d.ts +23 -0
- package/dist/models/defaults.d.ts.map +1 -0
- package/dist/models/defaults.js +19 -0
- package/dist/models/defaults.js.map +1 -0
- package/dist/models/index.d.ts +23 -0
- package/dist/models/index.d.ts.map +1 -0
- package/dist/models/index.js +19 -0
- package/dist/models/index.js.map +1 -0
- package/dist/models/prompt-caching.d.ts +19 -0
- package/dist/models/prompt-caching.d.ts.map +1 -0
- package/dist/models/prompt-caching.js +18 -0
- package/dist/models/prompt-caching.js.map +1 -0
- package/dist/models/provider.d.ts +19 -0
- package/dist/models/provider.d.ts.map +1 -0
- package/dist/models/provider.js +22 -0
- package/dist/models/provider.js.map +1 -0
- package/dist/models/reasoning.d.ts +21 -0
- package/dist/models/reasoning.d.ts.map +1 -0
- package/dist/models/reasoning.js +59 -0
- package/dist/models/reasoning.js.map +1 -0
- package/dist/pricing.d.ts +52 -0
- package/dist/pricing.d.ts.map +1 -0
- package/dist/pricing.js +37 -0
- package/dist/pricing.js.map +1 -0
- package/dist/principal.d.ts +37 -0
- package/dist/principal.d.ts.map +1 -0
- package/dist/principal.js +30 -0
- package/dist/principal.js.map +1 -0
- package/dist/projections.d.ts +37 -0
- package/dist/projections.d.ts.map +1 -0
- package/dist/projections.js +128 -0
- package/dist/projections.js.map +1 -0
- package/dist/prompt-caching.d.ts +106 -0
- package/dist/prompt-caching.d.ts.map +1 -0
- package/dist/prompt-caching.js +165 -0
- package/dist/prompt-caching.js.map +1 -0
- package/dist/runtime.d.ts +670 -0
- package/dist/runtime.d.ts.map +1 -0
- package/dist/runtime.js +2425 -0
- package/dist/runtime.js.map +1 -0
- package/dist/scheduler.d.ts +31 -0
- package/dist/scheduler.d.ts.map +1 -0
- package/dist/scheduler.js +43 -0
- package/dist/scheduler.js.map +1 -0
- package/dist/storage.d.ts +389 -0
- package/dist/storage.d.ts.map +1 -0
- package/dist/storage.js +38 -0
- package/dist/storage.js.map +1 -0
- package/dist/telemetry.d.ts +155 -0
- package/dist/telemetry.d.ts.map +1 -0
- package/dist/telemetry.js +2 -0
- package/dist/telemetry.js.map +1 -0
- package/dist/vault-node.d.ts +20 -0
- package/dist/vault-node.d.ts.map +1 -0
- package/dist/vault-node.js +30 -0
- package/dist/vault-node.js.map +1 -0
- package/dist/vault.d.ts +62 -0
- package/dist/vault.d.ts.map +1 -0
- package/dist/vault.js +88 -0
- package/dist/vault.js.map +1 -0
- package/package.json +95 -0
package/dist/runtime.js
ADDED
|
@@ -0,0 +1,2425 @@
|
|
|
1
|
+
import { ToolLoopAgent, consumeStream, createIdGenerator, createUIMessageStream, createUIMessageStreamResponse, jsonSchema, readUIMessageStream, smoothStream, stepCountIs, tool, toUIMessageStream, } from "ai";
|
|
2
|
+
import { currentDateLine } from "./current-date.js";
|
|
3
|
+
import { COMPACTION_PART_TYPE, sliceAtCompaction, toClientMessages, toModelMessages, } from "./projections.js";
|
|
4
|
+
import { summarizeForCompaction } from "./compaction.js";
|
|
5
|
+
import { computeCost } from "./pricing.js";
|
|
6
|
+
import { defaultPromptCachingPlan, markLastFunctionTool, mergeProviderOptions, toolsetHash, } from "./prompt-caching.js";
|
|
7
|
+
import { LimitExceededError, exceededPolicy, remainingTokens, windowRef, } from "./limits.js";
|
|
8
|
+
/** Ids for assistant response messages — `msg_<random>`, matching storage ids. */
|
|
9
|
+
const generateMessageId = createIdGenerator({ prefix: "msg", separator: "_" });
|
|
10
|
+
/** Sub-thread chat ids start with `sub_` so the host can keep them out of the
|
|
11
|
+
* main chat list (they're nested under the parent's spawn_agent tool calls). */
|
|
12
|
+
const generateSubchatId = createIdGenerator({ prefix: "sub", separator: "_" });
|
|
13
|
+
/** Parent-level id for a subagent approval bubbled up via `spawn_agent`. */
|
|
14
|
+
const generateSubApprovalId = createIdGenerator({ prefix: "sapv", separator: "_" });
|
|
15
|
+
/** Recursion ceiling for subagent delegation. */
|
|
16
|
+
const MAX_SUBAGENT_DEPTH = 4;
|
|
17
|
+
/**
|
|
18
|
+
* Below this many messages in the live window, compaction isn't worth a model
|
|
19
|
+
* call: one exchange (user + assistant) plus the message asking for it summarizes
|
|
20
|
+
* to roughly what it replaces.
|
|
21
|
+
*/
|
|
22
|
+
const MIN_COMPACTABLE_MESSAGES = 4;
|
|
23
|
+
const newInstanceId = createIdGenerator({ prefix: "inst", separator: "_" });
|
|
24
|
+
/** One-line summary of a gated tool call, for the approval dialog. */
|
|
25
|
+
function summarizeApproval(toolName, input) {
|
|
26
|
+
let args = "";
|
|
27
|
+
try {
|
|
28
|
+
args = JSON.stringify(input ?? {});
|
|
29
|
+
}
|
|
30
|
+
catch {
|
|
31
|
+
args = "";
|
|
32
|
+
}
|
|
33
|
+
if (args.length > 200)
|
|
34
|
+
args = `${args.slice(0, 200)}…`;
|
|
35
|
+
return `${toolName}(${args})`;
|
|
36
|
+
}
|
|
37
|
+
/** The undecided (`approved === undefined`) approvals in a thread's messages. */
|
|
38
|
+
function collectPendingApprovals(messages) {
|
|
39
|
+
const out = [];
|
|
40
|
+
for (const m of messages) {
|
|
41
|
+
for (const p of (m.parts ?? [])) {
|
|
42
|
+
if (p.state === "approval-requested" &&
|
|
43
|
+
p.approval &&
|
|
44
|
+
p.approval.approved === undefined) {
|
|
45
|
+
const toolName = p.type.replace(/^tool-/, "");
|
|
46
|
+
out.push({
|
|
47
|
+
approvalId: p.approval.id,
|
|
48
|
+
toolName,
|
|
49
|
+
summary: summarizeApproval(toolName, p.input),
|
|
50
|
+
});
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
return out;
|
|
55
|
+
}
|
|
56
|
+
/** The concatenated text of a message's text parts (for a chat title). */
|
|
57
|
+
function firstText(message) {
|
|
58
|
+
const text = (message.parts ?? [])
|
|
59
|
+
.filter((p) => p.type === "text")
|
|
60
|
+
.map((p) => p.text)
|
|
61
|
+
.join(" ")
|
|
62
|
+
.trim();
|
|
63
|
+
return text.length > 0 ? text : undefined;
|
|
64
|
+
}
|
|
65
|
+
/** The text of the turn's incoming user message (for telemetry). */
|
|
66
|
+
/** The runtime's own user-role message: a compaction summary + boundary marker, not a turn. */
|
|
67
|
+
function isCompactionMarker(m) {
|
|
68
|
+
return (m.parts ?? []).some((p) => p.type === COMPACTION_PART_TYPE);
|
|
69
|
+
}
|
|
70
|
+
/**
|
|
71
|
+
* The model that ran THIS run, off the assistant message it produced (each one
|
|
72
|
+
* carries its run id). Falls back to the tail when nothing is correlated —
|
|
73
|
+
* the legacy shape — and to nothing when the run never produced a message.
|
|
74
|
+
*/
|
|
75
|
+
function ranAs(messages, runId) {
|
|
76
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
77
|
+
const m = messages[i];
|
|
78
|
+
if (m.role !== "assistant")
|
|
79
|
+
continue;
|
|
80
|
+
const owner = m.metadata?.runId;
|
|
81
|
+
if (owner === runId || owner === undefined)
|
|
82
|
+
return m.metadata?.model;
|
|
83
|
+
}
|
|
84
|
+
return undefined;
|
|
85
|
+
}
|
|
86
|
+
function lastUserText(messages) {
|
|
87
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
88
|
+
const m = messages[i];
|
|
89
|
+
if (m?.role === "user")
|
|
90
|
+
return firstText(m);
|
|
91
|
+
}
|
|
92
|
+
return undefined;
|
|
93
|
+
}
|
|
94
|
+
/**
|
|
95
|
+
* Thrown when work is attempted on a chat whose last turn is still paused
|
|
96
|
+
* awaiting a human — by `compactChat`, where the pause can also be a parked
|
|
97
|
+
* client tool call and there's no sensible way to summarize around it. Carries
|
|
98
|
+
* a stable `code` so the handler can map it to a 409 and the client can prompt
|
|
99
|
+
* the user to resolve it first.
|
|
100
|
+
*
|
|
101
|
+
* `handleChat` deliberately does NOT throw this: a new message supersedes the
|
|
102
|
+
* pause (see `declinePendingApprovals`).
|
|
103
|
+
*/
|
|
104
|
+
export class PendingApprovalError extends Error {
|
|
105
|
+
code = "pending_approval";
|
|
106
|
+
constructor(message) {
|
|
107
|
+
super(message);
|
|
108
|
+
this.name = "PendingApprovalError";
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
/**
|
|
112
|
+
* Is the chat parked on an undecided approval RIGHT NOW?
|
|
113
|
+
*
|
|
114
|
+
* Only the last assistant message counts. A paused turn's gated calls all live
|
|
115
|
+
* on the assistant message it paused on (a resume starts a new one), so that
|
|
116
|
+
* message is the live pause point. Undecided parts in OLDER messages are
|
|
117
|
+
* abandoned turns — `closeDanglingToolCalls` already fabricates a terminal
|
|
118
|
+
* result for them in the model projection, and stored history is never
|
|
119
|
+
* rewritten. Scanning all of history instead wedged chats permanently in both
|
|
120
|
+
* directions: `handleChat` refused every new message with PendingApprovalError,
|
|
121
|
+
* AND `applyApproval` returned an empty 200 without resuming, so no button
|
|
122
|
+
* press could clear it either.
|
|
123
|
+
*/
|
|
124
|
+
function hasPendingApproval(messages) {
|
|
125
|
+
const last = messages[messages.length - 1];
|
|
126
|
+
// Strictly the FINAL message: a later user message means the turn moved on.
|
|
127
|
+
if (last?.role !== "assistant")
|
|
128
|
+
return false;
|
|
129
|
+
for (const p of (last.parts ?? [])) {
|
|
130
|
+
if (p.state === "approval-requested" && p.approval?.approved === undefined) {
|
|
131
|
+
return true;
|
|
132
|
+
}
|
|
133
|
+
// A bubbled-up subagent approval still awaiting a decision.
|
|
134
|
+
if (p.type === "data-subagent-approval" && p.data && p.data.approved === undefined) {
|
|
135
|
+
return true;
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
return false;
|
|
139
|
+
}
|
|
140
|
+
/** Why an approval was closed without the user ever deciding it. */
|
|
141
|
+
const SUPERSEDED_REASON = "Superseded — the user sent a new message instead of responding to this request.";
|
|
142
|
+
/**
|
|
143
|
+
* Decline every undecided approval on the paused turn, because the user moved
|
|
144
|
+
* on: they typed a new message instead of pressing a button.
|
|
145
|
+
*
|
|
146
|
+
* Without this a chat could wedge for good. A gate the user CAN'T satisfy — a
|
|
147
|
+
* misconfigured `connect_<name>` whose OAuth never completes is the reported
|
|
148
|
+
* case — left `handleChat` refusing every message with `PendingApprovalError`,
|
|
149
|
+
* so the only escape was a decision UI that (on the web) had no deny button.
|
|
150
|
+
* Declining is always the safe direction: no gated tool runs, and the model
|
|
151
|
+
* sees a denial plus the new instruction, so the user can simply redirect it.
|
|
152
|
+
*
|
|
153
|
+
* Bubbled-up subagent approvals are marked `resolved` too, so
|
|
154
|
+
* `collectDecidedSubagentApprovals` won't later try to resume a sub thread the
|
|
155
|
+
* user has walked away from. Mutates in place; returns the changed messages.
|
|
156
|
+
*/
|
|
157
|
+
function declinePendingApprovals(messages) {
|
|
158
|
+
const last = messages[messages.length - 1];
|
|
159
|
+
if (last?.role !== "assistant")
|
|
160
|
+
return [];
|
|
161
|
+
const changed = new Set();
|
|
162
|
+
for (const p of (last.parts ?? [])) {
|
|
163
|
+
if (p.state === "approval-requested" && p.approval && p.approval.approved === undefined) {
|
|
164
|
+
p.state = "approval-responded";
|
|
165
|
+
p.approval = { ...p.approval, approved: false, reason: SUPERSEDED_REASON };
|
|
166
|
+
changed.add(last);
|
|
167
|
+
}
|
|
168
|
+
if (p.type === "data-subagent-approval" && p.data && p.data.approved === undefined) {
|
|
169
|
+
const { spawnToolCallId, subChatId } = p.data;
|
|
170
|
+
p.data = { ...p.data, approved: false, reason: SUPERSEDED_REASON, resolved: true };
|
|
171
|
+
changed.add(last);
|
|
172
|
+
// Close the parent's `spawn_agent` call too. Its stored result is the
|
|
173
|
+
// `awaiting_approval` payload, whose note tells the model to stop and
|
|
174
|
+
// wait for a decision that is never coming — and the model projection
|
|
175
|
+
// drops data parts, so that note is ALL the next turn would see of this.
|
|
176
|
+
if (spawnToolCallId) {
|
|
177
|
+
for (const m of overwriteToolResult(messages, spawnToolCallId, {
|
|
178
|
+
status: "declined",
|
|
179
|
+
threadId: subChatId,
|
|
180
|
+
note: SUPERSEDED_REASON,
|
|
181
|
+
})) {
|
|
182
|
+
changed.add(m);
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
return [...changed];
|
|
188
|
+
}
|
|
189
|
+
/**
|
|
190
|
+
* Record decisions onto matching `approval-requested` parts (→ `approval-responded`).
|
|
191
|
+
* Mutates in place and returns the messages that changed (to re-persist).
|
|
192
|
+
*/
|
|
193
|
+
function applyDecisions(messages, decisions) {
|
|
194
|
+
const byId = new Map(decisions.map((d) => [d.approvalId, d]));
|
|
195
|
+
const changed = [];
|
|
196
|
+
for (const m of messages) {
|
|
197
|
+
let touched = false;
|
|
198
|
+
for (const p of (m.parts ?? [])) {
|
|
199
|
+
if (p.state === "approval-requested" && p.approval && byId.has(p.approval.id)) {
|
|
200
|
+
const d = byId.get(p.approval.id);
|
|
201
|
+
p.state = "approval-responded";
|
|
202
|
+
p.approval = { ...p.approval, approved: d.approved, reason: d.reason };
|
|
203
|
+
touched = true;
|
|
204
|
+
}
|
|
205
|
+
// A bubbled-up subagent approval — record the decision on the data part;
|
|
206
|
+
// applyApproval routes it to the sub thread.
|
|
207
|
+
if (p.type === "data-subagent-approval" &&
|
|
208
|
+
p.data &&
|
|
209
|
+
byId.has(p.data.approvalId)) {
|
|
210
|
+
const d = byId.get(p.data.approvalId);
|
|
211
|
+
p.data = { ...p.data, approved: d.approved, reason: d.reason };
|
|
212
|
+
touched = true;
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
if (touched)
|
|
216
|
+
changed.push(m);
|
|
217
|
+
}
|
|
218
|
+
return changed;
|
|
219
|
+
}
|
|
220
|
+
/** Overwrite a tool call's result (regardless of current state) — used to fold
|
|
221
|
+
* a resumed subagent's answer back into the `spawn_agent` call on the parent. */
|
|
222
|
+
function overwriteToolResult(messages, toolCallId, output) {
|
|
223
|
+
const changed = [];
|
|
224
|
+
for (const m of messages) {
|
|
225
|
+
let touched = false;
|
|
226
|
+
for (const p of (m.parts ?? [])) {
|
|
227
|
+
if (p.toolCallId === toolCallId &&
|
|
228
|
+
(p.state === "output-available" || p.state === "input-available")) {
|
|
229
|
+
p.state = "output-available";
|
|
230
|
+
p.output = output;
|
|
231
|
+
touched = true;
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
if (touched)
|
|
235
|
+
changed.push(m);
|
|
236
|
+
}
|
|
237
|
+
return changed;
|
|
238
|
+
}
|
|
239
|
+
/** Decided-but-unresolved subagent approvals to route to their sub threads. */
|
|
240
|
+
function collectDecidedSubagentApprovals(messages) {
|
|
241
|
+
const out = [];
|
|
242
|
+
for (const m of messages) {
|
|
243
|
+
for (const p of (m.parts ?? [])) {
|
|
244
|
+
if (p.type === "data-subagent-approval" &&
|
|
245
|
+
p.data &&
|
|
246
|
+
p.data.approved !== undefined &&
|
|
247
|
+
!p.data.resolved) {
|
|
248
|
+
out.push(p.data);
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
return out;
|
|
253
|
+
}
|
|
254
|
+
/**
|
|
255
|
+
* Re-bubble: append a fresh `data-subagent-approval` part next to a resolved
|
|
256
|
+
* one (same parent assistant message), for a resumed subagent that paused on
|
|
257
|
+
* ANOTHER gated tool. The undecided part keeps the parent `awaiting_input`;
|
|
258
|
+
* deciding it round-trips through `applyApproval` again, so bubbling recurses
|
|
259
|
+
* for as many rounds as the sub needs.
|
|
260
|
+
*/
|
|
261
|
+
function appendSubagentApprovalPart(messages, afterApprovalId, data) {
|
|
262
|
+
for (const m of messages) {
|
|
263
|
+
const parts = (m.parts ?? []);
|
|
264
|
+
if (parts.some((p) => p.type === "data-subagent-approval" &&
|
|
265
|
+
p.data?.approvalId === afterApprovalId)) {
|
|
266
|
+
parts.push({
|
|
267
|
+
type: "data-subagent-approval",
|
|
268
|
+
id: data.approvalId,
|
|
269
|
+
data,
|
|
270
|
+
});
|
|
271
|
+
m.parts = parts;
|
|
272
|
+
return [m];
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
return [];
|
|
276
|
+
}
|
|
277
|
+
/** Mark a routed subagent approval resolved so it isn't re-processed. */
|
|
278
|
+
function markSubagentApprovalResolved(messages, approvalId) {
|
|
279
|
+
const changed = [];
|
|
280
|
+
for (const m of messages) {
|
|
281
|
+
let touched = false;
|
|
282
|
+
for (const p of (m.parts ?? [])) {
|
|
283
|
+
if (p.type === "data-subagent-approval" &&
|
|
284
|
+
p.data &&
|
|
285
|
+
p.data.approvalId === approvalId) {
|
|
286
|
+
p.data = { ...p.data, resolved: true };
|
|
287
|
+
touched = true;
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
if (touched)
|
|
291
|
+
changed.push(m);
|
|
292
|
+
}
|
|
293
|
+
return changed;
|
|
294
|
+
}
|
|
295
|
+
/**
|
|
296
|
+
* Has this chat opened another turn since `run` started? Then `run` was
|
|
297
|
+
* superseded — the conversation moved on while it sat reaped — and it must
|
|
298
|
+
* settle rather than run: re-running would generate against a conversation that
|
|
299
|
+
* has moved on and, because the message id is reused, append the result to the
|
|
300
|
+
* later turn's message.
|
|
301
|
+
*
|
|
302
|
+
* Read off the RUNS, not the messages. Every user message is persisted together
|
|
303
|
+
* with its run (`openTurn`), so a later run in the chat is the one fact that
|
|
304
|
+
* proves a later turn exists. Messages cannot tell us this: user messages carry
|
|
305
|
+
* no run id, and when a turn crashes before its first checkpoint the previous
|
|
306
|
+
* turn's assistant message is legitimately the last assistant in history —
|
|
307
|
+
* reading either as "someone else's" marks a live run superseded and drops the
|
|
308
|
+
* user's request without a model call, which is the one failure worse than the
|
|
309
|
+
* bug this all fixes.
|
|
310
|
+
*
|
|
311
|
+
* Compaction runs summarise the conversation so far; they are not turns.
|
|
312
|
+
*/
|
|
313
|
+
export function supersededBy(run, runs) {
|
|
314
|
+
return runs.some((r) => r.chatId === run.chatId && r.id !== run.id && r.kind !== "compaction" && r.startedAt > run.startedAt);
|
|
315
|
+
}
|
|
316
|
+
/**
|
|
317
|
+
* What a reclaimed run's OWN history says it needs. Pure: whether the chat has
|
|
318
|
+
* since moved on is `supersededBy`'s question, answered from the runs table.
|
|
319
|
+
*/
|
|
320
|
+
export function resumeShapeOf(messages) {
|
|
321
|
+
// Compaction markers are not turns, so the tail is read past them: a summary
|
|
322
|
+
// appended after a reaped run's finished message must not make that message
|
|
323
|
+
// look like it was followed by a user turn (which would say "continue" and
|
|
324
|
+
// regenerate against the summary). `compactChat` refuses while a gate or
|
|
325
|
+
// client tool is pending, so a marker can never sit after an undecided one.
|
|
326
|
+
let end = messages.length;
|
|
327
|
+
while (end > 0 && isCompactionMarker(messages[end - 1]))
|
|
328
|
+
end--;
|
|
329
|
+
const turns = end === messages.length ? messages : messages.slice(0, end);
|
|
330
|
+
// Checked next: a pending gate outranks the tail's shape, and its own tail
|
|
331
|
+
// (narration after the gated call) would otherwise read as `loop-finished`.
|
|
332
|
+
if (hasPendingApproval(turns))
|
|
333
|
+
return "decision-pending";
|
|
334
|
+
const last = turns[turns.length - 1];
|
|
335
|
+
// Nothing of this turn was checkpointed (or a user message came last) — the
|
|
336
|
+
// model has not spoken yet, so run it.
|
|
337
|
+
if (last?.role !== "assistant")
|
|
338
|
+
return "continue";
|
|
339
|
+
const parts = (last.parts ?? []);
|
|
340
|
+
let stepStart = -1;
|
|
341
|
+
for (let i = parts.length - 1; i >= 0; i--) {
|
|
342
|
+
if (parts[i].type === "step-start") {
|
|
343
|
+
stepStart = i;
|
|
344
|
+
break;
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
const finalStep = parts.slice(stepStart + 1);
|
|
348
|
+
// A step that opened and produced nothing: the crash landed inside it, so
|
|
349
|
+
// there is unfinished work regardless of what earlier steps did.
|
|
350
|
+
if (finalStep.length === 0)
|
|
351
|
+
return "continue";
|
|
352
|
+
const toolParts = finalStep.filter((p) => p.type.startsWith("tool-") || p.type === "dynamic-tool");
|
|
353
|
+
// A CLIENT-handled tool (no `execute`, e.g. askUser) is the one tool part a
|
|
354
|
+
// checkpoint can hold with no result: the step's other tools ran, this one is
|
|
355
|
+
// for the client to answer, and the loop stops because not every call has an
|
|
356
|
+
// output. `onFinish` records that turn as `completed` and the answer arrives
|
|
357
|
+
// later through `applyToolResult`, which opens its own run — so that is what
|
|
358
|
+
// recovery must record too. Re-running instead would make the projection
|
|
359
|
+
// close the call as "interrupted" and the model ask its question twice.
|
|
360
|
+
// (A provider-executed tool still pending is the loop continuing, not a
|
|
361
|
+
// client wait, so it must not match.)
|
|
362
|
+
const awaitingClient = toolParts.some((p) => p.state === "input-available" && typeof p.toolCallId === "string" && !p.providerExecuted);
|
|
363
|
+
if (awaitingClient)
|
|
364
|
+
return "loop-finished";
|
|
365
|
+
return toolParts.length > 0 ? "continue" : "loop-finished";
|
|
366
|
+
}
|
|
367
|
+
/**
|
|
368
|
+
* The resume paths ({@link applyApproval} / {@link applyToolResult}) re-run the
|
|
369
|
+
* model from history whose LAST message is the assistant turn that paused — no
|
|
370
|
+
* fresh user message is appended. If that turn streamed trailing user-facing
|
|
371
|
+
* text before pausing (e.g. "…ready to create the page but needs your approval
|
|
372
|
+
* first"), the model projection ends on an assistant message. Some providers
|
|
373
|
+
* reject that ("This model does not support assistant message prefill. The
|
|
374
|
+
* conversation must end with a user message.").
|
|
375
|
+
*
|
|
376
|
+
* Drop the trailing text/reasoning that follows the final TOOL part of the last
|
|
377
|
+
* assistant message, so the projection ends on the tool result (a user turn) and
|
|
378
|
+
* the model generates a fresh continuation. Returns a shallow copy touching only
|
|
379
|
+
* the last message.
|
|
380
|
+
*
|
|
381
|
+
* **The anchor must be a part the model projection keeps.** `toModelMessages`
|
|
382
|
+
* drops every `data-*` part, so anchoring on a bubbled `data-subagent-approval`
|
|
383
|
+
* kept the text BEFORE it — leaving the projection ending on assistant text,
|
|
384
|
+
* i.e. exactly the prefill this exists to prevent. That is the shape the
|
|
385
|
+
* agent-builder produces on every approval round: `… tool-spawn_agent,
|
|
386
|
+
* data-subagent-progress, data-subagent-approval, step-start, text,
|
|
387
|
+
* data-subagent-approval`. Only `tool-*` / `dynamic-tool` anchor.
|
|
388
|
+
*
|
|
389
|
+
* `data-*` parts after the anchor are KEPT: the resumed turn's `onFinish`
|
|
390
|
+
* re-persists this message, so removing them would erase the approval markers
|
|
391
|
+
* the UI renders and `hasPendingApproval` reads. Trailing `step-start` is
|
|
392
|
+
* DROPPED along with the text: it is invisible content, but it instructs
|
|
393
|
+
* `convertToModelMessages` to open a NEW assistant message — and with the text
|
|
394
|
+
* gone, that message is empty, so the projection would still end on the
|
|
395
|
+
* assistant (`{ role: "assistant", content: [] }`), trading the prefill
|
|
396
|
+
* rejection for an empty-content one. The only thing in that step was the
|
|
397
|
+
* narration being removed, so history loses nothing but an empty divider.
|
|
398
|
+
* The trailing text is dropped from stored history too — it is provisional
|
|
399
|
+
* commentary about a decision that has now been made; the resumed turn
|
|
400
|
+
* regenerates the real continuation.
|
|
401
|
+
*/
|
|
402
|
+
export function trimTrailingAssistantPrefill(messages, opts) {
|
|
403
|
+
if (messages.length === 0)
|
|
404
|
+
return messages;
|
|
405
|
+
const lastIndex = messages.length - 1;
|
|
406
|
+
const last = messages[lastIndex];
|
|
407
|
+
if (last.role !== "assistant")
|
|
408
|
+
return messages;
|
|
409
|
+
const parts = (last.parts ?? []);
|
|
410
|
+
// Last part that survives into the model projection as a turn boundary.
|
|
411
|
+
let anchor = -1;
|
|
412
|
+
for (let i = parts.length - 1; i >= 0; i--) {
|
|
413
|
+
const t = parts[i].type;
|
|
414
|
+
if (t.startsWith("tool-") || t === "dynamic-tool") {
|
|
415
|
+
anchor = i;
|
|
416
|
+
break;
|
|
417
|
+
}
|
|
418
|
+
}
|
|
419
|
+
// No tool anchor at all. On the DECISION paths that shape is not theirs to
|
|
420
|
+
// judge, so leave it (a bubbled `data-subagent-approval` message carries its
|
|
421
|
+
// narration and no `tool-*` part; trimming there would silently lose it).
|
|
422
|
+
//
|
|
423
|
+
// `unanchored` is belt and braces for CRASH RECOVERY, which cannot afford to
|
|
424
|
+
// hand the model a prefill under any tail shape: with no anchor every prefill
|
|
425
|
+
// part is dropped, which empties the message out of the model projection
|
|
426
|
+
// entirely. Today `resumeShapeOf` settles the one anchorless tail a
|
|
427
|
+
// checkpoint can actually produce (`[step-start, text]` — a tool-free turn,
|
|
428
|
+
// which is `loop-finished`) before the trim is reached, so this branch is a
|
|
429
|
+
// guard against a future classifier change, not a live path.
|
|
430
|
+
if (anchor === -1 && !opts?.unanchored)
|
|
431
|
+
return messages;
|
|
432
|
+
// step-start joins text/reasoning: see the doc comment — keeping it ends the
|
|
433
|
+
// projection on an EMPTY assistant message, which providers also reject.
|
|
434
|
+
const isPrefill = (t) => t === "text" || t === "reasoning" || t === "step-start";
|
|
435
|
+
const tail = parts.slice(anchor + 1);
|
|
436
|
+
if (!tail.some((p) => isPrefill(p.type)))
|
|
437
|
+
return messages;
|
|
438
|
+
const kept = [...parts.slice(0, anchor + 1), ...tail.filter((p) => !isPrefill(p.type))];
|
|
439
|
+
const trimmed = { ...last, parts: kept };
|
|
440
|
+
return [...messages.slice(0, lastIndex), trimmed];
|
|
441
|
+
}
|
|
442
|
+
/**
|
|
443
|
+
* True if any tool call is still awaiting a (client-supplied) result.
|
|
444
|
+
*
|
|
445
|
+
* Callers must pass only the messages of the step being resolved (see
|
|
446
|
+
* `applyToolResult`) — NOT full history. A client tool call abandoned turns ago
|
|
447
|
+
* stays `input-available` in stored history forever (`closeDanglingToolCalls`
|
|
448
|
+
* repairs the model projection, never the store), so an all-history scan would
|
|
449
|
+
* make every later resume in that chat return "still waiting" and never re-run.
|
|
450
|
+
*/
|
|
451
|
+
function hasPendingToolResult(messages) {
|
|
452
|
+
for (const m of messages) {
|
|
453
|
+
for (const p of (m.parts ?? [])) {
|
|
454
|
+
if (p.state === "input-available" && typeof p.toolCallId === "string")
|
|
455
|
+
return true;
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
return false;
|
|
459
|
+
}
|
|
460
|
+
/**
|
|
461
|
+
* Record a tool `output` onto the matching tool call (input-available →
|
|
462
|
+
* output-available). Mutates in place and returns the messages that changed.
|
|
463
|
+
*/
|
|
464
|
+
function recordToolResult(messages, toolCallId, output) {
|
|
465
|
+
// Only the live pause point may be answered. A parked client tool call always
|
|
466
|
+
// sits on the assistant message the turn paused on (a resume starts a new
|
|
467
|
+
// one), so anything older was abandoned and buried by a later user turn.
|
|
468
|
+
// Recording onto it would mutate dead history AND resume an obsolete turn —
|
|
469
|
+
// a stale button, or a redelivered submit for a long-dead question, would
|
|
470
|
+
// trigger a spurious agent run instead of doing nothing.
|
|
471
|
+
const last = messages[messages.length - 1];
|
|
472
|
+
if (last?.role !== "assistant")
|
|
473
|
+
return [];
|
|
474
|
+
let touched = false;
|
|
475
|
+
for (const p of (last.parts ?? [])) {
|
|
476
|
+
if (p.toolCallId === toolCallId && p.state === "input-available") {
|
|
477
|
+
p.state = "output-available";
|
|
478
|
+
p.output = output;
|
|
479
|
+
touched = true;
|
|
480
|
+
}
|
|
481
|
+
}
|
|
482
|
+
return touched ? [last] : [];
|
|
483
|
+
}
|
|
484
|
+
/** Run a hook fire-and-forget — a thrown/rejected hook is swallowed so
|
|
485
|
+
* telemetry never breaks a turn. */
|
|
486
|
+
function fireAndForget(fn) {
|
|
487
|
+
if (!fn)
|
|
488
|
+
return;
|
|
489
|
+
try {
|
|
490
|
+
void Promise.resolve(fn()).catch(() => { });
|
|
491
|
+
}
|
|
492
|
+
catch {
|
|
493
|
+
// synchronous throw from the hook — ignore
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
/**
|
|
497
|
+
* The one place an error becomes a client-visible string. Latch deliberately
|
|
498
|
+
* forwards `Error` messages — they are the debugging surface for sandbox and
|
|
499
|
+
* tool failures, and the audience is the org's own authenticated users — but
|
|
500
|
+
* every stream/tool error path must route through here so the exposure policy
|
|
501
|
+
* can be tightened in a single place. Non-`Error` throws and unbounded
|
|
502
|
+
* payloads never pass through verbatim.
|
|
503
|
+
*/
|
|
504
|
+
/** Keys whose values are masked when a non-`Error` throw is serialized. */
|
|
505
|
+
const SECRETISH_KEY = /token|secret|password|authorization|cookie|api[-_]?key|credential/i;
|
|
506
|
+
export function clientErrorMessage(error) {
|
|
507
|
+
const message = error instanceof Error ? error.message.trim() : "";
|
|
508
|
+
if (message)
|
|
509
|
+
return message.length > 500 ? `${message.slice(0, 499)}…` : message;
|
|
510
|
+
// A non-Error throw or an empty-message Error still carries signal — a raw
|
|
511
|
+
// HTTP body, a JSON-RPC error object, a string. Describe it instead of
|
|
512
|
+
// masking: the bare mask turned an MCP server's `404 {"error":"Session not
|
|
513
|
+
// found"}` into an undebuggable generic string. Serialization masks
|
|
514
|
+
// secret-shaped fields (a thrown response object can embed headers or
|
|
515
|
+
// tokens) and stays bounded; JSON.stringify can throw on cycles, hence the
|
|
516
|
+
// try. Exported so hosts render connection failures through the SAME
|
|
517
|
+
// exposure policy instead of reinventing (String(err) → "[object Object]").
|
|
518
|
+
if (error instanceof Error) {
|
|
519
|
+
const name = error.name?.trim();
|
|
520
|
+
if (name && name !== "Error")
|
|
521
|
+
return `${name.slice(0, 100)} (no message)`;
|
|
522
|
+
}
|
|
523
|
+
try {
|
|
524
|
+
const desc = typeof error === "string"
|
|
525
|
+
? error
|
|
526
|
+
: JSON.stringify(error, (key, value) => key && SECRETISH_KEY.test(key) ? "[redacted]" : value);
|
|
527
|
+
if (desc && desc !== "{}" && desc !== "null" && desc !== '""' && desc !== "undefined") {
|
|
528
|
+
return `An error occurred: ${desc.length > 300 ? `${desc.slice(0, 299)}…` : desc}`;
|
|
529
|
+
}
|
|
530
|
+
}
|
|
531
|
+
catch {
|
|
532
|
+
// circular or non-serializable — fall through to the mask
|
|
533
|
+
}
|
|
534
|
+
return "An error occurred.";
|
|
535
|
+
}
|
|
536
|
+
export function createRuntime(config) {
|
|
537
|
+
const { storage } = config;
|
|
538
|
+
const reconstructPrincipal = config.reconstructPrincipal ?? ((identity) => identity);
|
|
539
|
+
const dur = {
|
|
540
|
+
enabled: config.durability?.enabled ?? false,
|
|
541
|
+
leaseTtlMs: config.durability?.leaseTtlMs ?? 30_000,
|
|
542
|
+
heartbeatMs: config.durability?.heartbeatMs ?? 10_000,
|
|
543
|
+
instanceId: config.durability?.instanceId ?? newInstanceId(),
|
|
544
|
+
};
|
|
545
|
+
if (dur.enabled && !storage.openTurn) {
|
|
546
|
+
// Without `openTurn` a turn opens as two writes (run, then message), and a
|
|
547
|
+
// crash between them leaves a run whose own message never landed. Recovery
|
|
548
|
+
// then reads the PREVIOUS turn's tail as this run's history. Run
|
|
549
|
+
// correlation catches the common shape, but the contract is explicit that
|
|
550
|
+
// a durable adapter should be atomic here — every shipped adapter is.
|
|
551
|
+
console.warn("[latch] durability is enabled but the storage adapter has no openTurn(): " +
|
|
552
|
+
"turn creation is not atomic, so a crash can leave a run without its message. " +
|
|
553
|
+
"Implement openTurn — see the StorageAdapter contract.");
|
|
554
|
+
}
|
|
555
|
+
/** Resolve an agent's per-request config + a display model id. */
|
|
556
|
+
async function resolveAgentConfig(agentName, principal, request, turnContext) {
|
|
557
|
+
const runtimeCtx = await config.context.build({ principal, request, turnContext });
|
|
558
|
+
// Code-declared agents first; otherwise a dynamic (e.g. DB-stored) agent.
|
|
559
|
+
const factory = config.agents[agentName];
|
|
560
|
+
const cfg = factory
|
|
561
|
+
? await factory({ context: runtimeCtx, principal, turnContext })
|
|
562
|
+
: await config.dynamicAgents?.resolve(agentName, principal);
|
|
563
|
+
if (!cfg)
|
|
564
|
+
throw new Error(`Unknown agent: ${agentName}`);
|
|
565
|
+
const modelId = typeof cfg.model === "string"
|
|
566
|
+
? cfg.model
|
|
567
|
+
: (cfg.model.modelId ?? "unknown");
|
|
568
|
+
return { cfg, modelId, runtimeCtx };
|
|
569
|
+
}
|
|
570
|
+
/**
|
|
571
|
+
* Build (and start) one turn's stream — shared by a fresh chat turn and a
|
|
572
|
+
* resume. With durability on, it heartbeats the lease and aborts if it's
|
|
573
|
+
* lost; onFinish writes are fenced by `lease`. The caller decides whether to
|
|
574
|
+
* hand the stream to a client (handleChat) or drive it to completion
|
|
575
|
+
* server-side (resume).
|
|
576
|
+
*/
|
|
577
|
+
async function buildTurn(args) {
|
|
578
|
+
const { resolvedChatId, principal, modelId, cfg, run, lease, uiMessages, runtimeCtx } = args;
|
|
579
|
+
const admitted = args.admitted;
|
|
580
|
+
const autoApprove = args.autoApprove ?? false;
|
|
581
|
+
const blockGated = args.blockGated ?? false;
|
|
582
|
+
const depth = args.depth ?? 0;
|
|
583
|
+
const trigger = args.trigger ?? "chat";
|
|
584
|
+
// The top-level conversation id — this turn's own chat unless a spawn
|
|
585
|
+
// threaded a root down. Subagent turns spawned below inherit it, so a whole
|
|
586
|
+
// delegation tree shares one telemetry session (see telemetryMeta).
|
|
587
|
+
const rootChatId = args.rootChatId ?? resolvedChatId;
|
|
588
|
+
const abort = new AbortController();
|
|
589
|
+
let heartbeat;
|
|
590
|
+
const stopHeartbeat = () => {
|
|
591
|
+
if (heartbeat !== undefined) {
|
|
592
|
+
clearInterval(heartbeat);
|
|
593
|
+
heartbeat = undefined;
|
|
594
|
+
}
|
|
595
|
+
};
|
|
596
|
+
if (dur.enabled && lease) {
|
|
597
|
+
// A THROWN heartbeat write means "we don't know if we still hold the
|
|
598
|
+
// lease" — very different from `held === false` ("someone else owns this
|
|
599
|
+
// run"). Swallowing it silently let a run whose beats were all failing
|
|
600
|
+
// keep streaming at full token cost while the reaper reclaimed it and a
|
|
601
|
+
// second worker resumed the same turn. So: retry transient failures, but
|
|
602
|
+
// once consecutive failures span a full TTL the lease has provably
|
|
603
|
+
// expired — abort, exactly as if `held` had come back false.
|
|
604
|
+
let missedBeats = 0;
|
|
605
|
+
let beatInFlight = false;
|
|
606
|
+
const maxMissedBeats = Math.max(1, Math.ceil(dur.leaseTtlMs / dur.heartbeatMs));
|
|
607
|
+
const missBeat = (why, e) => {
|
|
608
|
+
missedBeats++;
|
|
609
|
+
console.warn(`[durability] heartbeat ${why} for run ${run.id} (${missedBeats}/${maxMissedBeats})`, e ?? "");
|
|
610
|
+
if (missedBeats >= maxMissedBeats) {
|
|
611
|
+
stopHeartbeat();
|
|
612
|
+
abort.abort(new Error("lease lost: heartbeats failing for a full TTL"));
|
|
613
|
+
}
|
|
614
|
+
};
|
|
615
|
+
heartbeat = setInterval(() => {
|
|
616
|
+
// A write that never settles is as blind as one that throws — count it
|
|
617
|
+
// against the TTL budget instead of stacking another write on top.
|
|
618
|
+
if (beatInFlight) {
|
|
619
|
+
missBeat("write still unsettled from the previous interval");
|
|
620
|
+
return;
|
|
621
|
+
}
|
|
622
|
+
beatInFlight = true;
|
|
623
|
+
void storage
|
|
624
|
+
.heartbeatRun(run.id, lease.owner, dur.leaseTtlMs, lease.fencingToken)
|
|
625
|
+
.then((held) => {
|
|
626
|
+
missedBeats = 0;
|
|
627
|
+
if (!held) {
|
|
628
|
+
stopHeartbeat();
|
|
629
|
+
abort.abort(new Error("lease lost"));
|
|
630
|
+
}
|
|
631
|
+
})
|
|
632
|
+
.catch((e) => missBeat("write failed", e))
|
|
633
|
+
.finally(() => {
|
|
634
|
+
beatInFlight = false;
|
|
635
|
+
});
|
|
636
|
+
}, dur.heartbeatMs);
|
|
637
|
+
// Don't let the heartbeat alone keep a Node process alive.
|
|
638
|
+
heartbeat.unref?.();
|
|
639
|
+
}
|
|
640
|
+
// Open dynamic tool sources (e.g. MCP) for this tenant, then merge their
|
|
641
|
+
// tools with the static ones. They're closed when the turn ends. If opening
|
|
642
|
+
// one fails, close any already opened so we don't leak connections.
|
|
643
|
+
// Effective tool sources: the agent's own, plus a registry-resolved source
|
|
644
|
+
// for its declared `connections` (its MCP servers / integrations).
|
|
645
|
+
const sources = [...(cfg.toolSources ?? [])];
|
|
646
|
+
if (config.connections && cfg.connections && cfg.connections.length > 0) {
|
|
647
|
+
sources.push(config.connections.hostFor(cfg.connections));
|
|
648
|
+
}
|
|
649
|
+
const opened = [];
|
|
650
|
+
try {
|
|
651
|
+
for (const source of sources) {
|
|
652
|
+
opened.push(await source.open(principal));
|
|
653
|
+
}
|
|
654
|
+
}
|
|
655
|
+
catch (error) {
|
|
656
|
+
await Promise.all(opened.map((o) => o.close().catch(() => { })));
|
|
657
|
+
stopHeartbeat();
|
|
658
|
+
throw error;
|
|
659
|
+
}
|
|
660
|
+
const closeSources = () => Promise.all(opened.map((o) => o.close().catch(() => { }))).then(() => undefined);
|
|
661
|
+
const tools = Object.assign({}, cfg.tools, ...opened.map((o) => o.tools));
|
|
662
|
+
// Default harness tools (eve-mirrored), merged per the agent's flags. The
|
|
663
|
+
// platform supplies them (core carries no sandbox/QuickJS dep): `sandbox`
|
|
664
|
+
// tools (bash/read_file/write_file/glob/grep) operate on the resolved
|
|
665
|
+
// Experimental_SandboxSession; `app` tools (web_fetch/todo/ask_question) run
|
|
666
|
+
// in the app process. `disableTools` opts any of them back out.
|
|
667
|
+
// Names that must stay callable even when `enabledTools` restricts the set
|
|
668
|
+
// (it only restricts *connection* tools): the agent's own declared tools,
|
|
669
|
+
// the harness tools, and `spawn_agent` are capabilities, not connection picks.
|
|
670
|
+
const capabilityTools = Object.keys(cfg.tools ?? {});
|
|
671
|
+
const harness = config.harnessTools ?? {};
|
|
672
|
+
// Gate defaults for the harness tools this agent actually gets (a policy
|
|
673
|
+
// naming an unattached tool would be dead weight in the merge below).
|
|
674
|
+
const harnessApproval = {};
|
|
675
|
+
const attachHarness = (name, t) => {
|
|
676
|
+
tools[name] = t;
|
|
677
|
+
capabilityTools.push(name);
|
|
678
|
+
const gate = harness.approval?.[name];
|
|
679
|
+
if (gate)
|
|
680
|
+
harnessApproval[name] = gate;
|
|
681
|
+
};
|
|
682
|
+
const wantSandbox = !!cfg.sandbox && !!config.sandbox;
|
|
683
|
+
if (wantSandbox && harness.sandbox) {
|
|
684
|
+
for (const [name, t] of Object.entries(harness.sandbox))
|
|
685
|
+
attachHarness(name, t);
|
|
686
|
+
}
|
|
687
|
+
if (cfg.defaultTools && harness.app) {
|
|
688
|
+
let appTools;
|
|
689
|
+
if (typeof harness.app === "function") {
|
|
690
|
+
try {
|
|
691
|
+
appTools = await harness.app({ principal, chatId: resolvedChatId, agent: run.agent });
|
|
692
|
+
}
|
|
693
|
+
catch (error) {
|
|
694
|
+
// Same teardown as a failing tool-source open above: the sources are
|
|
695
|
+
// already open and the lease heartbeat is running — release both
|
|
696
|
+
// before surfacing the factory error.
|
|
697
|
+
await closeSources();
|
|
698
|
+
stopHeartbeat();
|
|
699
|
+
throw error;
|
|
700
|
+
}
|
|
701
|
+
}
|
|
702
|
+
else {
|
|
703
|
+
appTools = harness.app;
|
|
704
|
+
}
|
|
705
|
+
const only = Array.isArray(cfg.defaultTools) ? new Set(cfg.defaultTools) : null;
|
|
706
|
+
for (const [name, t] of Object.entries(appTools)) {
|
|
707
|
+
if (!only || only.has(name))
|
|
708
|
+
attachHarness(name, t);
|
|
709
|
+
}
|
|
710
|
+
}
|
|
711
|
+
// Render tools — a separate capability (like sandbox), opted into via
|
|
712
|
+
// `renderTools: true`, never granted by `defaultTools`.
|
|
713
|
+
if (cfg.renderTools && harness.render) {
|
|
714
|
+
for (const [name, t] of Object.entries(harness.render))
|
|
715
|
+
attachHarness(name, t);
|
|
716
|
+
}
|
|
717
|
+
// Provider-native web tools (server-side search/fetch), if the agent opts in
|
|
718
|
+
// and the platform supplies a resolver for this model's provider.
|
|
719
|
+
if (cfg.providerWebTools && config.resolveProviderTools) {
|
|
720
|
+
for (const [name, t] of Object.entries(config.resolveProviderTools(modelId))) {
|
|
721
|
+
tools[name] = t;
|
|
722
|
+
capabilityTools.push(name);
|
|
723
|
+
}
|
|
724
|
+
}
|
|
725
|
+
// Memory seam: when the agent declares `memory` and the runtime has a
|
|
726
|
+
// provider, merge the scope-bound memory tools (capabilities, not
|
|
727
|
+
// connection picks — they must survive an `enabledTools` restriction) and
|
|
728
|
+
// append the provider's compiled-index block to the instructions. A
|
|
729
|
+
// provider failure degrades to a turn without memory, never a dead turn.
|
|
730
|
+
let instructions = cfg.instructions;
|
|
731
|
+
if (cfg.memory && config.memory) {
|
|
732
|
+
const memArgs = { principal, agent: run.agent, memory: cfg.memory };
|
|
733
|
+
try {
|
|
734
|
+
// Resolve both hooks BEFORE mutating the toolset, so a failure in
|
|
735
|
+
// either leaves the turn exactly as it was without memory.
|
|
736
|
+
const memoryTools = Object.entries(config.memory.toolsFor(memArgs));
|
|
737
|
+
const block = await config.memory.instructionsFor(memArgs);
|
|
738
|
+
for (const [name, t] of memoryTools) {
|
|
739
|
+
tools[name] = t;
|
|
740
|
+
capabilityTools.push(name);
|
|
741
|
+
}
|
|
742
|
+
if (block)
|
|
743
|
+
instructions = [instructions, block].filter(Boolean).join("\n\n");
|
|
744
|
+
}
|
|
745
|
+
catch {
|
|
746
|
+
// Memory is optional and degradable — continue without it.
|
|
747
|
+
}
|
|
748
|
+
}
|
|
749
|
+
for (const name of cfg.disableTools ?? [])
|
|
750
|
+
delete tools[name];
|
|
751
|
+
// Per-agent tool description overrides — swap the resolved tool's
|
|
752
|
+
// description so the model sees the agent's custom guidance.
|
|
753
|
+
if (cfg.toolDescriptions) {
|
|
754
|
+
for (const [name, description] of Object.entries(cfg.toolDescriptions)) {
|
|
755
|
+
const t = tools[name];
|
|
756
|
+
if (t && description)
|
|
757
|
+
tools[name] = { ...t, description };
|
|
758
|
+
}
|
|
759
|
+
}
|
|
760
|
+
// Subagent delegation: one `spawn_agent` tool that runs another agent and
|
|
761
|
+
// returns its final answer. Skipped past the recursion ceiling.
|
|
762
|
+
if (cfg.subagents?.length && depth < MAX_SUBAGENT_DEPTH) {
|
|
763
|
+
const allowed = new Set(cfg.subagents);
|
|
764
|
+
const infos = await agentInfos(principal);
|
|
765
|
+
const lines = cfg.subagents
|
|
766
|
+
.map((n) => {
|
|
767
|
+
const i = infos.find((a) => a.name === n);
|
|
768
|
+
return `- ${n}: ${i?.description ?? i?.title ?? "(no description)"}`;
|
|
769
|
+
})
|
|
770
|
+
.join("\n");
|
|
771
|
+
tools.spawn_agent = tool({
|
|
772
|
+
description: "Delegate a sub-task to one of your agents and get its final answer.\n" +
|
|
773
|
+
`Available agents:\n${lines}\n\n` +
|
|
774
|
+
"Omit threadId to start a new conversation; pass the threadId returned by a " +
|
|
775
|
+
"previous call to continue with the SAME agent (it keeps full context). " +
|
|
776
|
+
"Returns { threadId, answer }.",
|
|
777
|
+
inputSchema: jsonSchema({
|
|
778
|
+
type: "object",
|
|
779
|
+
properties: {
|
|
780
|
+
agent: { type: "string", description: "One of the available agent names above." },
|
|
781
|
+
prompt: { type: "string", description: "The task / message for the subagent." },
|
|
782
|
+
threadId: { type: "string", description: "Continue an existing subagent thread." },
|
|
783
|
+
},
|
|
784
|
+
required: ["agent", "prompt"],
|
|
785
|
+
additionalProperties: false,
|
|
786
|
+
}),
|
|
787
|
+
execute: async ({ agent, prompt, threadId }, options) => {
|
|
788
|
+
if (!allowed.has(agent))
|
|
789
|
+
return { error: `Not an allowed subagent: ${agent}` };
|
|
790
|
+
// Stream the delegated conversation live into the parent turn (one
|
|
791
|
+
// data part, reconciled in place by id). Pre-generate the sub-thread
|
|
792
|
+
// id so the very first write already carries it.
|
|
793
|
+
const writer = options.context?.writer;
|
|
794
|
+
// `||` (not `??`): strict-schema models fill optional params with "" — an
|
|
795
|
+
// empty threadId must mean "new thread", never become a real chat id.
|
|
796
|
+
const subChatId = threadId?.trim() || generateSubchatId();
|
|
797
|
+
let onProgress;
|
|
798
|
+
if (writer) {
|
|
799
|
+
const write = (messages) => {
|
|
800
|
+
writer.write({
|
|
801
|
+
type: "data-subagent-progress",
|
|
802
|
+
id: `subagent_${options.toolCallId ?? subChatId}`,
|
|
803
|
+
data: {
|
|
804
|
+
toolCallId: options.toolCallId,
|
|
805
|
+
agent,
|
|
806
|
+
threadId: subChatId,
|
|
807
|
+
messages,
|
|
808
|
+
},
|
|
809
|
+
});
|
|
810
|
+
};
|
|
811
|
+
write([]);
|
|
812
|
+
// Each write re-sends the whole snapshot — throttle the re-sends.
|
|
813
|
+
// The trailing tokens are covered by the tool RESULT (the client
|
|
814
|
+
// switches to the persisted thread once output lands).
|
|
815
|
+
let lastWrite = 0;
|
|
816
|
+
onProgress = (messages) => {
|
|
817
|
+
const now = Date.now();
|
|
818
|
+
if (now - lastWrite < 250)
|
|
819
|
+
return;
|
|
820
|
+
lastWrite = now;
|
|
821
|
+
write(messages);
|
|
822
|
+
};
|
|
823
|
+
}
|
|
824
|
+
// Propagate the PARENT turn's execution mode AS A SET: interactive →
|
|
825
|
+
// the subagent gates its tools (a pending write bubbles up here);
|
|
826
|
+
// autonomous → it auto-approves; a TEST run (blockGated) stays a test
|
|
827
|
+
// run all the way down, or a stubbed parent could delegate to a child
|
|
828
|
+
// whose gated tools run for real.
|
|
829
|
+
const result = await runToCompletion({
|
|
830
|
+
principal,
|
|
831
|
+
agent,
|
|
832
|
+
message: userMessageOf(prompt),
|
|
833
|
+
chatId: subChatId,
|
|
834
|
+
depth: depth + 1,
|
|
835
|
+
autoApprove,
|
|
836
|
+
blockGated,
|
|
837
|
+
onProgress,
|
|
838
|
+
// Telemetry linkage: this turn is the parent; carry the root chat
|
|
839
|
+
// down so the subagent's trace nests under this conversation.
|
|
840
|
+
parentRunId: run.id,
|
|
841
|
+
rootChatId,
|
|
842
|
+
// A subagent inherits what set the tree off, so a scheduled fire's
|
|
843
|
+
// delegate is still attributable to the schedule (see TurnTrigger).
|
|
844
|
+
trigger,
|
|
845
|
+
});
|
|
846
|
+
if (result.pending?.length) {
|
|
847
|
+
// The subagent paused on a gated tool. Surface it on THIS turn as a
|
|
848
|
+
// `data-subagent-approval` part (tagged with the sub thread + this
|
|
849
|
+
// spawn_agent call) so the turn ends `awaiting_input` and
|
|
850
|
+
// `applyApproval` can route the decision back → resume the sub →
|
|
851
|
+
// resume here. Raw sub output stays in the sub thread (isolation).
|
|
852
|
+
const approvalId = generateSubApprovalId();
|
|
853
|
+
writer?.write({
|
|
854
|
+
type: "data-subagent-approval",
|
|
855
|
+
id: approvalId,
|
|
856
|
+
data: {
|
|
857
|
+
approvalId,
|
|
858
|
+
subChatId: result.chatId,
|
|
859
|
+
subApprovalIds: result.pending.map((p) => p.approvalId),
|
|
860
|
+
spawnToolCallId: options.toolCallId,
|
|
861
|
+
summaries: result.pending.map((p) => p.summary),
|
|
862
|
+
// Persist the sub-turn's telemetry linkage so its resumed turn
|
|
863
|
+
// (after approval) stays nested under this conversation/run.
|
|
864
|
+
parentRunId: run.id,
|
|
865
|
+
rootChatId,
|
|
866
|
+
},
|
|
867
|
+
});
|
|
868
|
+
return {
|
|
869
|
+
status: "awaiting_approval",
|
|
870
|
+
threadId: result.chatId,
|
|
871
|
+
pending: result.pending.map((p) => p.summary),
|
|
872
|
+
note: "Awaiting the user's approval — do not retry; stop here, the user will decide and you'll continue automatically.",
|
|
873
|
+
};
|
|
874
|
+
}
|
|
875
|
+
return { threadId: result.chatId, answer: result.answer };
|
|
876
|
+
},
|
|
877
|
+
});
|
|
878
|
+
capabilityTools.push("spawn_agent");
|
|
879
|
+
}
|
|
880
|
+
// Subagent runs are autonomous — there's no human to authorize, so waive
|
|
881
|
+
// approvals and drop the OAuth `connect_` gates (they'd pause forever).
|
|
882
|
+
if (autoApprove) {
|
|
883
|
+
for (const k of Object.keys(tools))
|
|
884
|
+
if (k.startsWith("connect_"))
|
|
885
|
+
delete tools[k];
|
|
886
|
+
// Test runs: stub the approval-gated (mutating) tools so a trial has no
|
|
887
|
+
// real side effects. Gated set = connection defaults + the agent's policy.
|
|
888
|
+
if (blockGated) {
|
|
889
|
+
const gated = new Set();
|
|
890
|
+
for (const [k, v] of Object.entries(harnessApproval))
|
|
891
|
+
if (v)
|
|
892
|
+
gated.add(k);
|
|
893
|
+
for (const o of opened) {
|
|
894
|
+
for (const [k, v] of Object.entries(o.toolApproval ?? {}))
|
|
895
|
+
if (v)
|
|
896
|
+
gated.add(k);
|
|
897
|
+
}
|
|
898
|
+
if (cfg.toolApproval && typeof cfg.toolApproval !== "function") {
|
|
899
|
+
for (const [k, v] of Object.entries(cfg.toolApproval)) {
|
|
900
|
+
if (v)
|
|
901
|
+
gated.add(k);
|
|
902
|
+
else
|
|
903
|
+
gated.delete(k);
|
|
904
|
+
}
|
|
905
|
+
}
|
|
906
|
+
else if (typeof cfg.toolApproval === "function") {
|
|
907
|
+
// A function policy can't be enumerated — it may gate ANY tool. To keep
|
|
908
|
+
// the test side-effect-free, conservatively block every executable tool
|
|
909
|
+
// (connection, authored, and harness — `connect_` is already dropped).
|
|
910
|
+
for (const k of Object.keys(tools))
|
|
911
|
+
gated.add(k);
|
|
912
|
+
}
|
|
913
|
+
for (const name of gated) {
|
|
914
|
+
const t = tools[name];
|
|
915
|
+
if (t && typeof t.execute === "function") {
|
|
916
|
+
tools[name] = {
|
|
917
|
+
...t,
|
|
918
|
+
execute: async (input) => ({
|
|
919
|
+
__blocked: true,
|
|
920
|
+
tool: name,
|
|
921
|
+
input,
|
|
922
|
+
reason: "approval-gated tool blocked during test run (no real side effects)",
|
|
923
|
+
}),
|
|
924
|
+
};
|
|
925
|
+
}
|
|
926
|
+
}
|
|
927
|
+
}
|
|
928
|
+
}
|
|
929
|
+
// Merge the default approval policies — harness-tool gates and the
|
|
930
|
+
// source-contributed ones (connection defaults + synthetic `connect_<name>`
|
|
931
|
+
// gates) — with the agent's. The AGENT WINS for tools it names, so an agent
|
|
932
|
+
// can require approval on a GET or waive it on a mutating op; tools it
|
|
933
|
+
// doesn't mention keep the default. Only for the object form (a function
|
|
934
|
+
// policy is left as-is).
|
|
935
|
+
const sourceApproval = Object.assign({}, harnessApproval, ...opened.map((o) => o.toolApproval ?? {}));
|
|
936
|
+
const toolApproval = autoApprove
|
|
937
|
+
? undefined
|
|
938
|
+
: cfg.toolApproval && typeof cfg.toolApproval !== "function"
|
|
939
|
+
? { ...sourceApproval, ...cfg.toolApproval }
|
|
940
|
+
: Object.keys(sourceApproval).length > 0 && !cfg.toolApproval
|
|
941
|
+
? sourceApproval
|
|
942
|
+
: cfg.toolApproval;
|
|
943
|
+
// Which tools the model may call. `enabledTools` restricts only CONNECTION
|
|
944
|
+
// tools, so union in each source's `alwaysActive` (e.g. connect_<name>) AND
|
|
945
|
+
// the capability tools (the agent's own tools, harness web_fetch/sandbox,
|
|
946
|
+
// spawn_agent) — otherwise picking specific connection tools would hide the
|
|
947
|
+
// harness. Omitted → all tools active.
|
|
948
|
+
const alwaysActive = opened.flatMap((o) => o.alwaysActive ?? []);
|
|
949
|
+
const activeTools = cfg.enabledTools
|
|
950
|
+
? // Exclude capability names that `disableTools` removed from `tools` (they
|
|
951
|
+
// were pushed to capabilityTools before the delete) — never list a tool
|
|
952
|
+
// in activeTools that isn't in the resolved toolset.
|
|
953
|
+
Array.from(new Set([...cfg.enabledTools, ...alwaysActive, ...capabilityTools])).filter((n) => n in tools)
|
|
954
|
+
: undefined;
|
|
955
|
+
// Tool-loop ceiling: a number caps it; "unlimited" runs until the model
|
|
956
|
+
// stops on its own; omitted → SDK default (stepCountIs(20)).
|
|
957
|
+
const stepCap = cfg.maxSteps === "unlimited"
|
|
958
|
+
? stepCountIs(Number.MAX_SAFE_INTEGER)
|
|
959
|
+
: typeof cfg.maxSteps === "number"
|
|
960
|
+
? stepCountIs(cfg.maxSteps)
|
|
961
|
+
: undefined;
|
|
962
|
+
// Mid-turn limit: stop the tool loop once THIS turn's tokens reach the
|
|
963
|
+
// tightest remaining budget it was admitted with. The model's last text is
|
|
964
|
+
// kept; no further step runs. (Counters were read at admission — concurrent
|
|
965
|
+
// turns can overshoot by at most one turn each, which settlement records.)
|
|
966
|
+
const tokenBudget = admitted?.remainingTokens;
|
|
967
|
+
const budgetStop = tokenBudget !== undefined
|
|
968
|
+
? ({ steps }) => steps.reduce((n, st) => n + (st.usage?.inputTokens ?? 0) + (st.usage?.outputTokens ?? 0), 0) >=
|
|
969
|
+
tokenBudget
|
|
970
|
+
: undefined;
|
|
971
|
+
const stopWhen = budgetStop ? [stepCap ?? stepCountIs(20), budgetStop] : stepCap;
|
|
972
|
+
// Reasoning/thinking effort → provider-specific options (the platform maps
|
|
973
|
+
// it per provider + gates to reasoning-capable models). Called even without
|
|
974
|
+
// an effort so the hook can set safe defaults (e.g. lift a provider's tiny
|
|
975
|
+
// default max_tokens for models that think by default); an absent/unknown
|
|
976
|
+
// effort must be a thinking-config no-op in the hook.
|
|
977
|
+
const reasoning = config.resolveReasoningOptions
|
|
978
|
+
? config.resolveReasoningOptions(modelId, cfg.effort ?? "")
|
|
979
|
+
: undefined;
|
|
980
|
+
// Prompt-caching plan for this model's provider (platform hook — see
|
|
981
|
+
// `prompt-caching.ts` for the breakpoint layout). `promptCaching: false`
|
|
982
|
+
// (e.g. the control arm of an A/B comparison) skips it for this turn only.
|
|
983
|
+
const caching = args.promptCaching === false
|
|
984
|
+
? undefined
|
|
985
|
+
: config.resolvePromptCaching
|
|
986
|
+
? config.resolvePromptCaching(modelId, {
|
|
987
|
+
agent: run.agent,
|
|
988
|
+
memoryScoped: !!(cfg.memory && config.memory),
|
|
989
|
+
principal,
|
|
990
|
+
})
|
|
991
|
+
: // No hook → caching is ON by default for first-party Anthropic /
|
|
992
|
+
// OpenAI models (derived from the model object's `provider`).
|
|
993
|
+
// Measured at ~6x unit cost on MCP-heavy agents — too expensive to
|
|
994
|
+
// leave off for hosts that never found the hook. Opt out with
|
|
995
|
+
// `resolvePromptCaching: () => undefined`.
|
|
996
|
+
defaultPromptCachingPlan(cfg.model, run.agent);
|
|
997
|
+
if (caching?.toolProviderOptions) {
|
|
998
|
+
markLastFunctionTool(tools, caching.toolProviderOptions, activeTools);
|
|
999
|
+
}
|
|
1000
|
+
// Fingerprint the toolset the model will actually see — persisted on the
|
|
1001
|
+
// run so churn across a chat's turns (a cache invalidator) is queryable.
|
|
1002
|
+
const toolsHash = toolsetHash(activeTools ?? Object.keys(tools));
|
|
1003
|
+
const providerOptions = mergeProviderOptions(reasoning?.providerOptions, caching?.requestProviderOptions);
|
|
1004
|
+
// Telemetry seam: per-run AI SDK telemetry options + trace correlation
|
|
1005
|
+
// identity, resolved once per turn (undefined → nothing traced).
|
|
1006
|
+
const telemetryMeta = {
|
|
1007
|
+
runId: run.id,
|
|
1008
|
+
chatId: resolvedChatId,
|
|
1009
|
+
agent: run.agent,
|
|
1010
|
+
modelId,
|
|
1011
|
+
principal,
|
|
1012
|
+
startedAt: run.startedAt,
|
|
1013
|
+
depth,
|
|
1014
|
+
trigger,
|
|
1015
|
+
parentRunId: args.parentRunId,
|
|
1016
|
+
// Top-level turns are their own root; subagent turns inherit the root the
|
|
1017
|
+
// spawn threaded down, so a whole delegation tree shares one session id.
|
|
1018
|
+
rootChatId,
|
|
1019
|
+
};
|
|
1020
|
+
const telemetryOptions = config.telemetry?.telemetryFor?.(telemetryMeta);
|
|
1021
|
+
// A telemetry impl that defines `telemetryFor` but returns undefined is
|
|
1022
|
+
// opting THIS run out — so the span wrapper and finish/flush hooks must go
|
|
1023
|
+
// silent too, not just the AI SDK settings. If `telemetryFor` is absent
|
|
1024
|
+
// entirely, the impl simply isn't customizing settings, so tracing stays on.
|
|
1025
|
+
const telemetryOn = !!config.telemetry &&
|
|
1026
|
+
!(config.telemetry.telemetryFor && telemetryOptions === undefined);
|
|
1027
|
+
const agent = new ToolLoopAgent({
|
|
1028
|
+
model: cfg.model,
|
|
1029
|
+
// Every agent gets a trailing "today's date" line — after the memory
|
|
1030
|
+
// block, so it stays the tail. Day-granularity by design — see
|
|
1031
|
+
// `currentDateLine` for the prompt-cache reasoning. With a caching plan,
|
|
1032
|
+
// the instructions become a MARKED system message and the date line a
|
|
1033
|
+
// separate unmarked one, so the daily flip lands after the breakpoint
|
|
1034
|
+
// instead of invalidating it at midnight UTC.
|
|
1035
|
+
instructions: caching?.systemProviderOptions
|
|
1036
|
+
? [
|
|
1037
|
+
...(instructions
|
|
1038
|
+
? [
|
|
1039
|
+
{
|
|
1040
|
+
role: "system",
|
|
1041
|
+
content: instructions,
|
|
1042
|
+
providerOptions: caching.systemProviderOptions,
|
|
1043
|
+
},
|
|
1044
|
+
]
|
|
1045
|
+
: []),
|
|
1046
|
+
{ role: "system", content: currentDateLine() },
|
|
1047
|
+
]
|
|
1048
|
+
: [instructions, currentDateLine()].filter(Boolean).join("\n\n"),
|
|
1049
|
+
tools,
|
|
1050
|
+
toolApproval,
|
|
1051
|
+
stopWhen,
|
|
1052
|
+
activeTools,
|
|
1053
|
+
...(providerOptions ? { providerOptions } : {}),
|
|
1054
|
+
...(reasoning?.maxOutputTokens ? { maxOutputTokens: reasoning.maxOutputTokens } : {}),
|
|
1055
|
+
...(telemetryOptions ? { telemetry: telemetryOptions } : {}),
|
|
1056
|
+
});
|
|
1057
|
+
let inputTokens = 0;
|
|
1058
|
+
let outputTokens = 0;
|
|
1059
|
+
let cacheReadTokens = 0;
|
|
1060
|
+
let cacheWriteTokens = 0;
|
|
1061
|
+
// Set by the FIRST onError, consumed by onFinish (which the SDK still runs
|
|
1062
|
+
// after a stream error, and may precede with more than one onError). It is
|
|
1063
|
+
// what keeps the turn's terminal state single and ordered: the run row is
|
|
1064
|
+
// settled as errored FIRST, and exactly one errored report follows — only
|
|
1065
|
+
// from the writer whose fenced update applied.
|
|
1066
|
+
let failed;
|
|
1067
|
+
// Step checkpoints commit IN STEP ORDER. Each write is still fire-and-forget
|
|
1068
|
+
// for the stream, but chained behind the previous one, so a slow early
|
|
1069
|
+
// write can't commit after a later one and roll the row back to an older
|
|
1070
|
+
// step — and `onFinish` awaits the chain before its final write, so the
|
|
1071
|
+
// finished message is strictly the LAST write of the turn. Without that the
|
|
1072
|
+
// final step's checkpoint races `onFinish` for the same row: the parts are
|
|
1073
|
+
// equal by then, but a late checkpoint clobbers the usage/cost metadata
|
|
1074
|
+
// only `onFinish` stamps. The chain orders, it does not batch: each write
|
|
1075
|
+
// starts the moment the one before it commits, so a crash at step N loses
|
|
1076
|
+
// at most the write in flight and the one queued behind it. A failed write
|
|
1077
|
+
// logs and releases the chain — the next checkpoint still runs.
|
|
1078
|
+
let checkpoints = Promise.resolve();
|
|
1079
|
+
// Post-turn telemetry: fired after persistence on every terminal path
|
|
1080
|
+
// (completed / awaiting_input / errored). Must never fail the turn.
|
|
1081
|
+
const reportRunFinished = async (status, messages, error) => {
|
|
1082
|
+
const t = config.telemetry;
|
|
1083
|
+
if (!t || !telemetryOn)
|
|
1084
|
+
return;
|
|
1085
|
+
try {
|
|
1086
|
+
await t.onRunFinished?.(telemetryMeta, {
|
|
1087
|
+
status,
|
|
1088
|
+
...(error !== undefined ? { error } : {}),
|
|
1089
|
+
inputText: lastUserText(uiMessages),
|
|
1090
|
+
messages,
|
|
1091
|
+
usage: {
|
|
1092
|
+
inputTokens,
|
|
1093
|
+
outputTokens,
|
|
1094
|
+
...(cacheReadTokens ? { cacheReadTokens } : {}),
|
|
1095
|
+
...(cacheWriteTokens ? { cacheWriteTokens } : {}),
|
|
1096
|
+
},
|
|
1097
|
+
cost: computeCost(config.pricing?.(modelId), {
|
|
1098
|
+
inputTokens,
|
|
1099
|
+
outputTokens,
|
|
1100
|
+
cacheReadTokens,
|
|
1101
|
+
cacheWriteTokens,
|
|
1102
|
+
}),
|
|
1103
|
+
endedAt: Date.now(),
|
|
1104
|
+
});
|
|
1105
|
+
}
|
|
1106
|
+
catch (e) {
|
|
1107
|
+
console.warn("latch telemetry onRunFinished failed:", e);
|
|
1108
|
+
}
|
|
1109
|
+
finally {
|
|
1110
|
+
// Always flush, even when the finish hook threw — otherwise a failed
|
|
1111
|
+
// judge/score pass would drop the turn's spans in serverless.
|
|
1112
|
+
try {
|
|
1113
|
+
await t.flush?.();
|
|
1114
|
+
}
|
|
1115
|
+
catch (e) {
|
|
1116
|
+
console.warn("latch telemetry flush failed:", e);
|
|
1117
|
+
}
|
|
1118
|
+
}
|
|
1119
|
+
};
|
|
1120
|
+
// Lower-level composition (vs `createAgentUIStreamResponse`) so we get the
|
|
1121
|
+
// UI-stream `onStepEnd` (per-step persistence) and the `writer`. The model
|
|
1122
|
+
// call runs INSIDE `execute` so the writer is live during the turn — code
|
|
1123
|
+
// here can write data parts (e.g. a compaction marker) before the model's
|
|
1124
|
+
// output, and tools can write via `toolsContext`.
|
|
1125
|
+
const stream = createUIMessageStream({
|
|
1126
|
+
originalMessages: uiMessages,
|
|
1127
|
+
// Persistence mode: one stable assistant-message id per turn (else every
|
|
1128
|
+
// turn would upsert onto the same row — see the dup-id regression test).
|
|
1129
|
+
generateId: generateMessageId,
|
|
1130
|
+
execute: async ({ writer }) => {
|
|
1131
|
+
// (in-turn writer writes — e.g. a "compacting…" / compaction marker —
|
|
1132
|
+
// would go here, before the model output is merged in)
|
|
1133
|
+
// Model projection: drops `sendToModel:false` messages (and data-* parts).
|
|
1134
|
+
const modelMessages = await toModelMessages(uiMessages, { tools });
|
|
1135
|
+
// Per-session sandbox (one per chat) for agents that declared `sandbox`.
|
|
1136
|
+
// Exposed to sandbox tools as `options.experimental_sandbox`.
|
|
1137
|
+
const sandbox = wantSandbox
|
|
1138
|
+
? await config.sandbox({
|
|
1139
|
+
principal,
|
|
1140
|
+
chatId: resolvedChatId,
|
|
1141
|
+
agent: run.agent,
|
|
1142
|
+
provider: cfg.sandboxProvider,
|
|
1143
|
+
})
|
|
1144
|
+
: undefined;
|
|
1145
|
+
// Start the model call inside the telemetry run span (when provided)
|
|
1146
|
+
// so the impl can root every AI SDK span in one seeded trace.
|
|
1147
|
+
const startStream = () => agent.stream({
|
|
1148
|
+
prompt: modelMessages,
|
|
1149
|
+
experimental_sandbox: sandbox,
|
|
1150
|
+
abortSignal: dur.enabled ? abort.signal : undefined,
|
|
1151
|
+
// Optional: smooth the visible token cadence before deltas stream out.
|
|
1152
|
+
experimental_transform: cfg.smoothStream
|
|
1153
|
+
? smoothStream(cfg.smoothStream === true ? undefined : cfg.smoothStream)
|
|
1154
|
+
: undefined,
|
|
1155
|
+
// Shared context for hooks / prepareStep / approval policies.
|
|
1156
|
+
runtimeContext: runtimeCtx,
|
|
1157
|
+
// `toolsContext` is keyed BY TOOL NAME (the SDK does
|
|
1158
|
+
// `toolsContext[toolName]`), so point every tool at the same context:
|
|
1159
|
+
// the project's runtime context + who it's for + the run + the writer
|
|
1160
|
+
// + the runtime itself (so a harness tool can schedule, list chats or
|
|
1161
|
+
// spawn a run without reaching for a host-global). A tool reads it as
|
|
1162
|
+
// `options.context`.
|
|
1163
|
+
toolsContext: Object.fromEntries(Object.keys(tools).map((name) => [
|
|
1164
|
+
name,
|
|
1165
|
+
{
|
|
1166
|
+
...runtimeCtx,
|
|
1167
|
+
principal,
|
|
1168
|
+
runtime: self,
|
|
1169
|
+
// `agent` is the ref the turn resolved (scope-pinned as stored
|
|
1170
|
+
// on the run), so a tool that acts on "the agent I'm running
|
|
1171
|
+
// as" names the same record the turn is using.
|
|
1172
|
+
run: { id: run.id, chatId: resolvedChatId, agent: run.agent },
|
|
1173
|
+
writer,
|
|
1174
|
+
},
|
|
1175
|
+
])),
|
|
1176
|
+
// generate-text step callback: accumulate usage across steps,
|
|
1177
|
+
// including the cached-prompt breakdown so cost reflects cache reads.
|
|
1178
|
+
onStepEnd: (step) => {
|
|
1179
|
+
inputTokens += step.usage?.inputTokens ?? 0;
|
|
1180
|
+
outputTokens += step.usage?.outputTokens ?? 0;
|
|
1181
|
+
cacheReadTokens += step.usage?.inputTokenDetails?.cacheReadTokens ?? 0;
|
|
1182
|
+
cacheWriteTokens += step.usage?.inputTokenDetails?.cacheWriteTokens ?? 0;
|
|
1183
|
+
if (telemetryOn && config.telemetry?.onStep) {
|
|
1184
|
+
const onStep = config.telemetry.onStep.bind(config.telemetry);
|
|
1185
|
+
fireAndForget(() => onStep(telemetryMeta, {
|
|
1186
|
+
text: step.text,
|
|
1187
|
+
finishReason: step.finishReason,
|
|
1188
|
+
usage: {
|
|
1189
|
+
inputTokens: step.usage?.inputTokens,
|
|
1190
|
+
outputTokens: step.usage?.outputTokens,
|
|
1191
|
+
cacheReadTokens: step.usage?.inputTokenDetails?.cacheReadTokens,
|
|
1192
|
+
cacheWriteTokens: step.usage?.inputTokenDetails?.cacheWriteTokens,
|
|
1193
|
+
},
|
|
1194
|
+
toolCalls: step.toolCalls,
|
|
1195
|
+
toolResults: step.toolResults,
|
|
1196
|
+
}));
|
|
1197
|
+
}
|
|
1198
|
+
},
|
|
1199
|
+
});
|
|
1200
|
+
const result = telemetryOn && config.telemetry?.withRunSpan
|
|
1201
|
+
? await config.telemetry.withRunSpan(telemetryMeta, startStream)
|
|
1202
|
+
: await startStream();
|
|
1203
|
+
// Fold the model's own output into the stream.
|
|
1204
|
+
writer.merge(toUIMessageStream({
|
|
1205
|
+
stream: result.stream,
|
|
1206
|
+
// Without this the AI SDK collapses every tool `execute` throw to
|
|
1207
|
+
// the literal "An error occurred." — surface the real message so
|
|
1208
|
+
// failures (sandbox init, credential lookups, …) are debuggable
|
|
1209
|
+
// from the chat itself.
|
|
1210
|
+
onError: clientErrorMessage,
|
|
1211
|
+
messageMetadata: ({ part }) => part.type === "start"
|
|
1212
|
+
? {
|
|
1213
|
+
visibility: "user",
|
|
1214
|
+
model: modelId,
|
|
1215
|
+
createdAt: Date.now(),
|
|
1216
|
+
// Correlates the assistant message to its run (and trace)
|
|
1217
|
+
// — e.g. per-message feedback posts against this id.
|
|
1218
|
+
runId: run.id,
|
|
1219
|
+
}
|
|
1220
|
+
: undefined,
|
|
1221
|
+
}));
|
|
1222
|
+
},
|
|
1223
|
+
// UI-stream per-step hook: checkpoint the in-progress assistant message
|
|
1224
|
+
// (with any completed tool results) so a crash mid-turn doesn't repeat
|
|
1225
|
+
// them on resume. Durable turns only — keeps T1 write counts unchanged.
|
|
1226
|
+
//
|
|
1227
|
+
// Guarded on the turn's abort signal: `appendMessages` has no lease
|
|
1228
|
+
// fence (only `updateRun` does), so once this worker has lost the lease
|
|
1229
|
+
// a reclaimed copy of the run may be replaying into this same chat —
|
|
1230
|
+
// an un-guarded checkpoint from the zombie would interleave with it,
|
|
1231
|
+
// and every later resume replays the polluted history.
|
|
1232
|
+
//
|
|
1233
|
+
// Best-effort, not a fence: the lease can be lost between this check and
|
|
1234
|
+
// the write committing. Closing that window needs a lease/fencing-token
|
|
1235
|
+
// check inside the storage write itself — an adapter interface change,
|
|
1236
|
+
// tracked as a follow-up. With the heartbeat now aborting within one TTL
|
|
1237
|
+
// of losing contact, the exposure is bounded to ~leaseTtlMs.
|
|
1238
|
+
onStepEnd: dur.enabled
|
|
1239
|
+
? ({ responseMessage }) => {
|
|
1240
|
+
if (abort.signal.aborted)
|
|
1241
|
+
return;
|
|
1242
|
+
checkpoints = checkpoints
|
|
1243
|
+
.then(() => {
|
|
1244
|
+
// Re-checked at write time: the lease may have gone while this
|
|
1245
|
+
// checkpoint waited behind the previous one.
|
|
1246
|
+
if (abort.signal.aborted)
|
|
1247
|
+
return;
|
|
1248
|
+
return storage.appendMessages(resolvedChatId, [responseMessage]);
|
|
1249
|
+
})
|
|
1250
|
+
.catch((e) => console.warn(`[durability] step checkpoint failed for run ${run.id}:`, e));
|
|
1251
|
+
}
|
|
1252
|
+
: undefined,
|
|
1253
|
+
onFinish: async ({ messages, isAborted, }) => {
|
|
1254
|
+
stopHeartbeat();
|
|
1255
|
+
await closeSources();
|
|
1256
|
+
// Cost for this turn (0 if no pricing configured). Stamp usage + cost
|
|
1257
|
+
// onto the assistant message — the message is complete; this is metadata.
|
|
1258
|
+
const cost = computeCost(config.pricing?.(modelId), {
|
|
1259
|
+
inputTokens,
|
|
1260
|
+
outputTokens,
|
|
1261
|
+
cacheReadTokens,
|
|
1262
|
+
cacheWriteTokens,
|
|
1263
|
+
});
|
|
1264
|
+
const last = messages[messages.length - 1];
|
|
1265
|
+
if (last?.role === "assistant") {
|
|
1266
|
+
last.metadata = {
|
|
1267
|
+
...last.metadata,
|
|
1268
|
+
usage: {
|
|
1269
|
+
inputTokens,
|
|
1270
|
+
outputTokens,
|
|
1271
|
+
...(cacheReadTokens ? { cacheReadTokens } : {}),
|
|
1272
|
+
...(cacheWriteTokens ? { cacheWriteTokens } : {}),
|
|
1273
|
+
},
|
|
1274
|
+
cost,
|
|
1275
|
+
};
|
|
1276
|
+
}
|
|
1277
|
+
// Errored (the stream failed — onError already reported it) or aborted
|
|
1278
|
+
// (lease lost): don't claim completion, and DON'T persist the partial
|
|
1279
|
+
// messages/usage — an aborted run may already be reaped and re-claimed
|
|
1280
|
+
// by another node whose writes must win. The guarded update no-ops if
|
|
1281
|
+
// the run was reassigned. Exactly one terminal report either way.
|
|
1282
|
+
if (failed !== undefined || isAborted) {
|
|
1283
|
+
const settled = await storage.updateRun(run.id, { status: "errored", error: failed ?? "aborted", endedAt: Date.now() }, lease);
|
|
1284
|
+
// Only the winning writer reports (same rule as the completion
|
|
1285
|
+
// path): a stale worker whose fenced update did not apply is not
|
|
1286
|
+
// the one that ended this run. The tokens WERE spent, though — a
|
|
1287
|
+
// failing turn settles against its budgets like any other, or a
|
|
1288
|
+
// token-only cap would never bound a run that keeps erroring.
|
|
1289
|
+
if (settled) {
|
|
1290
|
+
await settleLimits(admitted, { tokens: inputTokens + outputTokens, cost });
|
|
1291
|
+
await reportRunFinished("errored", messages, failed ?? "aborted");
|
|
1292
|
+
}
|
|
1293
|
+
return;
|
|
1294
|
+
}
|
|
1295
|
+
// A gated tool ended the turn awaiting a decision — pause, don't
|
|
1296
|
+
// complete. The decision arrives via applyApproval (message state).
|
|
1297
|
+
const paused = hasPendingApproval(messages);
|
|
1298
|
+
// Every step checkpoint has committed (or failed and been logged)
|
|
1299
|
+
// BEFORE the row is settled. Recovery never revisits a completed run,
|
|
1300
|
+
// so the moment the row says `completed` the stored message must
|
|
1301
|
+
// already hold the final step — otherwise a crash between the settle
|
|
1302
|
+
// and the final write below loses the answer text, not just metadata.
|
|
1303
|
+
// Awaiting here also makes the final write the row's last one.
|
|
1304
|
+
await checkpoints;
|
|
1305
|
+
// Fenced finish: settle the run FIRST. If the guarded update doesn't
|
|
1306
|
+
// apply (our lease was reaped + the run re-claimed elsewhere), we're
|
|
1307
|
+
// a stale writer — skip the final message/usage persistence so we
|
|
1308
|
+
// don't overwrite the new owner's state.
|
|
1309
|
+
const applied = await storage.updateRun(run.id, {
|
|
1310
|
+
status: paused ? "awaiting_input" : "completed",
|
|
1311
|
+
usageInputTokens: inputTokens,
|
|
1312
|
+
usageOutputTokens: outputTokens,
|
|
1313
|
+
usageCacheReadTokens: cacheReadTokens || undefined,
|
|
1314
|
+
usageCacheWriteTokens: cacheWriteTokens || undefined,
|
|
1315
|
+
costTotal: cost,
|
|
1316
|
+
toolsHash,
|
|
1317
|
+
endedAt: paused ? undefined : Date.now(),
|
|
1318
|
+
}, lease);
|
|
1319
|
+
if (applied) {
|
|
1320
|
+
await storage.appendMessages(resolvedChatId, messages);
|
|
1321
|
+
await storage.recordUsage(principal, {
|
|
1322
|
+
runId: run.id,
|
|
1323
|
+
model: modelId,
|
|
1324
|
+
inputTokens,
|
|
1325
|
+
outputTokens,
|
|
1326
|
+
...(cacheReadTokens ? { cacheReadTokens } : {}),
|
|
1327
|
+
...(cacheWriteTokens ? { cacheWriteTokens } : {}),
|
|
1328
|
+
cost,
|
|
1329
|
+
});
|
|
1330
|
+
await settleLimits(admitted, { tokens: inputTokens + outputTokens, cost });
|
|
1331
|
+
}
|
|
1332
|
+
// Only the winning writer reports terminal telemetry. If our lease was
|
|
1333
|
+
// reaped and the run re-claimed elsewhere (`!applied`), that node will
|
|
1334
|
+
// finish it — a stale writer must not re-run the judge or emit a
|
|
1335
|
+
// duplicate "completed"/"awaiting_input" trace.
|
|
1336
|
+
if (applied)
|
|
1337
|
+
await reportRunFinished(paused ? "awaiting_input" : "completed", messages);
|
|
1338
|
+
},
|
|
1339
|
+
onError: (error) => {
|
|
1340
|
+
stopHeartbeat();
|
|
1341
|
+
void closeSources();
|
|
1342
|
+
const message = clientErrorMessage(error);
|
|
1343
|
+
// Record only — the first error wins; later onError calls for the same
|
|
1344
|
+
// stream are noise. The SDK's onError must return the client string
|
|
1345
|
+
// synchronously and still runs onFinish afterwards, so that is where
|
|
1346
|
+
// the run row is settled as errored and the single terminal report is
|
|
1347
|
+
// made (after, and only if, the fenced update applied).
|
|
1348
|
+
if (failed === undefined)
|
|
1349
|
+
failed = message;
|
|
1350
|
+
return message;
|
|
1351
|
+
},
|
|
1352
|
+
});
|
|
1353
|
+
// Drain the stream server-side regardless of the client. If the browser
|
|
1354
|
+
// closes mid-turn, the response stops being pulled — without this the
|
|
1355
|
+
// generation stalls/cuts short and a refresh shows a partial answer. With
|
|
1356
|
+
// it, the agent runs to completion in the background and onFinish persists
|
|
1357
|
+
// the full message (the model's abort is tied to the lease, not the
|
|
1358
|
+
// request, so a disconnect doesn't cancel generation).
|
|
1359
|
+
let out = stream;
|
|
1360
|
+
if (args.observeStream) {
|
|
1361
|
+
const [main, observed] = out.tee();
|
|
1362
|
+
out = main;
|
|
1363
|
+
args.observeStream(observed);
|
|
1364
|
+
}
|
|
1365
|
+
return createUIMessageStreamResponse({
|
|
1366
|
+
stream: out,
|
|
1367
|
+
consumeSseStream: consumeStream,
|
|
1368
|
+
});
|
|
1369
|
+
}
|
|
1370
|
+
async function reapTick(opts) {
|
|
1371
|
+
const reaped = await storage.reapExpiredRuns(opts?.now ?? Date.now(), {
|
|
1372
|
+
error: opts?.error,
|
|
1373
|
+
});
|
|
1374
|
+
return { reaped };
|
|
1375
|
+
}
|
|
1376
|
+
/**
|
|
1377
|
+
* The admission gate. A FRESH turn (chat, schedule, subagent spawn, runAgent)
|
|
1378
|
+
* is RESERVED: one atomic increment of every policy's turn counter, which
|
|
1379
|
+
* returns the new counts; the first policy now over its cap refuses the turn
|
|
1380
|
+
* (`LimitExceededError`) and the reservation is released. Increment-then-
|
|
1381
|
+
* check is what makes a one-turn cap admit exactly one of N concurrent
|
|
1382
|
+
* requests — a read-then-write check would let them all through.
|
|
1383
|
+
*
|
|
1384
|
+
* A CONTINUATION is the second half of an already-admitted turn: a crash
|
|
1385
|
+
* resume, an approval or tool-result decision, a subagent's decided gate. It
|
|
1386
|
+
* passes `{ countTurn: false, refuse: false }`: never counted twice, never
|
|
1387
|
+
* refused (the persisted decision would strand until the window turned
|
|
1388
|
+
* over), but its tokens still settle and the mid-turn stop still applies —
|
|
1389
|
+
* with an exhausted budget it gets one step and stops. Undefined when no
|
|
1390
|
+
* limits are configured or none apply.
|
|
1391
|
+
*/
|
|
1392
|
+
async function admitTurn(principal, agent, trigger, opts = { countTurn: true, refuse: true }) {
|
|
1393
|
+
const limits = config.limits;
|
|
1394
|
+
if (!limits)
|
|
1395
|
+
return undefined;
|
|
1396
|
+
const policies = await limits.limitsFor({ principal, agent, trigger });
|
|
1397
|
+
if (policies.length === 0)
|
|
1398
|
+
return undefined;
|
|
1399
|
+
const now = Date.now();
|
|
1400
|
+
const refs = policies.map((p) => windowRef(p, now));
|
|
1401
|
+
// Reserve (fresh turn) or just look (continuation). Either way `usage` is
|
|
1402
|
+
// the state this turn is judged against, reservation included.
|
|
1403
|
+
const usage = opts.countTurn
|
|
1404
|
+
? await limits.store.add(refs.map((ref) => ({ ...ref, usage: { turns: 1 } })))
|
|
1405
|
+
: await limits.store.read(refs);
|
|
1406
|
+
if (opts.refuse) {
|
|
1407
|
+
const over = exceededPolicy(policies, usage, now, { nextTurn: false });
|
|
1408
|
+
if (over) {
|
|
1409
|
+
// Release the reservation: the window should count admitted turns, not
|
|
1410
|
+
// refused attempts. Best-effort — a failed release only over-counts.
|
|
1411
|
+
if (opts.countTurn) {
|
|
1412
|
+
await limits.store
|
|
1413
|
+
.add(refs.map((ref) => ({ ...ref, usage: { turns: -1 } })))
|
|
1414
|
+
.catch((e) => console.error("[limits] reservation release failed:", e));
|
|
1415
|
+
}
|
|
1416
|
+
// A host hook must never turn a refusal into a 500: isolate sync throws too.
|
|
1417
|
+
try {
|
|
1418
|
+
await limits.onRefused?.({ principal, agent, trigger, ...over });
|
|
1419
|
+
}
|
|
1420
|
+
catch (e) {
|
|
1421
|
+
console.error("[limits] onRefused failed:", e);
|
|
1422
|
+
}
|
|
1423
|
+
throw new LimitExceededError(over.policy, over.usage, over.retryAfterMs);
|
|
1424
|
+
}
|
|
1425
|
+
}
|
|
1426
|
+
return { policies, refs, remainingTokens: remainingTokens(policies, usage, now), counted: opts.countTurn };
|
|
1427
|
+
}
|
|
1428
|
+
/**
|
|
1429
|
+
* Give a reserved turn back when the turn never opened (busy chat, agent
|
|
1430
|
+
* resolution or run creation failed): the window counts turns that RAN, or a
|
|
1431
|
+
* user retrying into a busy chat would exhaust their own turn cap.
|
|
1432
|
+
*/
|
|
1433
|
+
async function releaseReservation(admitted) {
|
|
1434
|
+
if (!admitted?.counted || !config.limits)
|
|
1435
|
+
return;
|
|
1436
|
+
await config.limits.store
|
|
1437
|
+
.add(admitted.refs.map((ref) => ({ ...ref, usage: { turns: -1 } })))
|
|
1438
|
+
.catch((e) => console.error("[limits] reservation release failed:", e));
|
|
1439
|
+
}
|
|
1440
|
+
/** Run `open` for a freshly admitted turn; on failure the reservation is released and the error rethrown. */
|
|
1441
|
+
async function withReservation(admitted, open) {
|
|
1442
|
+
try {
|
|
1443
|
+
return await open();
|
|
1444
|
+
}
|
|
1445
|
+
catch (e) {
|
|
1446
|
+
await releaseReservation(admitted);
|
|
1447
|
+
throw e;
|
|
1448
|
+
}
|
|
1449
|
+
}
|
|
1450
|
+
/** `admitTurn` for continuations (see above). */
|
|
1451
|
+
const admitContinuation = (principal, agent, trigger) => admitTurn(principal, agent, trigger, { countTurn: false, refuse: false });
|
|
1452
|
+
/** Settlement: add this turn's tokens and cost to every policy it was admitted under. */
|
|
1453
|
+
async function settleLimits(admitted, usage) {
|
|
1454
|
+
if (!admitted || !config.limits)
|
|
1455
|
+
return;
|
|
1456
|
+
const now = Date.now();
|
|
1457
|
+
await config.limits.store
|
|
1458
|
+
.add(admitted.policies.map((p) => ({ ...windowRef(p, now), usage })))
|
|
1459
|
+
.catch((e) => console.error("[limits] settlement failed:", e));
|
|
1460
|
+
}
|
|
1461
|
+
/**
|
|
1462
|
+
* Open a turn: the run (one active per chat — `ChatBusyError` if not ours to
|
|
1463
|
+
* run) and the user message that starts it. Atomic when the adapter offers
|
|
1464
|
+
* `openTurn`; otherwise the run first, then the message, so a refusal still
|
|
1465
|
+
* persists nothing (the crash window between the two writes is the trade-off
|
|
1466
|
+
* an adapter without `openTurn` accepts).
|
|
1467
|
+
*/
|
|
1468
|
+
async function openTurn(principal, chatId, agentName, messages) {
|
|
1469
|
+
if (storage.openTurn) {
|
|
1470
|
+
const run = await storage.openTurn(principal, {
|
|
1471
|
+
chatId,
|
|
1472
|
+
agent: agentName,
|
|
1473
|
+
messages,
|
|
1474
|
+
...(dur.enabled ? { lease: { owner: dur.instanceId, ttlMs: dur.leaseTtlMs } } : {}),
|
|
1475
|
+
});
|
|
1476
|
+
return {
|
|
1477
|
+
run,
|
|
1478
|
+
lease: dur.enabled ? { owner: dur.instanceId, fencingToken: run.fencingToken ?? 1 } : undefined,
|
|
1479
|
+
};
|
|
1480
|
+
}
|
|
1481
|
+
const opened = await openRun(principal, chatId, agentName);
|
|
1482
|
+
if (messages.length > 0)
|
|
1483
|
+
await storage.appendMessages(chatId, messages);
|
|
1484
|
+
return opened;
|
|
1485
|
+
}
|
|
1486
|
+
async function openRun(principal, chatId, agentName) {
|
|
1487
|
+
if (dur.enabled) {
|
|
1488
|
+
const run = await storage.claimRun({
|
|
1489
|
+
principal,
|
|
1490
|
+
chatId,
|
|
1491
|
+
agent: agentName,
|
|
1492
|
+
owner: dur.instanceId,
|
|
1493
|
+
ttlMs: dur.leaseTtlMs,
|
|
1494
|
+
});
|
|
1495
|
+
return { run, lease: { owner: dur.instanceId, fencingToken: run.fencingToken ?? 1 } };
|
|
1496
|
+
}
|
|
1497
|
+
return {
|
|
1498
|
+
run: await storage.createRun(principal, { chatId, agent: agentName, kind: "turn" }),
|
|
1499
|
+
lease: undefined,
|
|
1500
|
+
};
|
|
1501
|
+
}
|
|
1502
|
+
async function handleChat({ agent: agentName, chatId, principal, message, request, turnContext, }) {
|
|
1503
|
+
// chat: scoped create-or-continue (others' ids resolve to "not found")
|
|
1504
|
+
let chat = chatId ? await storage.getChat(principal, chatId) : null;
|
|
1505
|
+
if (!chat) {
|
|
1506
|
+
chat = await storage.createChat(principal, {
|
|
1507
|
+
id: chatId, // use the provided id (create-or-continue); generated if omitted
|
|
1508
|
+
agent: agentName,
|
|
1509
|
+
// Title from the first user message — for the chat list.
|
|
1510
|
+
title: firstText(message)?.slice(0, 80),
|
|
1511
|
+
});
|
|
1512
|
+
}
|
|
1513
|
+
const resolvedChatId = chat.id;
|
|
1514
|
+
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(agentName, principal, request, turnContext);
|
|
1515
|
+
const prior = await storage.loadMessages(principal, resolvedChatId);
|
|
1516
|
+
// A new message while the last turn is parked on an approval is itself the
|
|
1517
|
+
// decision: the user declined to answer and wants to steer elsewhere. Close
|
|
1518
|
+
// the gate as declined (persisted, so the card stops rendering as live) and
|
|
1519
|
+
// carry on — the gated tool never runs, and the model sees the denial
|
|
1520
|
+
// followed by the new instruction. Refusing the message instead is what
|
|
1521
|
+
// wedged chats whose gate could not be satisfied at all.
|
|
1522
|
+
if (hasPendingApproval(prior)) {
|
|
1523
|
+
const declined = declinePendingApprovals(prior);
|
|
1524
|
+
if (declined.length > 0)
|
|
1525
|
+
await storage.appendMessages(resolvedChatId, declined);
|
|
1526
|
+
}
|
|
1527
|
+
// Limits gate BEFORE anything is persisted: a refused turn leaves no
|
|
1528
|
+
// trace (the client gets 429 and can retry when the window turns over).
|
|
1529
|
+
const admitted = await admitTurn(principal, agentName, "chat");
|
|
1530
|
+
// Run + message open together (atomically where the adapter can): a chat
|
|
1531
|
+
// with a live turn refuses (`ChatBusyError` → 409) and the message never
|
|
1532
|
+
// lands in history — the client retries it (with its turn reservation released).
|
|
1533
|
+
const { run, lease } = await withReservation(admitted, () => openTurn(principal, resolvedChatId, agentName, [message]));
|
|
1534
|
+
return buildTurn({
|
|
1535
|
+
resolvedChatId,
|
|
1536
|
+
principal,
|
|
1537
|
+
modelId,
|
|
1538
|
+
cfg,
|
|
1539
|
+
run,
|
|
1540
|
+
lease,
|
|
1541
|
+
uiMessages: [...prior, message],
|
|
1542
|
+
runtimeCtx,
|
|
1543
|
+
admitted,
|
|
1544
|
+
});
|
|
1545
|
+
}
|
|
1546
|
+
async function resumeRun(runId) {
|
|
1547
|
+
if (!dur.enabled)
|
|
1548
|
+
throw new Error("resume requires durability.enabled");
|
|
1549
|
+
// Exactly-once re-claim: only one node wins; a concurrent resumer / an
|
|
1550
|
+
// already-completed run yields null.
|
|
1551
|
+
const run = await storage.reclaimRun(runId, dur.instanceId, dur.leaseTtlMs);
|
|
1552
|
+
if (!run)
|
|
1553
|
+
return null;
|
|
1554
|
+
// System op (no request): rebuild the principal from the run's identity blob.
|
|
1555
|
+
const principal = await reconstructPrincipal(run.identity);
|
|
1556
|
+
const uiMessages = await storage.loadMessages(principal, run.chatId);
|
|
1557
|
+
const lease = {
|
|
1558
|
+
owner: dur.instanceId,
|
|
1559
|
+
fencingToken: run.fencingToken ?? 1,
|
|
1560
|
+
};
|
|
1561
|
+
// What does this run actually need? A crash between the last step's
|
|
1562
|
+
// checkpoint and `onFinish`'s `updateRun` leaves a FINISHED turn on an
|
|
1563
|
+
// `active` row, and re-running the model there would pay twice for an answer
|
|
1564
|
+
// already in history — and append the second copy to the same message. So
|
|
1565
|
+
// reconcile the row instead, and only re-run when work is genuinely left.
|
|
1566
|
+
// Did the chat move on while this run sat reaped? Asked of the runs table,
|
|
1567
|
+
// never inferred from message shapes — see `supersededBy` for why. The list
|
|
1568
|
+
// is owner-scoped and most-recent first; a later sibling is more recent than
|
|
1569
|
+
// we are, so it is present unless more than `limit` runs happened owner-wide
|
|
1570
|
+
// after IT. A miss errs toward re-running (today's behaviour), never toward
|
|
1571
|
+
// settling a live run.
|
|
1572
|
+
const siblings = await storage.listRuns(principal, { limit: 500 });
|
|
1573
|
+
const shape = supersededBy(run, siblings) ? "superseded" : resumeShapeOf(uiMessages);
|
|
1574
|
+
if (shape !== "continue") {
|
|
1575
|
+
const status = shape === "superseded" ? "errored" : shape === "decision-pending" ? "awaiting_input" : "completed";
|
|
1576
|
+
// Same outcome `openTurn` records when a fresh message supersedes a stale
|
|
1577
|
+
// run, so a superseded run reads the same however it got there.
|
|
1578
|
+
const error = shape === "superseded" ? "superseded: a later turn moved the conversation on" : null;
|
|
1579
|
+
// Fenced like every other terminal write: if our lease was reaped again
|
|
1580
|
+
// mid-flight, the new owner decides this run's end state, not us.
|
|
1581
|
+
const applied = await storage.updateRun(run.id, {
|
|
1582
|
+
status,
|
|
1583
|
+
// The reaper stamped BOTH an error and an `endedAt` on the way in
|
|
1584
|
+
// (see `reapExpiredRuns`). A finished turn did not fail, so its error
|
|
1585
|
+
// has to go or the row reads as failed forever — and a run settling to
|
|
1586
|
+
// `awaiting_input` is paused, not over, so its `endedAt` has to go
|
|
1587
|
+
// too. `onFinish` leaves it unset for a paused turn; a crash-settled
|
|
1588
|
+
// one must match, or inter-run gap stats read a run that ended before
|
|
1589
|
+
// its decision arrived. `null` (not `undefined`) is what clears a
|
|
1590
|
+
// column: the adapters drop undefined from the patch.
|
|
1591
|
+
error,
|
|
1592
|
+
endedAt: status === "awaiting_input" ? null : Date.now(),
|
|
1593
|
+
}, lease);
|
|
1594
|
+
// The process that spent this turn's tokens died before recording them,
|
|
1595
|
+
// so usage is unknowable here — the row and this report agree on zero
|
|
1596
|
+
// rather than inventing a number. The notification still fires: a host
|
|
1597
|
+
// watching `onRunFinished` for "the turn ended" must not miss a turn just
|
|
1598
|
+
// because recovery was a row write instead of a model call.
|
|
1599
|
+
const meta = {
|
|
1600
|
+
runId: run.id,
|
|
1601
|
+
chatId: run.chatId,
|
|
1602
|
+
agent: run.agent,
|
|
1603
|
+
// The model that ACTUALLY ran, off the message it produced — not a
|
|
1604
|
+
// freshly resolved config. Reconciling a finished turn must not need
|
|
1605
|
+
// the agent to still exist (see above), and this is the truer answer
|
|
1606
|
+
// anyway: the agent's model may have been changed since.
|
|
1607
|
+
modelId: ranAs(uiMessages, run.id) ?? "unknown",
|
|
1608
|
+
principal,
|
|
1609
|
+
startedAt: run.startedAt,
|
|
1610
|
+
depth: 0,
|
|
1611
|
+
trigger: "resume",
|
|
1612
|
+
rootChatId: run.chatId,
|
|
1613
|
+
};
|
|
1614
|
+
// Same opt-out rule as a streamed turn: a `telemetryFor` that returns
|
|
1615
|
+
// undefined silences this run's hooks too (see buildTurn).
|
|
1616
|
+
const telemetryOn = !!config.telemetry &&
|
|
1617
|
+
!(config.telemetry.telemetryFor && config.telemetry.telemetryFor(meta) === undefined);
|
|
1618
|
+
if (applied && telemetryOn) {
|
|
1619
|
+
try {
|
|
1620
|
+
await config.telemetry?.onRunFinished?.(meta, {
|
|
1621
|
+
status,
|
|
1622
|
+
...(error !== null ? { error } : {}),
|
|
1623
|
+
inputText: lastUserText(uiMessages),
|
|
1624
|
+
messages: uiMessages,
|
|
1625
|
+
usage: { inputTokens: 0, outputTokens: 0 },
|
|
1626
|
+
cost: 0,
|
|
1627
|
+
endedAt: Date.now(),
|
|
1628
|
+
});
|
|
1629
|
+
}
|
|
1630
|
+
catch (e) {
|
|
1631
|
+
console.warn("latch telemetry onRunFinished failed:", e);
|
|
1632
|
+
}
|
|
1633
|
+
try {
|
|
1634
|
+
await config.telemetry?.flush?.();
|
|
1635
|
+
}
|
|
1636
|
+
catch (e) {
|
|
1637
|
+
console.warn("latch telemetry flush failed:", e);
|
|
1638
|
+
}
|
|
1639
|
+
}
|
|
1640
|
+
const settled = await storage.getRun(runId);
|
|
1641
|
+
return {
|
|
1642
|
+
runId,
|
|
1643
|
+
status: settled?.status ?? status,
|
|
1644
|
+
attempt: settled?.attempt ?? run.attempt ?? null,
|
|
1645
|
+
};
|
|
1646
|
+
}
|
|
1647
|
+
// Only now resolve the agent — `resolveAgentConfig` throws on an agent that
|
|
1648
|
+
// no longer resolves, and a run whose answer is already stored must be
|
|
1649
|
+
// reconcilable even if its (dynamic, DB-stored) agent was since deleted.
|
|
1650
|
+
// Before this ordering that throw also aborted `sweep`'s loop, so one such
|
|
1651
|
+
// run held up recovery for every other run until its attempts ran out.
|
|
1652
|
+
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(run.agent, principal);
|
|
1653
|
+
// A resume was admitted when the turn first ran — never refuse it (the
|
|
1654
|
+
// half-done work would strand), but its tokens still settle against the
|
|
1655
|
+
// same counters, and the mid-turn stop still applies.
|
|
1656
|
+
const admitted = await admitContinuation(principal, run.agent, "resume");
|
|
1657
|
+
// Re-run from persisted history and drive to completion server-side
|
|
1658
|
+
// (there's no client to consume the stream).
|
|
1659
|
+
const res = await buildTurn({
|
|
1660
|
+
resolvedChatId: run.chatId,
|
|
1661
|
+
principal,
|
|
1662
|
+
modelId,
|
|
1663
|
+
cfg,
|
|
1664
|
+
run,
|
|
1665
|
+
lease,
|
|
1666
|
+
// Same prefill guard the decision paths use: a tail ending in the turn's
|
|
1667
|
+
// own half-written sentence would reach the model as an assistant
|
|
1668
|
+
// prefill, which models that reject one refuse on the request SHAPE — so
|
|
1669
|
+
// every identical retry fails until `maxAttempts` is spent. `unanchored`
|
|
1670
|
+
// extends it to a tail with no tool part to anchor on; no checkpoint
|
|
1671
|
+
// produces such a tail on this path today (see the note there), it is
|
|
1672
|
+
// here so a classifier change cannot reintroduce the prefill.
|
|
1673
|
+
uiMessages: trimTrailingAssistantPrefill(uiMessages, { unanchored: true }),
|
|
1674
|
+
runtimeCtx,
|
|
1675
|
+
admitted,
|
|
1676
|
+
// Crash recovery, driven server-side: nobody is watching this stream.
|
|
1677
|
+
trigger: "resume",
|
|
1678
|
+
});
|
|
1679
|
+
await res.text();
|
|
1680
|
+
const final = await storage.getRun(runId);
|
|
1681
|
+
return {
|
|
1682
|
+
runId,
|
|
1683
|
+
status: final?.status ?? "errored",
|
|
1684
|
+
attempt: final?.attempt ?? run.attempt ?? null,
|
|
1685
|
+
};
|
|
1686
|
+
}
|
|
1687
|
+
async function sweep(opts) {
|
|
1688
|
+
if (!dur.enabled)
|
|
1689
|
+
throw new Error("sweep requires durability.enabled");
|
|
1690
|
+
const maxAttempts = opts?.maxAttempts ?? 5;
|
|
1691
|
+
const { reaped } = await reapTick({ now: opts?.now, error: opts?.error });
|
|
1692
|
+
const resumed = [];
|
|
1693
|
+
for (const run of reaped) {
|
|
1694
|
+
// Give up on poison runs that keep dying — don't loop forever.
|
|
1695
|
+
if ((run.attempt ?? 1) >= maxAttempts)
|
|
1696
|
+
continue;
|
|
1697
|
+
const outcome = await resumeRun(run.id);
|
|
1698
|
+
if (outcome)
|
|
1699
|
+
resumed.push(outcome);
|
|
1700
|
+
}
|
|
1701
|
+
return { reaped, resumed };
|
|
1702
|
+
}
|
|
1703
|
+
async function cron(opts) {
|
|
1704
|
+
const errors = [];
|
|
1705
|
+
const describe = (e) => (e instanceof Error ? e.message : String(e));
|
|
1706
|
+
let due = { fired: 0, errors: 0, skipped: 0, parked: 0 };
|
|
1707
|
+
try {
|
|
1708
|
+
due = await runDue({ now: opts?.now });
|
|
1709
|
+
}
|
|
1710
|
+
catch (e) {
|
|
1711
|
+
errors.push(`runDue: ${describe(e)}`);
|
|
1712
|
+
}
|
|
1713
|
+
let reaped = [];
|
|
1714
|
+
let resumed = [];
|
|
1715
|
+
// Without durability there are no leases to reap — nothing to sweep.
|
|
1716
|
+
if (dur.enabled) {
|
|
1717
|
+
try {
|
|
1718
|
+
({ reaped, resumed } = await sweep({ now: opts?.now, maxAttempts: opts?.maxAttempts }));
|
|
1719
|
+
}
|
|
1720
|
+
catch (e) {
|
|
1721
|
+
errors.push(`sweep: ${describe(e)}`);
|
|
1722
|
+
}
|
|
1723
|
+
}
|
|
1724
|
+
// Limit counters: windows that ended more than a day ago are dead weight.
|
|
1725
|
+
if (config.limits?.store.prune) {
|
|
1726
|
+
try {
|
|
1727
|
+
await config.limits.store.prune((opts?.now ?? Date.now()) - 24 * 3_600_000);
|
|
1728
|
+
}
|
|
1729
|
+
catch (e) {
|
|
1730
|
+
errors.push(`limits.prune: ${describe(e)}`);
|
|
1731
|
+
}
|
|
1732
|
+
}
|
|
1733
|
+
return { due, reaped, resumed, errors };
|
|
1734
|
+
}
|
|
1735
|
+
async function applyApproval({ chatId, principal, decisions, request, turnContext, }) {
|
|
1736
|
+
const chat = await storage.getChat(principal, chatId);
|
|
1737
|
+
if (!chat)
|
|
1738
|
+
throw new Error(`Unknown chat: ${chatId}`);
|
|
1739
|
+
// Merge the decisions onto the persisted assistant message, then re-persist
|
|
1740
|
+
// only what changed. The agent continues from this updated history.
|
|
1741
|
+
const messages = await storage.loadMessages(principal, chatId);
|
|
1742
|
+
const changed = applyDecisions(messages, decisions);
|
|
1743
|
+
if (changed.length > 0)
|
|
1744
|
+
await storage.appendMessages(chatId, changed);
|
|
1745
|
+
// Route any just-decided SUBAGENT approvals (bubbled up via spawn_agent):
|
|
1746
|
+
// resume the sub thread with the decision, then fold its answer back into
|
|
1747
|
+
// the parent's spawn_agent result so this turn continues with the digest.
|
|
1748
|
+
const subApprovals = collectDecidedSubagentApprovals(messages);
|
|
1749
|
+
if (subApprovals.length > 0) {
|
|
1750
|
+
for (const sa of subApprovals) {
|
|
1751
|
+
const resumed = await resumeSubagentApproval(principal, sa.subChatId, sa.subApprovalIds.map((id) => ({
|
|
1752
|
+
approvalId: id,
|
|
1753
|
+
approved: sa.approved,
|
|
1754
|
+
reason: sa.reason,
|
|
1755
|
+
})), { parentRunId: sa.parentRunId, rootChatId: sa.rootChatId });
|
|
1756
|
+
if (resumed.pending?.length) {
|
|
1757
|
+
// The resumed sub paused on ANOTHER gated tool — re-bubble it as a
|
|
1758
|
+
// fresh data-subagent-approval part on the same parent message. The
|
|
1759
|
+
// undecided part keeps this chat awaiting_input (hasPendingApproval
|
|
1760
|
+
// below), and the next decision routes through here again, so
|
|
1761
|
+
// bubbling recurses per round.
|
|
1762
|
+
const nextApprovalId = generateSubApprovalId();
|
|
1763
|
+
appendSubagentApprovalPart(messages, sa.approvalId, {
|
|
1764
|
+
approvalId: nextApprovalId,
|
|
1765
|
+
subChatId: sa.subChatId,
|
|
1766
|
+
subApprovalIds: resumed.pending.map((p) => p.approvalId),
|
|
1767
|
+
spawnToolCallId: sa.spawnToolCallId,
|
|
1768
|
+
summaries: resumed.pending.map((p) => p.summary),
|
|
1769
|
+
// Carry the linkage forward so each further resume stays nested.
|
|
1770
|
+
parentRunId: sa.parentRunId,
|
|
1771
|
+
rootChatId: sa.rootChatId,
|
|
1772
|
+
});
|
|
1773
|
+
if (sa.spawnToolCallId) {
|
|
1774
|
+
overwriteToolResult(messages, sa.spawnToolCallId, {
|
|
1775
|
+
status: "awaiting_approval",
|
|
1776
|
+
threadId: sa.subChatId,
|
|
1777
|
+
pending: resumed.pending.map((p) => p.summary),
|
|
1778
|
+
note: "Awaiting the user's approval — do not retry; stop here, the user will decide and you'll continue automatically.",
|
|
1779
|
+
});
|
|
1780
|
+
}
|
|
1781
|
+
}
|
|
1782
|
+
else if (sa.spawnToolCallId) {
|
|
1783
|
+
overwriteToolResult(messages, sa.spawnToolCallId, {
|
|
1784
|
+
threadId: sa.subChatId,
|
|
1785
|
+
answer: resumed.answer,
|
|
1786
|
+
});
|
|
1787
|
+
}
|
|
1788
|
+
markSubagentApprovalResolved(messages, sa.approvalId);
|
|
1789
|
+
}
|
|
1790
|
+
// Persist the folded spawn_agent result (or re-bubbled approval part)
|
|
1791
|
+
// + resolved markers.
|
|
1792
|
+
await storage.appendMessages(chatId, messages);
|
|
1793
|
+
}
|
|
1794
|
+
// If the step had several gated calls and some are still undecided, DON'T
|
|
1795
|
+
// re-run yet — a tool call without a result makes convertToModelMessages
|
|
1796
|
+
// throw MissingToolResultsError. Persist the decision, stay awaiting_input,
|
|
1797
|
+
// and let the client submit the rest. Only continue once none are pending.
|
|
1798
|
+
if (hasPendingApproval(messages)) {
|
|
1799
|
+
return new Response(null, { status: 200 });
|
|
1800
|
+
}
|
|
1801
|
+
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(chat.agent, principal, request, turnContext);
|
|
1802
|
+
const admitted = await admitContinuation(principal, chat.agent, "decision");
|
|
1803
|
+
const { run, lease } = await openRun(principal, chatId, chat.agent);
|
|
1804
|
+
return buildTurn({
|
|
1805
|
+
resolvedChatId: chatId,
|
|
1806
|
+
principal,
|
|
1807
|
+
modelId,
|
|
1808
|
+
cfg,
|
|
1809
|
+
run,
|
|
1810
|
+
lease,
|
|
1811
|
+
// Resume: strip the paused turn's trailing text so the model call ends on
|
|
1812
|
+
// the tool result (a user turn), not an assistant prefill.
|
|
1813
|
+
uiMessages: trimTrailingAssistantPrefill(messages),
|
|
1814
|
+
runtimeCtx,
|
|
1815
|
+
// A human just decided something, so a human is present — even if this
|
|
1816
|
+
// turn parks again on the NEXT gate (see TurnTrigger).
|
|
1817
|
+
trigger: "decision",
|
|
1818
|
+
admitted,
|
|
1819
|
+
});
|
|
1820
|
+
}
|
|
1821
|
+
/**
|
|
1822
|
+
* Resume a turn paused on a client-handled tool (one with no `execute`, e.g.
|
|
1823
|
+
* `askUser`). Records the supplied `output` onto the matching tool call in the
|
|
1824
|
+
* persisted assistant message (input-available → output-available), then
|
|
1825
|
+
* re-runs the agent from history. Mirrors `applyApproval`: the result
|
|
1826
|
+
* round-trips as message state, not a client-message ingest (the handler only
|
|
1827
|
+
* ever accepts a fresh user message otherwise).
|
|
1828
|
+
*/
|
|
1829
|
+
async function applyToolResult({ chatId, principal, toolCallId, output, request, turnContext, }) {
|
|
1830
|
+
const chat = await storage.getChat(principal, chatId);
|
|
1831
|
+
if (!chat)
|
|
1832
|
+
throw new Error(`Unknown chat: ${chatId}`);
|
|
1833
|
+
const messages = await storage.loadMessages(principal, chatId);
|
|
1834
|
+
const changed = recordToolResult(messages, toolCallId, output);
|
|
1835
|
+
// Unknown or already-answered call: nothing to resume. Makes a duplicate
|
|
1836
|
+
// submit (double-tapped button, redelivered webhook) a no-op instead of a
|
|
1837
|
+
// spurious extra turn.
|
|
1838
|
+
if (changed.length === 0)
|
|
1839
|
+
return new Response(null, { status: 200 });
|
|
1840
|
+
// Another client tool call in the SAME step may still be awaiting a result;
|
|
1841
|
+
// re-running now would throw MissingToolResultsError. Record this one and
|
|
1842
|
+
// wait for the rest. Scope this to the messages we just touched: a call
|
|
1843
|
+
// abandoned turns ago is still `input-available` in stored history, and
|
|
1844
|
+
// counting it would park every future resume in this chat forever.
|
|
1845
|
+
if (hasPendingToolResult(changed)) {
|
|
1846
|
+
await storage.appendMessages(chatId, changed);
|
|
1847
|
+
return new Response(null, { status: 200 });
|
|
1848
|
+
}
|
|
1849
|
+
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(chat.agent, principal, request, turnContext);
|
|
1850
|
+
const admitted = await admitContinuation(principal, chat.agent, "decision");
|
|
1851
|
+
// The result is persisted WITH the run (atomically where the adapter can):
|
|
1852
|
+
// a busy chat refuses before anything is recorded, so the retry finds the
|
|
1853
|
+
// call still unanswered and resumes it — persisting first would make the
|
|
1854
|
+
// retry a "duplicate submit" no-op and lose the answer.
|
|
1855
|
+
const { run, lease } = await openTurn(principal, chatId, chat.agent, changed);
|
|
1856
|
+
return buildTurn({
|
|
1857
|
+
resolvedChatId: chatId,
|
|
1858
|
+
principal,
|
|
1859
|
+
modelId,
|
|
1860
|
+
cfg,
|
|
1861
|
+
run,
|
|
1862
|
+
lease,
|
|
1863
|
+
// Resume: same prefill guard as applyApproval (see trimTrailingAssistantPrefill).
|
|
1864
|
+
uiMessages: trimTrailingAssistantPrefill(messages),
|
|
1865
|
+
runtimeCtx,
|
|
1866
|
+
// A client-handled tool result is a human answering (see TurnTrigger).
|
|
1867
|
+
trigger: "decision",
|
|
1868
|
+
admitted,
|
|
1869
|
+
});
|
|
1870
|
+
}
|
|
1871
|
+
/**
|
|
1872
|
+
* Compact a conversation: summarize the live context window and append the
|
|
1873
|
+
* summary as a boundary marker, so later turns send the model the summary
|
|
1874
|
+
* instead of the turns it replaces (see `sliceAtCompaction`). Stored history is
|
|
1875
|
+
* never rewritten — nothing is lost, and a chat can be compacted repeatedly.
|
|
1876
|
+
*
|
|
1877
|
+
* Returns null when there's too little to be worth compacting (or the chat
|
|
1878
|
+
* isn't the caller's). Throws `PendingApprovalError` when the turn is paused:
|
|
1879
|
+
* the parked `tool_use` would fall behind the boundary and vanish from the
|
|
1880
|
+
* model view while stored history still gates every new message.
|
|
1881
|
+
*/
|
|
1882
|
+
async function compactChat({ chatId, principal, focus, signal, request, turnContext, }) {
|
|
1883
|
+
const chat = await storage.getChat(principal, chatId);
|
|
1884
|
+
if (!chat)
|
|
1885
|
+
return null;
|
|
1886
|
+
const stored = await storage.loadMessages(principal, chatId);
|
|
1887
|
+
// Only the live window is summarized: everything before the previous
|
|
1888
|
+
// boundary is already represented by that boundary's summary.
|
|
1889
|
+
const live = sliceAtCompaction(stored);
|
|
1890
|
+
if (live.length < MIN_COMPACTABLE_MESSAGES)
|
|
1891
|
+
return null;
|
|
1892
|
+
if (hasPendingApproval(live) || hasPendingToolResult([live[live.length - 1]])) {
|
|
1893
|
+
throw new PendingApprovalError("This chat is awaiting a decision — resolve the pending request before compacting.");
|
|
1894
|
+
}
|
|
1895
|
+
const { cfg, modelId } = await resolveAgentConfig(chat.agent, principal, request, turnContext);
|
|
1896
|
+
// Tool definitions only affect how a recorded tool RESULT is rendered for the
|
|
1897
|
+
// model (`toModelOutput`); an unknown tool falls back to its raw JSON rather
|
|
1898
|
+
// than failing. So take every definition available without I/O — the agent's
|
|
1899
|
+
// own plus the harness ones — and deliberately skip opening dynamic sources
|
|
1900
|
+
// (MCP / connections): summarizing must not depend on a remote server being
|
|
1901
|
+
// reachable, and raw JSON is a fine thing to summarize.
|
|
1902
|
+
const harness = config.harnessTools ?? {};
|
|
1903
|
+
const tools = {
|
|
1904
|
+
...cfg.tools,
|
|
1905
|
+
...harness.app,
|
|
1906
|
+
...harness.sandbox,
|
|
1907
|
+
};
|
|
1908
|
+
// Summarizing needs no reasoning block, and on a model that thinks by
|
|
1909
|
+
// DEFAULT that block spends the same output budget as the summary — leaving
|
|
1910
|
+
// the call truncated with nothing usable. `"minimal"` is the effort every
|
|
1911
|
+
// provider mapping turns into "thinking off", so reuse the host's hook rather
|
|
1912
|
+
// than hardcoding provider shapes here.
|
|
1913
|
+
const noThinking = config.resolveReasoningOptions?.(modelId, "minimal")?.providerOptions;
|
|
1914
|
+
const run = await storage.createRun(principal, {
|
|
1915
|
+
chatId,
|
|
1916
|
+
agent: chat.agent,
|
|
1917
|
+
// Not a conversational turn — stats/rollups filter on kind so a
|
|
1918
|
+
// compaction's cache-less usage never depresses an agent's hit rate.
|
|
1919
|
+
kind: "compaction",
|
|
1920
|
+
});
|
|
1921
|
+
try {
|
|
1922
|
+
const { text, truncated, usage } = await summarizeForCompaction({
|
|
1923
|
+
model: cfg.model,
|
|
1924
|
+
messages: await toModelMessages(live, { tools }),
|
|
1925
|
+
...(focus ? { focus } : {}),
|
|
1926
|
+
...(noThinking ? { providerOptions: noThinking } : {}),
|
|
1927
|
+
...(signal ? { signal } : {}),
|
|
1928
|
+
});
|
|
1929
|
+
if (!text)
|
|
1930
|
+
throw new Error("The model returned an empty summary.");
|
|
1931
|
+
// Still truncated after the shortened retry (see summarizeForCompaction).
|
|
1932
|
+
// A summary cut off mid-thought must not be stored — it would become the
|
|
1933
|
+
// permanent head of every later prompt. History is untouched, so retrying
|
|
1934
|
+
// is safe.
|
|
1935
|
+
if (truncated) {
|
|
1936
|
+
throw new Error("Couldn't fit a summary of this conversation — nothing was changed. Try /compact again, or /new to start fresh.");
|
|
1937
|
+
}
|
|
1938
|
+
const cost = computeCost(config.pricing?.(modelId), usage);
|
|
1939
|
+
// `internal` so the client projection never shows a "user" message the user
|
|
1940
|
+
// didn't write; `sendToModel` stays default-true so the model DOES see it.
|
|
1941
|
+
// Role `user` because after slicing this is the conversation's first
|
|
1942
|
+
// message, and a model prompt may not open with an assistant turn.
|
|
1943
|
+
await storage.appendMessages(chatId, [
|
|
1944
|
+
{
|
|
1945
|
+
id: generateMessageId(),
|
|
1946
|
+
role: "user",
|
|
1947
|
+
parts: [
|
|
1948
|
+
{ type: "text", text },
|
|
1949
|
+
{
|
|
1950
|
+
type: COMPACTION_PART_TYPE,
|
|
1951
|
+
data: { compacted: live.length, at: Date.now() },
|
|
1952
|
+
},
|
|
1953
|
+
],
|
|
1954
|
+
metadata: {
|
|
1955
|
+
visibility: "internal",
|
|
1956
|
+
createdAt: Date.now(),
|
|
1957
|
+
model: modelId,
|
|
1958
|
+
usage: {
|
|
1959
|
+
inputTokens: usage.inputTokens,
|
|
1960
|
+
outputTokens: usage.outputTokens,
|
|
1961
|
+
...(usage.cacheReadTokens ? { cacheReadTokens: usage.cacheReadTokens } : {}),
|
|
1962
|
+
...(usage.cacheWriteTokens ? { cacheWriteTokens: usage.cacheWriteTokens } : {}),
|
|
1963
|
+
},
|
|
1964
|
+
cost,
|
|
1965
|
+
},
|
|
1966
|
+
},
|
|
1967
|
+
]);
|
|
1968
|
+
await storage.updateRun(run.id, {
|
|
1969
|
+
status: "completed",
|
|
1970
|
+
usageInputTokens: usage.inputTokens,
|
|
1971
|
+
usageOutputTokens: usage.outputTokens,
|
|
1972
|
+
usageCacheReadTokens: usage.cacheReadTokens || undefined,
|
|
1973
|
+
usageCacheWriteTokens: usage.cacheWriteTokens || undefined,
|
|
1974
|
+
costTotal: cost,
|
|
1975
|
+
endedAt: Date.now(),
|
|
1976
|
+
});
|
|
1977
|
+
await storage.recordUsage(principal, {
|
|
1978
|
+
runId: run.id,
|
|
1979
|
+
model: modelId,
|
|
1980
|
+
inputTokens: usage.inputTokens,
|
|
1981
|
+
outputTokens: usage.outputTokens,
|
|
1982
|
+
...(usage.cacheReadTokens ? { cacheReadTokens: usage.cacheReadTokens } : {}),
|
|
1983
|
+
...(usage.cacheWriteTokens ? { cacheWriteTokens: usage.cacheWriteTokens } : {}),
|
|
1984
|
+
cost,
|
|
1985
|
+
});
|
|
1986
|
+
return { summary: text, compacted: live.length };
|
|
1987
|
+
}
|
|
1988
|
+
catch (e) {
|
|
1989
|
+
await storage
|
|
1990
|
+
.updateRun(run.id, {
|
|
1991
|
+
status: "errored",
|
|
1992
|
+
error: e instanceof Error ? e.message : String(e),
|
|
1993
|
+
endedAt: Date.now(),
|
|
1994
|
+
})
|
|
1995
|
+
.catch(() => { });
|
|
1996
|
+
throw e;
|
|
1997
|
+
}
|
|
1998
|
+
}
|
|
1999
|
+
async function loadHistory({ chatId, principal, }) {
|
|
2000
|
+
const messages = await storage.loadMessages(principal, chatId);
|
|
2001
|
+
return toClientMessages(messages);
|
|
2002
|
+
}
|
|
2003
|
+
async function listChats({ principal, limit, kind, }) {
|
|
2004
|
+
// No default kind filter — passing nothing returns ALL chats (unchanged
|
|
2005
|
+
// behavior). Callers that want only user-facing chats pass kind: "user".
|
|
2006
|
+
return storage.listChats(principal, { limit, kind });
|
|
2007
|
+
}
|
|
2008
|
+
async function listRuns({ principal, limit, }) {
|
|
2009
|
+
return storage.listRuns(principal, { limit });
|
|
2010
|
+
}
|
|
2011
|
+
async function getChatById({ principal, id, }) {
|
|
2012
|
+
return storage.getChat(principal, id);
|
|
2013
|
+
}
|
|
2014
|
+
async function getRun({ principal, runId, }) {
|
|
2015
|
+
const run = await storage.getRun(runId);
|
|
2016
|
+
if (!run)
|
|
2017
|
+
return null;
|
|
2018
|
+
// Ownership: the run is the caller's only if its chat is (getChat is
|
|
2019
|
+
// principal-scoped). Guards cross-tenant reads/writes keyed by runId.
|
|
2020
|
+
const chat = await storage.getChat(principal, run.chatId);
|
|
2021
|
+
return chat ? run : null;
|
|
2022
|
+
}
|
|
2023
|
+
/** A fresh user message from plain text (for cron-fired turns). */
|
|
2024
|
+
function userMessageOf(text) {
|
|
2025
|
+
return {
|
|
2026
|
+
id: generateMessageId(),
|
|
2027
|
+
role: "user",
|
|
2028
|
+
parts: [{ type: "text", text }],
|
|
2029
|
+
metadata: { visibility: "user", createdAt: Date.now() },
|
|
2030
|
+
};
|
|
2031
|
+
}
|
|
2032
|
+
/** Merged agent directory (code-declared + dynamic) — name/title/description. */
|
|
2033
|
+
async function agentInfos(principal) {
|
|
2034
|
+
const stat = Object.entries(config.agents).map(([name, factory]) => ({
|
|
2035
|
+
name,
|
|
2036
|
+
title: factory.meta?.title,
|
|
2037
|
+
description: factory.meta?.description,
|
|
2038
|
+
hidden: factory.meta?.hidden,
|
|
2039
|
+
}));
|
|
2040
|
+
const dyn = (await config.dynamicAgents?.list(principal)) ?? [];
|
|
2041
|
+
// Dedup by name, static first — mirrors resolve-time precedence (a
|
|
2042
|
+
// code-declared agent shadows a same-named dynamic one).
|
|
2043
|
+
const seen = new Set(stat.map((a) => a.name));
|
|
2044
|
+
return [...stat, ...dyn.filter((a) => !seen.has(a.name) && seen.add(a.name))];
|
|
2045
|
+
}
|
|
2046
|
+
/** Concatenated text of the last assistant message in a chat (subagent answer). */
|
|
2047
|
+
function lastAssistantText(messages) {
|
|
2048
|
+
const last = [...messages].reverse().find((m) => m.role === "assistant");
|
|
2049
|
+
if (!last)
|
|
2050
|
+
return "";
|
|
2051
|
+
return (last.parts ?? [])
|
|
2052
|
+
.filter((p) => p.type === "text")
|
|
2053
|
+
.map((p) => p.text)
|
|
2054
|
+
.join("");
|
|
2055
|
+
}
|
|
2056
|
+
/**
|
|
2057
|
+
* Run one agent turn to completion, server-side, and report how it ended —
|
|
2058
|
+
* THE single non-streaming path. `runAgent` (programmatic / test runs),
|
|
2059
|
+
* `spawn_agent` (subagent delegation) and `runDue` (scheduled fires) all
|
|
2060
|
+
* funnel here, so approval semantics, telemetry linkage and the "did it
|
|
2061
|
+
* park?" verdict cannot drift between them.
|
|
2062
|
+
*
|
|
2063
|
+
* `chatId` absent (or "") → a fresh internal `sub_…` thread; present → that
|
|
2064
|
+
* chat continues with its full history (created as an internal thread when
|
|
2065
|
+
* it doesn't exist yet). `autoApprove` is REQUIRED, not defaulted: autonomous
|
|
2066
|
+
* runs (subagents, tests) waive HITL gates, while a scheduled fire keeps them
|
|
2067
|
+
* — and parks on them, which is the whole point of `parked`.
|
|
2068
|
+
*
|
|
2069
|
+
* `parked` = the turn ended waiting on a human (an undecided approval or
|
|
2070
|
+
* client tool result) rather than with an answer; the stream ending is not
|
|
2071
|
+
* the same as an answer, and a parked unattended turn has spent its tokens
|
|
2072
|
+
* and produced nothing. `pending` lifts an interactive subagent's undecided
|
|
2073
|
+
* approvals so `spawn_agent` can bubble them up to the parent turn. Both are
|
|
2074
|
+
* read from the fold over the persisted log — the same truth every
|
|
2075
|
+
* interactive path trusts — not the run row.
|
|
2076
|
+
*/
|
|
2077
|
+
async function runToCompletion(o) {
|
|
2078
|
+
// `||` (not `??`): an empty-string id must mint a fresh one, or the chat is
|
|
2079
|
+
// created with id "" and the UI can never open it.
|
|
2080
|
+
const chatId = o.chatId?.trim() || generateSubchatId();
|
|
2081
|
+
let chat = await storage.getChat(o.principal, chatId);
|
|
2082
|
+
if (!chat) {
|
|
2083
|
+
chat = await storage.createChat(o.principal, {
|
|
2084
|
+
id: chatId,
|
|
2085
|
+
agent: o.agent,
|
|
2086
|
+
title: firstText(o.message)?.slice(0, 80),
|
|
2087
|
+
// Runtime-created thread: stamp it internal so a host filtering its
|
|
2088
|
+
// chat list on kind never shows it (see ChatRecord.kind).
|
|
2089
|
+
kind: "internal",
|
|
2090
|
+
});
|
|
2091
|
+
}
|
|
2092
|
+
const prior = await storage.loadMessages(o.principal, chatId);
|
|
2093
|
+
// Same rule as handleChat: a new message while the thread is parked on an
|
|
2094
|
+
// approval IS the decision — close the stale gate as declined (persisted)
|
|
2095
|
+
// and carry on. Otherwise a continued thread (`runAgent({ threadId })`,
|
|
2096
|
+
// `spawn_agent({ threadId })`) would run with an undecided gate in history
|
|
2097
|
+
// and this turn's verdict below would re-report that stale approval.
|
|
2098
|
+
if (hasPendingApproval(prior)) {
|
|
2099
|
+
const declined = declinePendingApprovals(prior);
|
|
2100
|
+
if (declined.length > 0)
|
|
2101
|
+
await storage.appendMessages(chatId, declined);
|
|
2102
|
+
}
|
|
2103
|
+
// Gate before the message persists — a refused schedule/subagent turn
|
|
2104
|
+
// leaves nothing behind, same as handleChat. Same for a busy chat: open
|
|
2105
|
+
// the run first, persist the message only once it is ours to run.
|
|
2106
|
+
const admitted = await admitTurn(o.principal, o.agent, o.trigger);
|
|
2107
|
+
const { cfg, modelId, runtimeCtx, run, lease } = await withReservation(admitted, async () => {
|
|
2108
|
+
const resolved = await resolveAgentConfig(o.agent, o.principal, undefined, o.turnContext);
|
|
2109
|
+
const opened = await openTurn(o.principal, chatId, o.agent, [o.message]);
|
|
2110
|
+
return { ...resolved, ...opened };
|
|
2111
|
+
});
|
|
2112
|
+
const onProgress = o.onProgress;
|
|
2113
|
+
const res = await buildTurn({
|
|
2114
|
+
resolvedChatId: chatId,
|
|
2115
|
+
principal: o.principal,
|
|
2116
|
+
modelId,
|
|
2117
|
+
cfg,
|
|
2118
|
+
run,
|
|
2119
|
+
lease,
|
|
2120
|
+
uiMessages: [...prior, o.message],
|
|
2121
|
+
runtimeCtx,
|
|
2122
|
+
admitted,
|
|
2123
|
+
autoApprove: o.autoApprove,
|
|
2124
|
+
blockGated: o.blockGated,
|
|
2125
|
+
promptCaching: o.promptCaching,
|
|
2126
|
+
depth: o.depth ?? 0,
|
|
2127
|
+
parentRunId: o.parentRunId,
|
|
2128
|
+
rootChatId: o.rootChatId,
|
|
2129
|
+
trigger: o.trigger,
|
|
2130
|
+
// Forward progressive UI-message snapshots to the caller (spawn_agent
|
|
2131
|
+
// streams them into the parent turn). Observation must never break the
|
|
2132
|
+
// run — the primary stream is drained independently below.
|
|
2133
|
+
observeStream: onProgress
|
|
2134
|
+
? (observed) => {
|
|
2135
|
+
void (async () => {
|
|
2136
|
+
try {
|
|
2137
|
+
const byId = new Map();
|
|
2138
|
+
for await (const m of readUIMessageStream({
|
|
2139
|
+
stream: observed,
|
|
2140
|
+
})) {
|
|
2141
|
+
const msg = m;
|
|
2142
|
+
byId.set(msg.id, msg);
|
|
2143
|
+
onProgress([o.message, ...byId.values()]);
|
|
2144
|
+
}
|
|
2145
|
+
}
|
|
2146
|
+
catch {
|
|
2147
|
+
// ignore — live progress is best-effort
|
|
2148
|
+
}
|
|
2149
|
+
})();
|
|
2150
|
+
}
|
|
2151
|
+
: undefined,
|
|
2152
|
+
});
|
|
2153
|
+
await res.text();
|
|
2154
|
+
const after = await storage.loadMessages(o.principal, chatId);
|
|
2155
|
+
// The verdict is about THIS turn, so read the FINAL message only (the same
|
|
2156
|
+
// rule hasPendingApproval applies): an undecided gate in an older message
|
|
2157
|
+
// is an abandoned turn, not this one's pause point.
|
|
2158
|
+
const last = after[after.length - 1];
|
|
2159
|
+
const pending = last?.role === "assistant" ? collectPendingApprovals([last]) : [];
|
|
2160
|
+
const parked = pending.length > 0 ||
|
|
2161
|
+
hasPendingApproval(after) ||
|
|
2162
|
+
(!!last && hasPendingToolResult([last]));
|
|
2163
|
+
return {
|
|
2164
|
+
chatId,
|
|
2165
|
+
answer: pending.length > 0 ? "" : lastAssistantText(after),
|
|
2166
|
+
parked,
|
|
2167
|
+
...(pending.length > 0 ? { pending } : {}),
|
|
2168
|
+
};
|
|
2169
|
+
}
|
|
2170
|
+
/**
|
|
2171
|
+
* Resume a subagent thread after its bubbled-up approval was decided: apply
|
|
2172
|
+
* the decision to the sub's gated tool, re-run the sub from history (the
|
|
2173
|
+
* approved tool executes, or returns denied), and return the sub's digest.
|
|
2174
|
+
*
|
|
2175
|
+
* If the resumed sub pauses AGAIN on a new gated tool, its pending
|
|
2176
|
+
* approval(s) are returned so the caller (`applyApproval`) can re-bubble
|
|
2177
|
+
* them to the parent as a fresh `data-subagent-approval` part — bubbling
|
|
2178
|
+
* recurses one round per decision round-trip.
|
|
2179
|
+
*/
|
|
2180
|
+
async function resumeSubagentApproval(principal, subChatId, decisions,
|
|
2181
|
+
/** Telemetry linkage of the original sub-turn, so the trace stays nested. */
|
|
2182
|
+
linkage) {
|
|
2183
|
+
const subChat = await storage.getChat(principal, subChatId);
|
|
2184
|
+
if (!subChat)
|
|
2185
|
+
return { answer: "" };
|
|
2186
|
+
const subMessages = await storage.loadMessages(principal, subChatId);
|
|
2187
|
+
const changed = applyDecisions(subMessages, decisions);
|
|
2188
|
+
if (changed.length > 0)
|
|
2189
|
+
await storage.appendMessages(subChatId, changed);
|
|
2190
|
+
// Another gated call in the same sub step still undecided → can't finish.
|
|
2191
|
+
if (hasPendingApproval(subMessages))
|
|
2192
|
+
return { answer: "" };
|
|
2193
|
+
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(subChat.agent, principal);
|
|
2194
|
+
const admitted = await admitContinuation(principal, subChat.agent, "decision");
|
|
2195
|
+
const { run, lease } = await openRun(principal, subChatId, subChat.agent);
|
|
2196
|
+
const res = await buildTurn({
|
|
2197
|
+
resolvedChatId: subChatId,
|
|
2198
|
+
principal,
|
|
2199
|
+
modelId,
|
|
2200
|
+
cfg,
|
|
2201
|
+
run,
|
|
2202
|
+
admitted,
|
|
2203
|
+
lease,
|
|
2204
|
+
// Resume: same prefill guard as applyApproval — the paused sub turn may
|
|
2205
|
+
// have narrated after its gated call, and that text must not reach the
|
|
2206
|
+
// model as an assistant prefill.
|
|
2207
|
+
uiMessages: trimTrailingAssistantPrefill(subMessages),
|
|
2208
|
+
runtimeCtx,
|
|
2209
|
+
autoApprove: false, // keep gating any further writes in the sub
|
|
2210
|
+
depth: 1,
|
|
2211
|
+
// Keep the resumed turn nested under the original trace/session.
|
|
2212
|
+
parentRunId: linkage?.parentRunId,
|
|
2213
|
+
rootChatId: linkage?.rootChatId,
|
|
2214
|
+
// Reached only by a human deciding the bubbled-up approval.
|
|
2215
|
+
trigger: "decision",
|
|
2216
|
+
});
|
|
2217
|
+
await res.text();
|
|
2218
|
+
const after = await storage.loadMessages(principal, subChatId);
|
|
2219
|
+
// Paused on a SECOND gated tool → lift it so the caller re-bubbles.
|
|
2220
|
+
const pending = collectPendingApprovals(after);
|
|
2221
|
+
if (pending.length > 0)
|
|
2222
|
+
return { answer: "", pending };
|
|
2223
|
+
return { answer: lastAssistantText(after) };
|
|
2224
|
+
}
|
|
2225
|
+
async function schedule({ principal, agent, cron, timezone, prompt, delivery, onParked, }) {
|
|
2226
|
+
// Accept code-declared or dynamic (UI-created) agents.
|
|
2227
|
+
const known = !!config.agents[agent] || !!(await config.dynamicAgents?.resolve(agent, principal));
|
|
2228
|
+
if (!known)
|
|
2229
|
+
throw new Error(`Unknown agent: ${agent}`);
|
|
2230
|
+
const now = Date.now();
|
|
2231
|
+
let nextRunAt = now; // one-shot → fire on the next runDue
|
|
2232
|
+
if (cron) {
|
|
2233
|
+
const next = config.cron?.(cron, now, timezone);
|
|
2234
|
+
if (next == null) {
|
|
2235
|
+
throw new Error("Recurring schedules need a `cron` evaluator in the runtime config (or the expression never fires)");
|
|
2236
|
+
}
|
|
2237
|
+
nextRunAt = next;
|
|
2238
|
+
}
|
|
2239
|
+
return storage.createSchedule(principal, {
|
|
2240
|
+
agent,
|
|
2241
|
+
cron,
|
|
2242
|
+
timezone,
|
|
2243
|
+
prompt,
|
|
2244
|
+
nextRunAt,
|
|
2245
|
+
delivery,
|
|
2246
|
+
onParked,
|
|
2247
|
+
});
|
|
2248
|
+
}
|
|
2249
|
+
async function runDue(opts) {
|
|
2250
|
+
const now = opts?.now ?? Date.now();
|
|
2251
|
+
const claimed = await storage.claimDueSchedules(now, dur.leaseTtlMs, dur.instanceId);
|
|
2252
|
+
let fired = 0;
|
|
2253
|
+
let errors = 0;
|
|
2254
|
+
let skipped = 0;
|
|
2255
|
+
let parked = 0;
|
|
2256
|
+
for (const s of claimed) {
|
|
2257
|
+
const next = s.cron ? (config.cron?.(s.cron, now, s.timezone) ?? null) : null;
|
|
2258
|
+
// Consume the occurrence BEFORE firing, not after. The claim lease is
|
|
2259
|
+
// stamped once (never heartbeated), so while a fire is in flight the row
|
|
2260
|
+
// still reads `enabled AND nextRunAt <= now` — and once the lease TTL
|
|
2261
|
+
// (default 30s) lapses, ANY claimer re-fires the same occurrence: a
|
|
2262
|
+
// second node, a deploy overlap, the POST /cron/run backstop, even this
|
|
2263
|
+
// same instance's next tick. A scheduled agent turn routinely outlives
|
|
2264
|
+
// the TTL, and each duplicate is a full paid turn. Advancing first makes
|
|
2265
|
+
// the occurrence at-most-once: after this write the row is no longer due
|
|
2266
|
+
// (or no longer enabled, for one-shots), so it cannot be claimed again
|
|
2267
|
+
// regardless of lease state. Run-level durability — not an occurrence
|
|
2268
|
+
// retry — is what recovers a fire that dies mid-turn: the run row
|
|
2269
|
+
// survives and `sweep()` resumes it. For work that is billed per
|
|
2270
|
+
// attempt, "ran twice" is strictly worse than "ran once, resumed".
|
|
2271
|
+
try {
|
|
2272
|
+
await storage.completeSchedule(s.id, { nextRunAt: next ?? now, lastRunAt: now, enabled: next != null }, dur.instanceId);
|
|
2273
|
+
}
|
|
2274
|
+
catch (e) {
|
|
2275
|
+
// Cannot consume the occurrence → do NOT fire. Firing anyway would
|
|
2276
|
+
// leave the row due, and every subsequent tick would fire it again —
|
|
2277
|
+
// the swallowed version of this write was itself a repeat-fire bug.
|
|
2278
|
+
console.error(`[schedule] failed to advance ${s.id}; skipping this fire:`, e);
|
|
2279
|
+
errors++;
|
|
2280
|
+
continue;
|
|
2281
|
+
}
|
|
2282
|
+
try {
|
|
2283
|
+
const principal = await reconstructPrincipal(s.identity);
|
|
2284
|
+
// Parked-predecessor guard: when the LAST fire's chat is still waiting
|
|
2285
|
+
// on a human (an undecided approval or tool result), firing again just
|
|
2286
|
+
// mints another parked chat and pays for the tokens up to the gate —
|
|
2287
|
+
// on every tick, forever. `onParked: "skip"` (the default) consumes
|
|
2288
|
+
// the occurrence without running instead; the answer is one approval
|
|
2289
|
+
// away in the existing chat, and the next occurrence re-checks. The
|
|
2290
|
+
// truth is the message-log fold, not run rows — same predicates the
|
|
2291
|
+
// interactive paths trust. Best-effort: a failed check must never
|
|
2292
|
+
// block a fire.
|
|
2293
|
+
if ((s.onParked ?? "skip") === "skip" && s.lastChatId) {
|
|
2294
|
+
try {
|
|
2295
|
+
const prior = await storage.loadMessages(principal, s.lastChatId);
|
|
2296
|
+
const last = prior[prior.length - 1];
|
|
2297
|
+
if (hasPendingApproval(prior) || (last && hasPendingToolResult([last]))) {
|
|
2298
|
+
console.warn(`[schedule] ${s.id} (${s.agent}): previous fire's chat ${s.lastChatId} ` +
|
|
2299
|
+
`is still waiting on a human — skipping this occurrence (onParked: skip)`);
|
|
2300
|
+
skipped++;
|
|
2301
|
+
continue;
|
|
2302
|
+
}
|
|
2303
|
+
}
|
|
2304
|
+
catch (e) {
|
|
2305
|
+
console.error(`[schedule] ${s.id}: parked check failed; firing anyway:`, e);
|
|
2306
|
+
}
|
|
2307
|
+
}
|
|
2308
|
+
// Give the host first refusal (e.g. run it in the user's channel thread
|
|
2309
|
+
// and stream the answer there). A hook that throws is treated as "not
|
|
2310
|
+
// handled": the fire still happens the default way, so a broken
|
|
2311
|
+
// delivery target degrades to a normal chat run instead of losing the
|
|
2312
|
+
// occurrence entirely. Hosts: only throw BEFORE doing paid work — a
|
|
2313
|
+
// hook that already ran the agent and then throws (say, on delivery)
|
|
2314
|
+
// makes the fallback a second full turn for the same occurrence.
|
|
2315
|
+
let handled = false;
|
|
2316
|
+
if (config.runSchedule) {
|
|
2317
|
+
try {
|
|
2318
|
+
const outcome = await config.runSchedule({ schedule: s, principal });
|
|
2319
|
+
handled = typeof outcome === "boolean" ? outcome : outcome.handled;
|
|
2320
|
+
// The host owns the chat a channel fire ran in, so it is the only
|
|
2321
|
+
// one that can report it. Same accounting as the default path when
|
|
2322
|
+
// it does: stamp the chat for the next `onParked` check, and count
|
|
2323
|
+
// a park that nobody can answer.
|
|
2324
|
+
if (handled && typeof outcome === "object") {
|
|
2325
|
+
if (outcome.chatId) {
|
|
2326
|
+
await storage.noteScheduleFire?.(s.id, {
|
|
2327
|
+
lastChatId: outcome.chatId,
|
|
2328
|
+
firedAt: now,
|
|
2329
|
+
});
|
|
2330
|
+
}
|
|
2331
|
+
if (outcome.parked) {
|
|
2332
|
+
parked++;
|
|
2333
|
+
console.warn(`[schedule] ${s.id} (${s.agent}): host-delivered fire parked waiting on a ` +
|
|
2334
|
+
`human${outcome.chatId ? ` in chat ${outcome.chatId}` : ""} — no answer was delivered`);
|
|
2335
|
+
}
|
|
2336
|
+
}
|
|
2337
|
+
}
|
|
2338
|
+
catch (e) {
|
|
2339
|
+
console.error(`[schedule] runSchedule hook failed for ${s.id}:`, e);
|
|
2340
|
+
}
|
|
2341
|
+
}
|
|
2342
|
+
if (!handled) {
|
|
2343
|
+
const chat = await storage.createChat(principal, {
|
|
2344
|
+
agent: s.agent,
|
|
2345
|
+
title: s.prompt.slice(0, 80),
|
|
2346
|
+
});
|
|
2347
|
+
// Remember where this fire ran so the NEXT occurrence can see a
|
|
2348
|
+
// parked predecessor. Stamped before the turn (a parked turn still
|
|
2349
|
+
// "completes" the await); optional — see StorageAdapter. `firedAt`
|
|
2350
|
+
// is the lastRunAt this fire's completeSchedule wrote above, so a
|
|
2351
|
+
// delayed stamp can't clobber a newer occurrence's.
|
|
2352
|
+
await storage.noteScheduleFire?.(s.id, { lastChatId: chat.id, firedAt: now });
|
|
2353
|
+
const outcome = await runToCompletion({
|
|
2354
|
+
principal,
|
|
2355
|
+
agent: s.agent,
|
|
2356
|
+
message: userMessageOf(s.prompt),
|
|
2357
|
+
chatId: chat.id,
|
|
2358
|
+
// A cron fire keeps HITL gates: nobody is here to waive them, and a
|
|
2359
|
+
// gated tool parks the turn (reported below) instead of running.
|
|
2360
|
+
autoApprove: false,
|
|
2361
|
+
trigger: "schedule",
|
|
2362
|
+
});
|
|
2363
|
+
if (outcome.parked) {
|
|
2364
|
+
// Say it on THIS tick, not the next one: a one-shot has no next
|
|
2365
|
+
// occurrence, and even a cron's operator shouldn't have to wait
|
|
2366
|
+
// for the skip message to learn that a paid fire produced nothing.
|
|
2367
|
+
parked++;
|
|
2368
|
+
console.warn(`[schedule] ${s.id} (${s.agent}): fire parked waiting on a human in ` +
|
|
2369
|
+
`chat ${chat.id} — no answer was delivered`);
|
|
2370
|
+
}
|
|
2371
|
+
}
|
|
2372
|
+
fired++;
|
|
2373
|
+
}
|
|
2374
|
+
catch (e) {
|
|
2375
|
+
// The occurrence is already consumed (above), so this cannot repeat-
|
|
2376
|
+
// fire — but a silent `errors++` left hosts unable to see WHY a
|
|
2377
|
+
// schedule produced nothing.
|
|
2378
|
+
console.error(`[schedule] fire failed for ${s.id} (${s.agent}):`, e);
|
|
2379
|
+
errors++;
|
|
2380
|
+
}
|
|
2381
|
+
}
|
|
2382
|
+
return { fired, errors, skipped, parked };
|
|
2383
|
+
}
|
|
2384
|
+
const self = {
|
|
2385
|
+
listAgents: agentInfos,
|
|
2386
|
+
handleChat,
|
|
2387
|
+
runAgent: async ({ principal, agent, prompt, threadId, blockGated = true, turnContext, promptCaching,
|
|
2388
|
+
// No client stream: assume unattended unless the host says otherwise, so
|
|
2389
|
+
// a park here can't be silently filtered out as "someone's watching".
|
|
2390
|
+
trigger = "programmatic", }) => {
|
|
2391
|
+
const r = await runToCompletion({
|
|
2392
|
+
principal,
|
|
2393
|
+
agent,
|
|
2394
|
+
message: userMessageOf(prompt),
|
|
2395
|
+
chatId: threadId,
|
|
2396
|
+
depth: 1,
|
|
2397
|
+
// Autonomous: no human to authorize (blockGated stubs the gated tools).
|
|
2398
|
+
autoApprove: true,
|
|
2399
|
+
blockGated,
|
|
2400
|
+
turnContext,
|
|
2401
|
+
promptCaching,
|
|
2402
|
+
trigger,
|
|
2403
|
+
});
|
|
2404
|
+
return { threadId: r.chatId, answer: r.answer };
|
|
2405
|
+
},
|
|
2406
|
+
compactChat,
|
|
2407
|
+
loadHistory,
|
|
2408
|
+
listChats,
|
|
2409
|
+
getChat: getChatById,
|
|
2410
|
+
listRuns,
|
|
2411
|
+
getRun,
|
|
2412
|
+
resume: resumeRun,
|
|
2413
|
+
sweep,
|
|
2414
|
+
cron,
|
|
2415
|
+
applyApproval,
|
|
2416
|
+
applyToolResult,
|
|
2417
|
+
schedule,
|
|
2418
|
+
listSchedules: ({ principal }) => storage.listSchedules(principal),
|
|
2419
|
+
unschedule: ({ principal, id }) => storage.deleteSchedule(principal, id),
|
|
2420
|
+
runScheduleNow: ({ principal, id }) => storage.bumpSchedule(principal, id),
|
|
2421
|
+
runDue,
|
|
2422
|
+
};
|
|
2423
|
+
return self;
|
|
2424
|
+
}
|
|
2425
|
+
//# sourceMappingURL=runtime.js.map
|