@intentface/latch-core 0.12.0 → 0.13.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/models/catalog.d.ts +7 -0
- package/dist/models/catalog.d.ts.map +1 -1
- package/dist/models/catalog.js +15 -5
- package/dist/models/catalog.js.map +1 -1
- package/dist/runtime.d.ts +54 -3
- package/dist/runtime.d.ts.map +1 -1
- package/dist/runtime.js +345 -138
- package/dist/runtime.js.map +1 -1
- package/dist/storage.d.ts +11 -0
- package/dist/storage.d.ts.map +1 -1
- package/package.json +3 -3
package/dist/runtime.js
CHANGED
|
@@ -6,6 +6,58 @@ import { computeCost } from "./pricing.js";
|
|
|
6
6
|
import { defaultPromptCachingPlan, markLastFunctionTool, mergeProviderOptions, toolsetHash, } from "./prompt-caching.js";
|
|
7
7
|
import { LimitExceededError, exceededPolicy, remainingTokens, windowRef, } from "./limits.js";
|
|
8
8
|
import { compose, fireClientToolResults, insertStepParts, isNewSession, lastUserTextOf, renderTranscript, replaceLastUserText, wrapTools, } from "./extensions/index.js";
|
|
9
|
+
/**
|
|
10
|
+
* The `data-subagent-progress` writer for one delegation: `now` sends a
|
|
11
|
+
* snapshot immediately, `throttled` at most every 250ms (each write re-sends
|
|
12
|
+
* the whole snapshot). Shared by `spawn_agent` and the approval resume so both
|
|
13
|
+
* write the SAME part id — a resumed delegation updates the one the client is
|
|
14
|
+
* already showing instead of adding a second.
|
|
15
|
+
*/
|
|
16
|
+
function subagentProgressWriter(writer, ids) {
|
|
17
|
+
const now = (messages) => {
|
|
18
|
+
writer.write({
|
|
19
|
+
type: "data-subagent-progress",
|
|
20
|
+
id: `subagent_${ids.toolCallId ?? ids.threadId}`,
|
|
21
|
+
data: { ...ids, messages },
|
|
22
|
+
});
|
|
23
|
+
};
|
|
24
|
+
let lastWrite = 0;
|
|
25
|
+
const throttled = (messages) => {
|
|
26
|
+
const t = Date.now();
|
|
27
|
+
if (t - lastWrite < 250)
|
|
28
|
+
return;
|
|
29
|
+
lastWrite = t;
|
|
30
|
+
now(messages);
|
|
31
|
+
};
|
|
32
|
+
return { now, throttled };
|
|
33
|
+
}
|
|
34
|
+
/**
|
|
35
|
+
* A `buildTurn` `observeStream` that folds the turn's UI chunks into message
|
|
36
|
+
* snapshots and hands each to `onProgress` as `[...lead, ...assistant]`.
|
|
37
|
+
* `continuing` is the assistant message a resume extends, so the snapshot keeps
|
|
38
|
+
* the parts it already had. Observation must never break the run: errors are
|
|
39
|
+
* swallowed, and the primary stream is drained independently.
|
|
40
|
+
*/
|
|
41
|
+
function progressObserver(lead, onProgress, continuing) {
|
|
42
|
+
return (observed) => {
|
|
43
|
+
void (async () => {
|
|
44
|
+
try {
|
|
45
|
+
const byId = new Map();
|
|
46
|
+
for await (const m of readUIMessageStream({
|
|
47
|
+
stream: observed,
|
|
48
|
+
...(continuing ? { message: structuredClone(continuing) } : {}),
|
|
49
|
+
})) {
|
|
50
|
+
const msg = m;
|
|
51
|
+
byId.set(msg.id, msg);
|
|
52
|
+
onProgress([...lead, ...byId.values()]);
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
catch {
|
|
56
|
+
// ignore — live progress is best-effort
|
|
57
|
+
}
|
|
58
|
+
})();
|
|
59
|
+
};
|
|
60
|
+
}
|
|
9
61
|
/** Ids for assistant response messages — `msg_<random>`, matching storage ids. */
|
|
10
62
|
const generateMessageId = createIdGenerator({ prefix: "msg", separator: "_" });
|
|
11
63
|
/** Sub-thread chat ids start with `sub_` so the host can keep them out of the
|
|
@@ -191,6 +243,15 @@ function declinePendingApprovals(messages) {
|
|
|
191
243
|
* Record decisions onto matching `approval-requested` parts (→ `approval-responded`).
|
|
192
244
|
* Mutates in place and returns the messages that changed (to re-persist).
|
|
193
245
|
*/
|
|
246
|
+
/**
|
|
247
|
+
* Whether an approval belongs to the chat's LAST message — the only one a turn
|
|
248
|
+
* can be paused on. Matches a gated tool part's `approval.id` or a bubbled
|
|
249
|
+
* subagent approval's `data.approvalId`.
|
|
250
|
+
*/
|
|
251
|
+
function approvalIsLast(messages, approvalId) {
|
|
252
|
+
const last = messages.at(-1);
|
|
253
|
+
return (last?.parts ?? []).some((p) => p.approval?.id === approvalId || (p.type === "data-subagent-approval" && p.data?.approvalId === approvalId));
|
|
254
|
+
}
|
|
194
255
|
function applyDecisions(messages, decisions) {
|
|
195
256
|
const byId = new Map(decisions.map((d) => [d.approvalId, d]));
|
|
196
257
|
const changed = [];
|
|
@@ -504,6 +565,23 @@ function fireAndForget(fn) {
|
|
|
504
565
|
*/
|
|
505
566
|
/** Keys whose values are masked when a non-`Error` throw is serialized. */
|
|
506
567
|
const SECRETISH_KEY = /token|secret|password|authorization|cookie|api[-_]?key|credential/i;
|
|
568
|
+
/** The last `error` chunk's text in a UI message stream body (SSE `data:` lines), if any. */
|
|
569
|
+
function lastStreamError(body) {
|
|
570
|
+
let error;
|
|
571
|
+
for (const line of body.split("\n")) {
|
|
572
|
+
if (!line.startsWith("data: "))
|
|
573
|
+
continue;
|
|
574
|
+
try {
|
|
575
|
+
const chunk = JSON.parse(line.slice(6));
|
|
576
|
+
if (chunk.type === "error" && chunk.errorText)
|
|
577
|
+
error = chunk.errorText;
|
|
578
|
+
}
|
|
579
|
+
catch {
|
|
580
|
+
// "[DONE]" and other non-JSON lines
|
|
581
|
+
}
|
|
582
|
+
}
|
|
583
|
+
return error;
|
|
584
|
+
}
|
|
507
585
|
export function clientErrorMessage(error) {
|
|
508
586
|
const message = error instanceof Error ? error.message.trim() : "";
|
|
509
587
|
if (message)
|
|
@@ -554,13 +632,15 @@ export function createRuntime(config) {
|
|
|
554
632
|
"Implement openTurn — see the StorageAdapter contract.");
|
|
555
633
|
}
|
|
556
634
|
/** Resolve an agent's per-request config + a display model id. */
|
|
557
|
-
async function resolveAgentConfig(agentName, principal, request, turnContext
|
|
635
|
+
async function resolveAgentConfig(agentName, principal, request, turnContext,
|
|
636
|
+
/** Passed through to `dynamicAgents.resolve` (see DynamicAgentResolveContext). */
|
|
637
|
+
resolveCtx) {
|
|
558
638
|
const runtimeCtx = await config.context.build({ principal, request, turnContext });
|
|
559
639
|
// Code-declared agents first; otherwise a dynamic (e.g. DB-stored) agent.
|
|
560
640
|
const factory = config.agents[agentName];
|
|
561
641
|
const cfg = factory
|
|
562
642
|
? await factory({ context: runtimeCtx, principal, turnContext })
|
|
563
|
-
: await config.dynamicAgents?.resolve(agentName, principal);
|
|
643
|
+
: await config.dynamicAgents?.resolve(agentName, principal, resolveCtx);
|
|
564
644
|
if (!cfg)
|
|
565
645
|
throw new Error(`Unknown agent: ${agentName}`);
|
|
566
646
|
const modelId = typeof cfg.model === "string"
|
|
@@ -568,15 +648,33 @@ export function createRuntime(config) {
|
|
|
568
648
|
: (cfg.model.modelId ?? "unknown");
|
|
569
649
|
return { cfg, modelId, runtimeCtx };
|
|
570
650
|
}
|
|
651
|
+
/** `buildTurnStream` as an SSE `Response` — what every chat-facing path returns. */
|
|
652
|
+
async function buildTurn(args) {
|
|
653
|
+
return createUIMessageStreamResponse({
|
|
654
|
+
stream: (await buildTurnStream(args)),
|
|
655
|
+
consumeSseStream: consumeStream,
|
|
656
|
+
});
|
|
657
|
+
}
|
|
571
658
|
/**
|
|
572
659
|
* Build (and start) one turn's stream — shared by a fresh chat turn and a
|
|
573
660
|
* resume. With durability on, it heartbeats the lease and aborts if it's
|
|
574
661
|
* lost; onFinish writes are fenced by `lease`. The caller decides whether to
|
|
575
662
|
* hand the stream to a client (handleChat) or drive it to completion
|
|
576
|
-
* server-side (resume).
|
|
663
|
+
* server-side (resume). Returned as the bare chunk stream (see `buildTurn`
|
|
664
|
+
* for the `Response`) so `applyApproval`, which may already be streaming a
|
|
665
|
+
* subagent resume, can merge the continuation into that response.
|
|
577
666
|
*/
|
|
578
|
-
async function
|
|
579
|
-
const { resolvedChatId, principal,
|
|
667
|
+
async function buildTurnStream(args) {
|
|
668
|
+
const { resolvedChatId, principal, run, lease, runtimeCtx } = args;
|
|
669
|
+
// A per-turn model override replaces the agent's own, and the telemetry id
|
|
670
|
+
// with it — a run row that says the agent's model while a different one
|
|
671
|
+
// answered makes every later comparison a lie.
|
|
672
|
+
const cfg = args.model ? { ...args.cfg, model: args.model } : args.cfg;
|
|
673
|
+
const modelId = args.model
|
|
674
|
+
? typeof args.model === "string"
|
|
675
|
+
? args.model
|
|
676
|
+
: (args.model.modelId ?? "unknown")
|
|
677
|
+
: args.modelId;
|
|
580
678
|
// Mutable: the start moments below may inject, replace or rewrite history
|
|
581
679
|
// before the model runs. From here on this is the turn's transcript.
|
|
582
680
|
let uiMessages = args.uiMessages;
|
|
@@ -777,7 +875,8 @@ export function createRuntime(config) {
|
|
|
777
875
|
// append the provider's compiled-index block to the instructions. A
|
|
778
876
|
// provider failure degrades to a turn without memory, never a dead turn.
|
|
779
877
|
let instructions = cfg.instructions;
|
|
780
|
-
|
|
878
|
+
const memoryOn = args.memory ?? true;
|
|
879
|
+
if (memoryOn && cfg.memory && config.memory) {
|
|
781
880
|
const memArgs = { principal, agent: run.agent, memory: cfg.memory };
|
|
782
881
|
try {
|
|
783
882
|
// Resolve both hooks BEFORE mutating the toolset, so a failure in
|
|
@@ -852,30 +951,15 @@ export function createRuntime(config) {
|
|
|
852
951
|
const subChatId = threadId?.trim() || generateSubchatId();
|
|
853
952
|
let onProgress;
|
|
854
953
|
if (writer) {
|
|
855
|
-
const
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
agent,
|
|
862
|
-
threadId: subChatId,
|
|
863
|
-
messages,
|
|
864
|
-
},
|
|
865
|
-
});
|
|
866
|
-
};
|
|
867
|
-
write([]);
|
|
868
|
-
// Each write re-sends the whole snapshot — throttle the re-sends.
|
|
954
|
+
const progress = subagentProgressWriter(writer, {
|
|
955
|
+
toolCallId: options.toolCallId,
|
|
956
|
+
agent,
|
|
957
|
+
threadId: subChatId,
|
|
958
|
+
});
|
|
959
|
+
progress.now([]);
|
|
869
960
|
// The trailing tokens are covered by the tool RESULT (the client
|
|
870
961
|
// switches to the persisted thread once output lands).
|
|
871
|
-
|
|
872
|
-
onProgress = (messages) => {
|
|
873
|
-
const now = Date.now();
|
|
874
|
-
if (now - lastWrite < 250)
|
|
875
|
-
return;
|
|
876
|
-
lastWrite = now;
|
|
877
|
-
write(messages);
|
|
878
|
-
};
|
|
962
|
+
onProgress = progress.throttled;
|
|
879
963
|
}
|
|
880
964
|
// Propagate the PARENT turn's execution mode AS A SET: interactive →
|
|
881
965
|
// the subagent gates its tools (a pending write bubbles up here);
|
|
@@ -890,7 +974,13 @@ export function createRuntime(config) {
|
|
|
890
974
|
depth: depth + 1,
|
|
891
975
|
autoApprove,
|
|
892
976
|
blockGated,
|
|
977
|
+
// Memory off stays off down the tree: an eval turn's delegate must
|
|
978
|
+
// not recall earlier inputs or save new memory either.
|
|
979
|
+
...(memoryOn ? {} : { memory: false }),
|
|
893
980
|
onProgress,
|
|
981
|
+
// Who is delegating — persisted on the sub-thread so every later
|
|
982
|
+
// resolve of it can be shaped per parent (DynamicAgentResolveContext).
|
|
983
|
+
parentAgent: run.agent,
|
|
894
984
|
// Telemetry linkage: this turn is the parent; carry the root chat
|
|
895
985
|
// down so the subagent's trace nests under this conversation.
|
|
896
986
|
parentRunId: run.id,
|
|
@@ -963,18 +1053,36 @@ export function createRuntime(config) {
|
|
|
963
1053
|
// Test runs: stub the approval-gated (mutating) tools so a trial has no
|
|
964
1054
|
// real side effects. Gated set = connection defaults + the agent's policy.
|
|
965
1055
|
if (blockGated) {
|
|
1056
|
+
/**
|
|
1057
|
+
* Does this approval value actually gate the tool?
|
|
1058
|
+
*
|
|
1059
|
+
* NOT a truthiness check. The SDK spells a waiver as the STRING
|
|
1060
|
+
* `"not-applicable"` — which is truthy — so `if (v)` treated every
|
|
1061
|
+
* tool an agent had explicitly waived as gated and stubbed it. On an
|
|
1062
|
+
* agent whose reads are waived to override a connection default, that
|
|
1063
|
+
* is every read tool: the agent could not retrieve anything, and an
|
|
1064
|
+
* eval run measured a model with no access to its own corpus.
|
|
1065
|
+
*
|
|
1066
|
+
* Gated: `true`, `"user-approval"`, an object (approval with a
|
|
1067
|
+
* reason), a function. Waived: `false`, `undefined`,
|
|
1068
|
+
* `"not-applicable"`.
|
|
1069
|
+
*/
|
|
1070
|
+
const gatedBy = (v) => v === true ||
|
|
1071
|
+
v === "user-approval" ||
|
|
1072
|
+
typeof v === "function" ||
|
|
1073
|
+
(typeof v === "object" && v !== null);
|
|
966
1074
|
const gated = new Set();
|
|
967
1075
|
for (const [k, v] of Object.entries(harnessApproval))
|
|
968
|
-
if (v)
|
|
1076
|
+
if (gatedBy(v))
|
|
969
1077
|
gated.add(k);
|
|
970
1078
|
for (const o of opened) {
|
|
971
1079
|
for (const [k, v] of Object.entries(o.toolApproval ?? {}))
|
|
972
|
-
if (v)
|
|
1080
|
+
if (gatedBy(v))
|
|
973
1081
|
gated.add(k);
|
|
974
1082
|
}
|
|
975
1083
|
if (cfg.toolApproval && typeof cfg.toolApproval !== "function") {
|
|
976
1084
|
for (const [k, v] of Object.entries(cfg.toolApproval)) {
|
|
977
|
-
if (v)
|
|
1085
|
+
if (gatedBy(v))
|
|
978
1086
|
gated.add(k);
|
|
979
1087
|
else
|
|
980
1088
|
gated.delete(k);
|
|
@@ -992,12 +1100,19 @@ export function createRuntime(config) {
|
|
|
992
1100
|
if (t && typeof t.execute === "function") {
|
|
993
1101
|
tools[name] = {
|
|
994
1102
|
...t,
|
|
995
|
-
execute: async (input) =>
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
|
|
999
|
-
|
|
1000
|
-
|
|
1103
|
+
execute: async (input) => blockGated === "simulate-success"
|
|
1104
|
+
? {
|
|
1105
|
+
ok: true,
|
|
1106
|
+
__stubbed: true,
|
|
1107
|
+
tool: name,
|
|
1108
|
+
input,
|
|
1109
|
+
}
|
|
1110
|
+
: {
|
|
1111
|
+
__blocked: true,
|
|
1112
|
+
tool: name,
|
|
1113
|
+
input,
|
|
1114
|
+
reason: "approval-gated tool blocked during test run (no real side effects)",
|
|
1115
|
+
},
|
|
1001
1116
|
};
|
|
1002
1117
|
}
|
|
1003
1118
|
}
|
|
@@ -1056,7 +1171,19 @@ export function createRuntime(config) {
|
|
|
1056
1171
|
? ({ steps }) => steps.reduce((n, st) => n + (st.usage?.inputTokens ?? 0) + (st.usage?.outputTokens ?? 0), 0) >=
|
|
1057
1172
|
tokenBudget
|
|
1058
1173
|
: undefined;
|
|
1059
|
-
|
|
1174
|
+
// A delegation that paused on the user's approval ends the turn, exactly as
|
|
1175
|
+
// a gated tool of the agent's own does. A gated call has no result, so the
|
|
1176
|
+
// SDK stops there by itself; `spawn_agent` RETURNS one (`awaiting_approval`),
|
|
1177
|
+
// so without this the loop continues and only the result's note stands
|
|
1178
|
+
// between the model and a second delegation — another bubbled card per
|
|
1179
|
+
// step, up to the step cap. Added only when the agent can delegate, so
|
|
1180
|
+
// other agents' stop conditions stay exactly as configured.
|
|
1181
|
+
const subagentPauseStop = tools.spawn_agent
|
|
1182
|
+
? ({ steps }) => (steps.at(-1)?.toolResults ?? []).some((r) => r.toolName === "spawn_agent" &&
|
|
1183
|
+
r.output?.status === "awaiting_approval")
|
|
1184
|
+
: undefined;
|
|
1185
|
+
const extraStops = [budgetStop, subagentPauseStop].filter((s) => s !== undefined);
|
|
1186
|
+
const baseStopWhen = extraStops.length > 0 ? [stepCap ?? stepCountIs(20), ...extraStops] : stepCap;
|
|
1060
1187
|
// --- step boundaries for the seam -------------------------------------------
|
|
1061
1188
|
// A step ends twice from here: on the model side (`agent.stream`'s
|
|
1062
1189
|
// onStepEnd — usage, finish reason, tool calls) and on the UI side (the
|
|
@@ -1130,7 +1257,7 @@ export function createRuntime(config) {
|
|
|
1130
1257
|
: config.resolvePromptCaching
|
|
1131
1258
|
? config.resolvePromptCaching(modelId, {
|
|
1132
1259
|
agent: run.agent,
|
|
1133
|
-
memoryScoped: !!(cfg.memory && config.memory),
|
|
1260
|
+
memoryScoped: !!(memoryOn && cfg.memory && config.memory),
|
|
1134
1261
|
principal,
|
|
1135
1262
|
})
|
|
1136
1263
|
: // No hook → caching is ON by default for first-party Anthropic /
|
|
@@ -1704,10 +1831,7 @@ export function createRuntime(config) {
|
|
|
1704
1831
|
out = main;
|
|
1705
1832
|
args.observeStream(observed);
|
|
1706
1833
|
}
|
|
1707
|
-
return
|
|
1708
|
-
stream: out,
|
|
1709
|
-
consumeSseStream: consumeStream,
|
|
1710
|
-
});
|
|
1834
|
+
return out;
|
|
1711
1835
|
}
|
|
1712
1836
|
async function reapTick(opts) {
|
|
1713
1837
|
const reaped = await storage.reapExpiredRuns(opts?.now ?? Date.now(), {
|
|
@@ -1860,7 +1984,10 @@ export function createRuntime(config) {
|
|
|
1860
1984
|
});
|
|
1861
1985
|
}
|
|
1862
1986
|
const resolvedChatId = chat.id;
|
|
1863
|
-
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(agentName, principal, request, turnContext
|
|
1987
|
+
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(agentName, principal, request, turnContext,
|
|
1988
|
+
// A host may continue a subagent thread through the chat surface; keep
|
|
1989
|
+
// resolving it the way its spawner would.
|
|
1990
|
+
{ parentAgent: chat.parentAgent });
|
|
1864
1991
|
const prior = await storage.loadMessages(principal, resolvedChatId);
|
|
1865
1992
|
// A new message while the last turn is parked on an approval is itself the
|
|
1866
1993
|
// decision: the user declined to answer and wants to steer elsewhere. Close
|
|
@@ -2028,7 +2155,12 @@ export function createRuntime(config) {
|
|
|
2028
2155
|
// reconcilable even if its (dynamic, DB-stored) agent was since deleted.
|
|
2029
2156
|
// Before this ordering that throw also aborted `sweep`'s loop, so one such
|
|
2030
2157
|
// run held up recovery for every other run until its attempts ran out.
|
|
2031
|
-
|
|
2158
|
+
// A recovered subagent run must resolve exactly as its spawner shaped it —
|
|
2159
|
+
// crash recovery and `sweep` both land here.
|
|
2160
|
+
const chat = await storage.getChat(principal, run.chatId);
|
|
2161
|
+
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(run.agent, principal, undefined, undefined, {
|
|
2162
|
+
parentAgent: chat?.parentAgent,
|
|
2163
|
+
});
|
|
2032
2164
|
// A resume was admitted when the turn first ran — never refuse it (the
|
|
2033
2165
|
// half-done work would strand), but its tokens still settle against the
|
|
2034
2166
|
// same counters, and the mid-turn stop still applies.
|
|
@@ -2133,57 +2265,115 @@ export function createRuntime(config) {
|
|
|
2133
2265
|
// Merge the decisions onto the persisted assistant message, then re-persist
|
|
2134
2266
|
// only what changed. The agent continues from this updated history.
|
|
2135
2267
|
const messages = await storage.loadMessages(principal, chatId);
|
|
2268
|
+
// A decision only means something for the message the turn paused on —
|
|
2269
|
+
// the LAST one. An approval from further back (an old tab, an old Slack
|
|
2270
|
+
// card, a late retry after the user moved on) was already settled: the
|
|
2271
|
+
// next message declined it. Continuing anyway would run the agent again
|
|
2272
|
+
// and append an answer nobody asked for. A real retry (a crash or a busy
|
|
2273
|
+
// chat after the decision was saved) still finds its approval last.
|
|
2274
|
+
const live = decisions.filter((d) => approvalIsLast(messages, d.approvalId));
|
|
2275
|
+
if (live.length === 0)
|
|
2276
|
+
return new Response(null, { status: 200 });
|
|
2277
|
+
decisions = live;
|
|
2136
2278
|
const changed = applyDecisions(messages, decisions);
|
|
2137
2279
|
if (changed.length > 0)
|
|
2138
2280
|
await storage.appendMessages(chatId, changed);
|
|
2281
|
+
// Continue this chat from the decided history.
|
|
2282
|
+
const continueTurn = async () => {
|
|
2283
|
+
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(chat.agent, principal, request, turnContext, { parentAgent: chat.parentAgent });
|
|
2284
|
+
const admitted = await admitContinuation(principal, chat.agent, "decision");
|
|
2285
|
+
const { run, lease } = await openRun(principal, chatId, chat.agent, cfg.configVersion);
|
|
2286
|
+
return buildTurnStream({
|
|
2287
|
+
resolvedChatId: chatId,
|
|
2288
|
+
principal,
|
|
2289
|
+
modelId,
|
|
2290
|
+
cfg,
|
|
2291
|
+
run,
|
|
2292
|
+
lease,
|
|
2293
|
+
// Resume: strip the paused turn's trailing text so the model call ends on
|
|
2294
|
+
// the tool result (a user turn), not an assistant prefill.
|
|
2295
|
+
uiMessages: trimTrailingAssistantPrefill(messages),
|
|
2296
|
+
runtimeCtx,
|
|
2297
|
+
// A human just decided something, so a human is present — even if this
|
|
2298
|
+
// turn parks again on the NEXT gate (see TurnTrigger).
|
|
2299
|
+
trigger: "decision",
|
|
2300
|
+
admitted,
|
|
2301
|
+
});
|
|
2302
|
+
};
|
|
2139
2303
|
// Route any just-decided SUBAGENT approvals (bubbled up via spawn_agent):
|
|
2140
2304
|
// resume the sub thread with the decision, then fold its answer back into
|
|
2141
2305
|
// the parent's spawn_agent result so this turn continues with the digest.
|
|
2306
|
+
//
|
|
2307
|
+
// All of it streams in THIS response, which is the only channel back to a
|
|
2308
|
+
// client already showing the paused message. The resume can run for
|
|
2309
|
+
// minutes, pause on a SECOND gate, or fold an answer the client has never
|
|
2310
|
+
// seen — answered with an empty body instead, the user saw their approval
|
|
2311
|
+
// land and then nothing: no progress, and no card for the next gate, while
|
|
2312
|
+
// the chat sat awaiting a decision it never offered.
|
|
2142
2313
|
const subApprovals = collectDecidedSubagentApprovals(messages);
|
|
2143
2314
|
if (subApprovals.length > 0) {
|
|
2144
|
-
|
|
2145
|
-
|
|
2146
|
-
|
|
2147
|
-
|
|
2148
|
-
|
|
2149
|
-
|
|
2150
|
-
|
|
2151
|
-
|
|
2152
|
-
|
|
2153
|
-
|
|
2154
|
-
|
|
2155
|
-
|
|
2156
|
-
|
|
2157
|
-
|
|
2158
|
-
|
|
2159
|
-
|
|
2160
|
-
|
|
2161
|
-
|
|
2162
|
-
|
|
2163
|
-
|
|
2164
|
-
|
|
2165
|
-
|
|
2166
|
-
|
|
2167
|
-
|
|
2168
|
-
|
|
2169
|
-
|
|
2170
|
-
|
|
2171
|
-
|
|
2172
|
-
|
|
2173
|
-
|
|
2315
|
+
const stream = createUIMessageStream({
|
|
2316
|
+
execute: async ({ writer }) => {
|
|
2317
|
+
for (const sa of subApprovals) {
|
|
2318
|
+
const resumed = await resumeSubagentApproval(principal, sa.subChatId, sa.subApprovalIds.map((id) => ({
|
|
2319
|
+
approvalId: id,
|
|
2320
|
+
approved: sa.approved,
|
|
2321
|
+
reason: sa.reason,
|
|
2322
|
+
})), { parentRunId: sa.parentRunId, rootChatId: sa.rootChatId }, { writer, spawnToolCallId: sa.spawnToolCallId });
|
|
2323
|
+
let output;
|
|
2324
|
+
if (resumed.pending?.length) {
|
|
2325
|
+
// The resumed sub paused on ANOTHER gated tool — re-bubble it as a
|
|
2326
|
+
// fresh data-subagent-approval part on the same parent message. The
|
|
2327
|
+
// undecided part keeps this chat awaiting_input (hasPendingApproval
|
|
2328
|
+
// below), and the next decision routes through here again, so
|
|
2329
|
+
// bubbling recurses per round.
|
|
2330
|
+
const data = {
|
|
2331
|
+
approvalId: generateSubApprovalId(),
|
|
2332
|
+
subChatId: sa.subChatId,
|
|
2333
|
+
subApprovalIds: resumed.pending.map((p) => p.approvalId),
|
|
2334
|
+
spawnToolCallId: sa.spawnToolCallId,
|
|
2335
|
+
summaries: resumed.pending.map((p) => p.summary),
|
|
2336
|
+
// Carry the linkage forward so each further resume stays nested.
|
|
2337
|
+
parentRunId: sa.parentRunId,
|
|
2338
|
+
rootChatId: sa.rootChatId,
|
|
2339
|
+
};
|
|
2340
|
+
appendSubagentApprovalPart(messages, sa.approvalId, data);
|
|
2341
|
+
// Same id as the stored part, so the client reconciles it into the
|
|
2342
|
+
// paused message exactly as a reload would show it.
|
|
2343
|
+
writer.write({ type: "data-subagent-approval", id: data.approvalId, data });
|
|
2344
|
+
output = {
|
|
2345
|
+
status: "awaiting_approval",
|
|
2346
|
+
threadId: sa.subChatId,
|
|
2347
|
+
pending: resumed.pending.map((p) => p.summary),
|
|
2348
|
+
note: "Awaiting the user's approval — do not retry; stop here, the user will decide and you'll continue automatically.",
|
|
2349
|
+
};
|
|
2350
|
+
}
|
|
2351
|
+
else {
|
|
2352
|
+
output = { threadId: sa.subChatId, answer: resumed.answer };
|
|
2353
|
+
}
|
|
2354
|
+
if (sa.spawnToolCallId) {
|
|
2355
|
+
overwriteToolResult(messages, sa.spawnToolCallId, output);
|
|
2356
|
+
writer.write({ type: "tool-output-available", toolCallId: sa.spawnToolCallId, output });
|
|
2357
|
+
}
|
|
2358
|
+
markSubagentApprovalResolved(messages, sa.approvalId);
|
|
2174
2359
|
}
|
|
2175
|
-
|
|
2176
|
-
|
|
2177
|
-
|
|
2178
|
-
|
|
2179
|
-
|
|
2180
|
-
|
|
2181
|
-
|
|
2182
|
-
|
|
2183
|
-
|
|
2184
|
-
|
|
2185
|
-
|
|
2186
|
-
|
|
2360
|
+
// Persist the folded spawn_agent result (or re-bubbled approval part)
|
|
2361
|
+
// + resolved markers.
|
|
2362
|
+
await storage.appendMessages(chatId, messages);
|
|
2363
|
+
if (hasPendingApproval(messages)) {
|
|
2364
|
+
writer.write({ type: "finish" });
|
|
2365
|
+
return;
|
|
2366
|
+
}
|
|
2367
|
+
writer.merge(await continueTurn());
|
|
2368
|
+
},
|
|
2369
|
+
onError: (error) => clientErrorMessage(error),
|
|
2370
|
+
});
|
|
2371
|
+
return createUIMessageStreamResponse({
|
|
2372
|
+
stream: stream,
|
|
2373
|
+
// Drain server-side even if the client leaves: the resume must finish
|
|
2374
|
+
// and persist either way, as every turn does (see buildTurnStream).
|
|
2375
|
+
consumeSseStream: consumeStream,
|
|
2376
|
+
});
|
|
2187
2377
|
}
|
|
2188
2378
|
// If the step had several gated calls and some are still undecided, DON'T
|
|
2189
2379
|
// re-run yet — a tool call without a result makes convertToModelMessages
|
|
@@ -2192,24 +2382,9 @@ export function createRuntime(config) {
|
|
|
2192
2382
|
if (hasPendingApproval(messages)) {
|
|
2193
2383
|
return new Response(null, { status: 200 });
|
|
2194
2384
|
}
|
|
2195
|
-
|
|
2196
|
-
|
|
2197
|
-
|
|
2198
|
-
return buildTurn({
|
|
2199
|
-
resolvedChatId: chatId,
|
|
2200
|
-
principal,
|
|
2201
|
-
modelId,
|
|
2202
|
-
cfg,
|
|
2203
|
-
run,
|
|
2204
|
-
lease,
|
|
2205
|
-
// Resume: strip the paused turn's trailing text so the model call ends on
|
|
2206
|
-
// the tool result (a user turn), not an assistant prefill.
|
|
2207
|
-
uiMessages: trimTrailingAssistantPrefill(messages),
|
|
2208
|
-
runtimeCtx,
|
|
2209
|
-
// A human just decided something, so a human is present — even if this
|
|
2210
|
-
// turn parks again on the NEXT gate (see TurnTrigger).
|
|
2211
|
-
trigger: "decision",
|
|
2212
|
-
admitted,
|
|
2385
|
+
return createUIMessageStreamResponse({
|
|
2386
|
+
stream: (await continueTurn()),
|
|
2387
|
+
consumeSseStream: consumeStream,
|
|
2213
2388
|
});
|
|
2214
2389
|
}
|
|
2215
2390
|
/**
|
|
@@ -2240,7 +2415,7 @@ export function createRuntime(config) {
|
|
|
2240
2415
|
await storage.appendMessages(chatId, changed);
|
|
2241
2416
|
return new Response(null, { status: 200 });
|
|
2242
2417
|
}
|
|
2243
|
-
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(chat.agent, principal, request, turnContext);
|
|
2418
|
+
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(chat.agent, principal, request, turnContext, { parentAgent: chat.parentAgent });
|
|
2244
2419
|
const admitted = await admitContinuation(principal, chat.agent, "decision");
|
|
2245
2420
|
// The result is persisted WITH the run (atomically where the adapter can):
|
|
2246
2421
|
// a busy chat refuses before anything is recorded, so the retry finds the
|
|
@@ -2286,7 +2461,7 @@ export function createRuntime(config) {
|
|
|
2286
2461
|
if (hasPendingApproval(live) || hasPendingToolResult([live[live.length - 1]])) {
|
|
2287
2462
|
throw new PendingApprovalError("This chat is awaiting a decision — resolve the pending request before compacting.");
|
|
2288
2463
|
}
|
|
2289
|
-
const { cfg, modelId } = await resolveAgentConfig(chat.agent, principal, request, turnContext);
|
|
2464
|
+
const { cfg, modelId } = await resolveAgentConfig(chat.agent, principal, request, turnContext, { parentAgent: chat.parentAgent });
|
|
2290
2465
|
// Tool definitions only affect how a recorded tool RESULT is rendered for the
|
|
2291
2466
|
// model (`toModelOutput`); an unknown tool falls back to its raw JSON rather
|
|
2292
2467
|
// than failing. So take every definition available without I/O — the agent's
|
|
@@ -2419,11 +2594,14 @@ export function createRuntime(config) {
|
|
|
2419
2594
|
return chat ? run : null;
|
|
2420
2595
|
}
|
|
2421
2596
|
/** A fresh user message from plain text (for cron-fired turns). */
|
|
2422
|
-
function userMessageOf(text) {
|
|
2597
|
+
function userMessageOf(text, files) {
|
|
2423
2598
|
return {
|
|
2424
2599
|
id: generateMessageId(),
|
|
2425
2600
|
role: "user",
|
|
2426
|
-
parts: [
|
|
2601
|
+
parts: [
|
|
2602
|
+
...(files ?? []).map((f) => ({ type: "file", url: f.url, mediaType: f.mediaType, ...(f.filename ? { filename: f.filename } : {}) })),
|
|
2603
|
+
{ type: "text", text },
|
|
2604
|
+
],
|
|
2427
2605
|
metadata: { visibility: "user", createdAt: Date.now() },
|
|
2428
2606
|
};
|
|
2429
2607
|
}
|
|
@@ -2485,8 +2663,15 @@ export function createRuntime(config) {
|
|
|
2485
2663
|
// Runtime-created thread: stamp it internal so a host filtering its
|
|
2486
2664
|
// chat list on kind never shows it (see ChatRecord.kind).
|
|
2487
2665
|
kind: "internal",
|
|
2666
|
+
...(o.parentAgent ? { parentAgent: o.parentAgent } : {}),
|
|
2488
2667
|
});
|
|
2489
2668
|
}
|
|
2669
|
+
// A spawn may only continue a thread its own agent delegated. Otherwise
|
|
2670
|
+
// agent B could pass agent A's sub-thread id and have the child resolved
|
|
2671
|
+
// under A — borrowing whatever a host grants A's subagents.
|
|
2672
|
+
if (o.parentAgent && chat.parentAgent && chat.parentAgent !== o.parentAgent) {
|
|
2673
|
+
throw new Error(`Thread ${chatId} was delegated by another agent; omit threadId to start a new one.`);
|
|
2674
|
+
}
|
|
2490
2675
|
const prior = await storage.loadMessages(o.principal, chatId);
|
|
2491
2676
|
// Same rule as handleChat: a new message while the thread is parked on an
|
|
2492
2677
|
// approval IS the decision — close the stale gate as declined (persisted)
|
|
@@ -2503,7 +2688,14 @@ export function createRuntime(config) {
|
|
|
2503
2688
|
// the run first, persist the message only once it is ours to run.
|
|
2504
2689
|
const admitted = await admitTurn(o.principal, o.agent, o.trigger);
|
|
2505
2690
|
const { cfg, modelId, runtimeCtx, run, lease } = await withReservation(admitted, async () => {
|
|
2506
|
-
const resolved = await resolveAgentConfig(o.agent, o.principal, undefined, o.turnContext
|
|
2691
|
+
const resolved = await resolveAgentConfig(o.agent, o.principal, undefined, o.turnContext, {
|
|
2692
|
+
// A continued thread keeps the parent it was created under (a mismatch
|
|
2693
|
+
// was refused above). Without a stored one — an adapter that doesn't
|
|
2694
|
+
// persist the column — the spawner of THIS turn is the parent: a host
|
|
2695
|
+
// that restricts a subagent by its parent must never see none.
|
|
2696
|
+
parentAgent: chat.parentAgent ?? o.parentAgent,
|
|
2697
|
+
autoApprove: o.autoApprove,
|
|
2698
|
+
});
|
|
2507
2699
|
const opened = await openTurn(o.principal, chatId, o.agent, [o.message], resolved.cfg.configVersion);
|
|
2508
2700
|
return { ...resolved, ...opened };
|
|
2509
2701
|
});
|
|
@@ -2520,35 +2712,20 @@ export function createRuntime(config) {
|
|
|
2520
2712
|
admitted,
|
|
2521
2713
|
autoApprove: o.autoApprove,
|
|
2522
2714
|
blockGated: o.blockGated,
|
|
2715
|
+
memory: o.memory,
|
|
2716
|
+
model: o.model,
|
|
2523
2717
|
promptCaching: o.promptCaching,
|
|
2524
2718
|
depth: o.depth ?? 0,
|
|
2525
2719
|
parentRunId: o.parentRunId,
|
|
2526
2720
|
rootChatId: o.rootChatId,
|
|
2527
2721
|
trigger: o.trigger,
|
|
2528
2722
|
// Forward progressive UI-message snapshots to the caller (spawn_agent
|
|
2529
|
-
// streams them into the parent turn).
|
|
2530
|
-
|
|
2531
|
-
observeStream: onProgress
|
|
2532
|
-
? (observed) => {
|
|
2533
|
-
void (async () => {
|
|
2534
|
-
try {
|
|
2535
|
-
const byId = new Map();
|
|
2536
|
-
for await (const m of readUIMessageStream({
|
|
2537
|
-
stream: observed,
|
|
2538
|
-
})) {
|
|
2539
|
-
const msg = m;
|
|
2540
|
-
byId.set(msg.id, msg);
|
|
2541
|
-
onProgress([o.message, ...byId.values()]);
|
|
2542
|
-
}
|
|
2543
|
-
}
|
|
2544
|
-
catch {
|
|
2545
|
-
// ignore — live progress is best-effort
|
|
2546
|
-
}
|
|
2547
|
-
})();
|
|
2548
|
-
}
|
|
2549
|
-
: undefined,
|
|
2723
|
+
// streams them into the parent turn).
|
|
2724
|
+
observeStream: onProgress ? progressObserver([o.message], onProgress) : undefined,
|
|
2550
2725
|
});
|
|
2551
|
-
|
|
2726
|
+
// The turn's own failure travels as an `error` chunk — the stream, not a
|
|
2727
|
+
// throw. Keep its text: without it a failed turn reads as a silent one.
|
|
2728
|
+
const error = lastStreamError(await res.text());
|
|
2552
2729
|
const after = await storage.loadMessages(o.principal, chatId);
|
|
2553
2730
|
// The verdict is about THIS turn, so read the FINAL message only (the same
|
|
2554
2731
|
// rule hasPendingApproval applies): an undecided gate in an older message
|
|
@@ -2563,6 +2740,7 @@ export function createRuntime(config) {
|
|
|
2563
2740
|
answer: pending.length > 0 ? "" : lastAssistantText(after),
|
|
2564
2741
|
parked,
|
|
2565
2742
|
...(pending.length > 0 ? { pending } : {}),
|
|
2743
|
+
...(error ? { error } : {}),
|
|
2566
2744
|
};
|
|
2567
2745
|
}
|
|
2568
2746
|
/**
|
|
@@ -2577,7 +2755,13 @@ export function createRuntime(config) {
|
|
|
2577
2755
|
*/
|
|
2578
2756
|
async function resumeSubagentApproval(principal, subChatId, decisions,
|
|
2579
2757
|
/** Telemetry linkage of the original sub-turn, so the trace stays nested. */
|
|
2580
|
-
linkage
|
|
2758
|
+
linkage,
|
|
2759
|
+
/**
|
|
2760
|
+
* Stream the resumed sub-turn into the parent's response as the same
|
|
2761
|
+
* `data-subagent-progress` part `spawn_agent` wrote, so the delegation the
|
|
2762
|
+
* user just approved visibly runs again instead of sitting silent.
|
|
2763
|
+
*/
|
|
2764
|
+
live) {
|
|
2581
2765
|
const subChat = await storage.getChat(principal, subChatId);
|
|
2582
2766
|
if (!subChat)
|
|
2583
2767
|
return { answer: "" };
|
|
@@ -2588,9 +2772,29 @@ export function createRuntime(config) {
|
|
|
2588
2772
|
// Another gated call in the same sub step still undecided → can't finish.
|
|
2589
2773
|
if (hasPendingApproval(subMessages))
|
|
2590
2774
|
return { answer: "" };
|
|
2591
|
-
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(subChat.agent, principal
|
|
2775
|
+
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(subChat.agent, principal, undefined, undefined,
|
|
2776
|
+
// Same shaping as the first sub-turn: the parent is read off the persisted
|
|
2777
|
+
// sub-thread, and this path is only ever reached by a human decision.
|
|
2778
|
+
{ parentAgent: subChat.parentAgent, autoApprove: false });
|
|
2592
2779
|
const admitted = await admitContinuation(principal, subChat.agent, "decision");
|
|
2593
2780
|
const { run, lease } = await openRun(principal, subChatId, subChat.agent, cfg.configVersion);
|
|
2781
|
+
const resumeFrom = trimTrailingAssistantPrefill(subMessages);
|
|
2782
|
+
let observeStream;
|
|
2783
|
+
if (live) {
|
|
2784
|
+
const progress = subagentProgressWriter(live.writer, {
|
|
2785
|
+
toolCallId: live.spawnToolCallId,
|
|
2786
|
+
agent: subChat.agent,
|
|
2787
|
+
threadId: subChatId,
|
|
2788
|
+
});
|
|
2789
|
+
// The exchange being resumed: its prompt and the paused assistant
|
|
2790
|
+
// message, which the resumed turn extends in place.
|
|
2791
|
+
const last = resumeFrom[resumeFrom.length - 1];
|
|
2792
|
+
const continuing = last?.role === "assistant" ? last : undefined;
|
|
2793
|
+
const prompt = resumeFrom.findLast((m) => m.role === "user");
|
|
2794
|
+
const lead = prompt ? [prompt] : [];
|
|
2795
|
+
progress.now(continuing ? [...lead, continuing] : lead);
|
|
2796
|
+
observeStream = progressObserver(lead, progress.throttled, continuing);
|
|
2797
|
+
}
|
|
2594
2798
|
const res = await buildTurn({
|
|
2595
2799
|
resolvedChatId: subChatId,
|
|
2596
2800
|
principal,
|
|
@@ -2602,7 +2806,7 @@ export function createRuntime(config) {
|
|
|
2602
2806
|
// Resume: same prefill guard as applyApproval — the paused sub turn may
|
|
2603
2807
|
// have narrated after its gated call, and that text must not reach the
|
|
2604
2808
|
// model as an assistant prefill.
|
|
2605
|
-
uiMessages:
|
|
2809
|
+
uiMessages: resumeFrom,
|
|
2606
2810
|
runtimeCtx,
|
|
2607
2811
|
autoApprove: false, // keep gating any further writes in the sub
|
|
2608
2812
|
depth: 1,
|
|
@@ -2611,6 +2815,7 @@ export function createRuntime(config) {
|
|
|
2611
2815
|
rootChatId: linkage?.rootChatId,
|
|
2612
2816
|
// Reached only by a human deciding the bubbled-up approval.
|
|
2613
2817
|
trigger: "decision",
|
|
2818
|
+
observeStream,
|
|
2614
2819
|
});
|
|
2615
2820
|
await res.text();
|
|
2616
2821
|
const after = await storage.loadMessages(principal, subChatId);
|
|
@@ -2825,24 +3030,26 @@ export function createRuntime(config) {
|
|
|
2825
3030
|
const self = {
|
|
2826
3031
|
listAgents: agentInfos,
|
|
2827
3032
|
handleChat,
|
|
2828
|
-
runAgent: async ({ principal, agent, prompt, threadId, blockGated = true, turnContext, promptCaching,
|
|
3033
|
+
runAgent: async ({ principal, agent, prompt, files, threadId, blockGated = true, memory, model, turnContext, promptCaching,
|
|
2829
3034
|
// No client stream: assume unattended unless the host says otherwise, so
|
|
2830
3035
|
// a park here can't be silently filtered out as "someone's watching".
|
|
2831
3036
|
trigger = "programmatic", }) => {
|
|
2832
3037
|
const r = await runToCompletion({
|
|
2833
3038
|
principal,
|
|
2834
3039
|
agent,
|
|
2835
|
-
message: userMessageOf(prompt),
|
|
3040
|
+
message: userMessageOf(prompt, files),
|
|
2836
3041
|
chatId: threadId,
|
|
2837
3042
|
depth: 1,
|
|
2838
3043
|
// Autonomous: no human to authorize (blockGated stubs the gated tools).
|
|
2839
3044
|
autoApprove: true,
|
|
2840
3045
|
blockGated,
|
|
3046
|
+
memory,
|
|
3047
|
+
model,
|
|
2841
3048
|
turnContext,
|
|
2842
3049
|
promptCaching,
|
|
2843
3050
|
trigger,
|
|
2844
3051
|
});
|
|
2845
|
-
return { threadId: r.chatId, answer: r.answer };
|
|
3052
|
+
return { threadId: r.chatId, answer: r.answer, ...(r.error ? { error: r.error } : {}) };
|
|
2846
3053
|
},
|
|
2847
3054
|
compactChat,
|
|
2848
3055
|
loadHistory,
|