@intentface/latch-core 0.12.0 → 0.13.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/runtime.js CHANGED
@@ -6,6 +6,58 @@ import { computeCost } from "./pricing.js";
6
6
  import { defaultPromptCachingPlan, markLastFunctionTool, mergeProviderOptions, toolsetHash, } from "./prompt-caching.js";
7
7
  import { LimitExceededError, exceededPolicy, remainingTokens, windowRef, } from "./limits.js";
8
8
  import { compose, fireClientToolResults, insertStepParts, isNewSession, lastUserTextOf, renderTranscript, replaceLastUserText, wrapTools, } from "./extensions/index.js";
9
+ /**
10
+ * The `data-subagent-progress` writer for one delegation: `now` sends a
11
+ * snapshot immediately, `throttled` at most every 250ms (each write re-sends
12
+ * the whole snapshot). Shared by `spawn_agent` and the approval resume so both
13
+ * write the SAME part id — a resumed delegation updates the one the client is
14
+ * already showing instead of adding a second.
15
+ */
16
+ function subagentProgressWriter(writer, ids) {
17
+ const now = (messages) => {
18
+ writer.write({
19
+ type: "data-subagent-progress",
20
+ id: `subagent_${ids.toolCallId ?? ids.threadId}`,
21
+ data: { ...ids, messages },
22
+ });
23
+ };
24
+ let lastWrite = 0;
25
+ const throttled = (messages) => {
26
+ const t = Date.now();
27
+ if (t - lastWrite < 250)
28
+ return;
29
+ lastWrite = t;
30
+ now(messages);
31
+ };
32
+ return { now, throttled };
33
+ }
34
+ /**
35
+ * A `buildTurn` `observeStream` that folds the turn's UI chunks into message
36
+ * snapshots and hands each to `onProgress` as `[...lead, ...assistant]`.
37
+ * `continuing` is the assistant message a resume extends, so the snapshot keeps
38
+ * the parts it already had. Observation must never break the run: errors are
39
+ * swallowed, and the primary stream is drained independently.
40
+ */
41
+ function progressObserver(lead, onProgress, continuing) {
42
+ return (observed) => {
43
+ void (async () => {
44
+ try {
45
+ const byId = new Map();
46
+ for await (const m of readUIMessageStream({
47
+ stream: observed,
48
+ ...(continuing ? { message: structuredClone(continuing) } : {}),
49
+ })) {
50
+ const msg = m;
51
+ byId.set(msg.id, msg);
52
+ onProgress([...lead, ...byId.values()]);
53
+ }
54
+ }
55
+ catch {
56
+ // ignore — live progress is best-effort
57
+ }
58
+ })();
59
+ };
60
+ }
9
61
  /** Ids for assistant response messages — `msg_<random>`, matching storage ids. */
10
62
  const generateMessageId = createIdGenerator({ prefix: "msg", separator: "_" });
11
63
  /** Sub-thread chat ids start with `sub_` so the host can keep them out of the
@@ -191,6 +243,15 @@ function declinePendingApprovals(messages) {
191
243
  * Record decisions onto matching `approval-requested` parts (→ `approval-responded`).
192
244
  * Mutates in place and returns the messages that changed (to re-persist).
193
245
  */
246
+ /**
247
+ * Whether an approval belongs to the chat's LAST message — the only one a turn
248
+ * can be paused on. Matches a gated tool part's `approval.id` or a bubbled
249
+ * subagent approval's `data.approvalId`.
250
+ */
251
+ function approvalIsLast(messages, approvalId) {
252
+ const last = messages.at(-1);
253
+ return (last?.parts ?? []).some((p) => p.approval?.id === approvalId || (p.type === "data-subagent-approval" && p.data?.approvalId === approvalId));
254
+ }
194
255
  function applyDecisions(messages, decisions) {
195
256
  const byId = new Map(decisions.map((d) => [d.approvalId, d]));
196
257
  const changed = [];
@@ -504,6 +565,23 @@ function fireAndForget(fn) {
504
565
  */
505
566
  /** Keys whose values are masked when a non-`Error` throw is serialized. */
506
567
  const SECRETISH_KEY = /token|secret|password|authorization|cookie|api[-_]?key|credential/i;
568
+ /** The last `error` chunk's text in a UI message stream body (SSE `data:` lines), if any. */
569
+ function lastStreamError(body) {
570
+ let error;
571
+ for (const line of body.split("\n")) {
572
+ if (!line.startsWith("data: "))
573
+ continue;
574
+ try {
575
+ const chunk = JSON.parse(line.slice(6));
576
+ if (chunk.type === "error" && chunk.errorText)
577
+ error = chunk.errorText;
578
+ }
579
+ catch {
580
+ // "[DONE]" and other non-JSON lines
581
+ }
582
+ }
583
+ return error;
584
+ }
507
585
  export function clientErrorMessage(error) {
508
586
  const message = error instanceof Error ? error.message.trim() : "";
509
587
  if (message)
@@ -554,13 +632,15 @@ export function createRuntime(config) {
554
632
  "Implement openTurn — see the StorageAdapter contract.");
555
633
  }
556
634
  /** Resolve an agent's per-request config + a display model id. */
557
- async function resolveAgentConfig(agentName, principal, request, turnContext) {
635
+ async function resolveAgentConfig(agentName, principal, request, turnContext,
636
+ /** Passed through to `dynamicAgents.resolve` (see DynamicAgentResolveContext). */
637
+ resolveCtx) {
558
638
  const runtimeCtx = await config.context.build({ principal, request, turnContext });
559
639
  // Code-declared agents first; otherwise a dynamic (e.g. DB-stored) agent.
560
640
  const factory = config.agents[agentName];
561
641
  const cfg = factory
562
642
  ? await factory({ context: runtimeCtx, principal, turnContext })
563
- : await config.dynamicAgents?.resolve(agentName, principal);
643
+ : await config.dynamicAgents?.resolve(agentName, principal, resolveCtx);
564
644
  if (!cfg)
565
645
  throw new Error(`Unknown agent: ${agentName}`);
566
646
  const modelId = typeof cfg.model === "string"
@@ -568,15 +648,33 @@ export function createRuntime(config) {
568
648
  : (cfg.model.modelId ?? "unknown");
569
649
  return { cfg, modelId, runtimeCtx };
570
650
  }
651
+ /** `buildTurnStream` as an SSE `Response` — what every chat-facing path returns. */
652
+ async function buildTurn(args) {
653
+ return createUIMessageStreamResponse({
654
+ stream: (await buildTurnStream(args)),
655
+ consumeSseStream: consumeStream,
656
+ });
657
+ }
571
658
  /**
572
659
  * Build (and start) one turn's stream — shared by a fresh chat turn and a
573
660
  * resume. With durability on, it heartbeats the lease and aborts if it's
574
661
  * lost; onFinish writes are fenced by `lease`. The caller decides whether to
575
662
  * hand the stream to a client (handleChat) or drive it to completion
576
- * server-side (resume).
663
+ * server-side (resume). Returned as the bare chunk stream (see `buildTurn`
664
+ * for the `Response`) so `applyApproval`, which may already be streaming a
665
+ * subagent resume, can merge the continuation into that response.
577
666
  */
578
- async function buildTurn(args) {
579
- const { resolvedChatId, principal, modelId, cfg, run, lease, runtimeCtx } = args;
667
+ async function buildTurnStream(args) {
668
+ const { resolvedChatId, principal, run, lease, runtimeCtx } = args;
669
+ // A per-turn model override replaces the agent's own, and the telemetry id
670
+ // with it — a run row that says the agent's model while a different one
671
+ // answered makes every later comparison a lie.
672
+ const cfg = args.model ? { ...args.cfg, model: args.model } : args.cfg;
673
+ const modelId = args.model
674
+ ? typeof args.model === "string"
675
+ ? args.model
676
+ : (args.model.modelId ?? "unknown")
677
+ : args.modelId;
580
678
  // Mutable: the start moments below may inject, replace or rewrite history
581
679
  // before the model runs. From here on this is the turn's transcript.
582
680
  let uiMessages = args.uiMessages;
@@ -777,7 +875,8 @@ export function createRuntime(config) {
777
875
  // append the provider's compiled-index block to the instructions. A
778
876
  // provider failure degrades to a turn without memory, never a dead turn.
779
877
  let instructions = cfg.instructions;
780
- if (cfg.memory && config.memory) {
878
+ const memoryOn = args.memory ?? true;
879
+ if (memoryOn && cfg.memory && config.memory) {
781
880
  const memArgs = { principal, agent: run.agent, memory: cfg.memory };
782
881
  try {
783
882
  // Resolve both hooks BEFORE mutating the toolset, so a failure in
@@ -852,30 +951,15 @@ export function createRuntime(config) {
852
951
  const subChatId = threadId?.trim() || generateSubchatId();
853
952
  let onProgress;
854
953
  if (writer) {
855
- const write = (messages) => {
856
- writer.write({
857
- type: "data-subagent-progress",
858
- id: `subagent_${options.toolCallId ?? subChatId}`,
859
- data: {
860
- toolCallId: options.toolCallId,
861
- agent,
862
- threadId: subChatId,
863
- messages,
864
- },
865
- });
866
- };
867
- write([]);
868
- // Each write re-sends the whole snapshot — throttle the re-sends.
954
+ const progress = subagentProgressWriter(writer, {
955
+ toolCallId: options.toolCallId,
956
+ agent,
957
+ threadId: subChatId,
958
+ });
959
+ progress.now([]);
869
960
  // The trailing tokens are covered by the tool RESULT (the client
870
961
  // switches to the persisted thread once output lands).
871
- let lastWrite = 0;
872
- onProgress = (messages) => {
873
- const now = Date.now();
874
- if (now - lastWrite < 250)
875
- return;
876
- lastWrite = now;
877
- write(messages);
878
- };
962
+ onProgress = progress.throttled;
879
963
  }
880
964
  // Propagate the PARENT turn's execution mode AS A SET: interactive →
881
965
  // the subagent gates its tools (a pending write bubbles up here);
@@ -890,7 +974,13 @@ export function createRuntime(config) {
890
974
  depth: depth + 1,
891
975
  autoApprove,
892
976
  blockGated,
977
+ // Memory off stays off down the tree: an eval turn's delegate must
978
+ // not recall earlier inputs or save new memory either.
979
+ ...(memoryOn ? {} : { memory: false }),
893
980
  onProgress,
981
+ // Who is delegating — persisted on the sub-thread so every later
982
+ // resolve of it can be shaped per parent (DynamicAgentResolveContext).
983
+ parentAgent: run.agent,
894
984
  // Telemetry linkage: this turn is the parent; carry the root chat
895
985
  // down so the subagent's trace nests under this conversation.
896
986
  parentRunId: run.id,
@@ -963,18 +1053,36 @@ export function createRuntime(config) {
963
1053
  // Test runs: stub the approval-gated (mutating) tools so a trial has no
964
1054
  // real side effects. Gated set = connection defaults + the agent's policy.
965
1055
  if (blockGated) {
1056
+ /**
1057
+ * Does this approval value actually gate the tool?
1058
+ *
1059
+ * NOT a truthiness check. The SDK spells a waiver as the STRING
1060
+ * `"not-applicable"` — which is truthy — so `if (v)` treated every
1061
+ * tool an agent had explicitly waived as gated and stubbed it. On an
1062
+ * agent whose reads are waived to override a connection default, that
1063
+ * is every read tool: the agent could not retrieve anything, and an
1064
+ * eval run measured a model with no access to its own corpus.
1065
+ *
1066
+ * Gated: `true`, `"user-approval"`, an object (approval with a
1067
+ * reason), a function. Waived: `false`, `undefined`,
1068
+ * `"not-applicable"`.
1069
+ */
1070
+ const gatedBy = (v) => v === true ||
1071
+ v === "user-approval" ||
1072
+ typeof v === "function" ||
1073
+ (typeof v === "object" && v !== null);
966
1074
  const gated = new Set();
967
1075
  for (const [k, v] of Object.entries(harnessApproval))
968
- if (v)
1076
+ if (gatedBy(v))
969
1077
  gated.add(k);
970
1078
  for (const o of opened) {
971
1079
  for (const [k, v] of Object.entries(o.toolApproval ?? {}))
972
- if (v)
1080
+ if (gatedBy(v))
973
1081
  gated.add(k);
974
1082
  }
975
1083
  if (cfg.toolApproval && typeof cfg.toolApproval !== "function") {
976
1084
  for (const [k, v] of Object.entries(cfg.toolApproval)) {
977
- if (v)
1085
+ if (gatedBy(v))
978
1086
  gated.add(k);
979
1087
  else
980
1088
  gated.delete(k);
@@ -992,12 +1100,19 @@ export function createRuntime(config) {
992
1100
  if (t && typeof t.execute === "function") {
993
1101
  tools[name] = {
994
1102
  ...t,
995
- execute: async (input) => ({
996
- __blocked: true,
997
- tool: name,
998
- input,
999
- reason: "approval-gated tool blocked during test run (no real side effects)",
1000
- }),
1103
+ execute: async (input) => blockGated === "simulate-success"
1104
+ ? {
1105
+ ok: true,
1106
+ __stubbed: true,
1107
+ tool: name,
1108
+ input,
1109
+ }
1110
+ : {
1111
+ __blocked: true,
1112
+ tool: name,
1113
+ input,
1114
+ reason: "approval-gated tool blocked during test run (no real side effects)",
1115
+ },
1001
1116
  };
1002
1117
  }
1003
1118
  }
@@ -1056,7 +1171,19 @@ export function createRuntime(config) {
1056
1171
  ? ({ steps }) => steps.reduce((n, st) => n + (st.usage?.inputTokens ?? 0) + (st.usage?.outputTokens ?? 0), 0) >=
1057
1172
  tokenBudget
1058
1173
  : undefined;
1059
- const baseStopWhen = budgetStop ? [stepCap ?? stepCountIs(20), budgetStop] : stepCap;
1174
+ // A delegation that paused on the user's approval ends the turn, exactly as
1175
+ // a gated tool of the agent's own does. A gated call has no result, so the
1176
+ // SDK stops there by itself; `spawn_agent` RETURNS one (`awaiting_approval`),
1177
+ // so without this the loop continues and only the result's note stands
1178
+ // between the model and a second delegation — another bubbled card per
1179
+ // step, up to the step cap. Added only when the agent can delegate, so
1180
+ // other agents' stop conditions stay exactly as configured.
1181
+ const subagentPauseStop = tools.spawn_agent
1182
+ ? ({ steps }) => (steps.at(-1)?.toolResults ?? []).some((r) => r.toolName === "spawn_agent" &&
1183
+ r.output?.status === "awaiting_approval")
1184
+ : undefined;
1185
+ const extraStops = [budgetStop, subagentPauseStop].filter((s) => s !== undefined);
1186
+ const baseStopWhen = extraStops.length > 0 ? [stepCap ?? stepCountIs(20), ...extraStops] : stepCap;
1060
1187
  // --- step boundaries for the seam -------------------------------------------
1061
1188
  // A step ends twice from here: on the model side (`agent.stream`'s
1062
1189
  // onStepEnd — usage, finish reason, tool calls) and on the UI side (the
@@ -1130,7 +1257,7 @@ export function createRuntime(config) {
1130
1257
  : config.resolvePromptCaching
1131
1258
  ? config.resolvePromptCaching(modelId, {
1132
1259
  agent: run.agent,
1133
- memoryScoped: !!(cfg.memory && config.memory),
1260
+ memoryScoped: !!(memoryOn && cfg.memory && config.memory),
1134
1261
  principal,
1135
1262
  })
1136
1263
  : // No hook → caching is ON by default for first-party Anthropic /
@@ -1704,10 +1831,7 @@ export function createRuntime(config) {
1704
1831
  out = main;
1705
1832
  args.observeStream(observed);
1706
1833
  }
1707
- return createUIMessageStreamResponse({
1708
- stream: out,
1709
- consumeSseStream: consumeStream,
1710
- });
1834
+ return out;
1711
1835
  }
1712
1836
  async function reapTick(opts) {
1713
1837
  const reaped = await storage.reapExpiredRuns(opts?.now ?? Date.now(), {
@@ -1860,7 +1984,10 @@ export function createRuntime(config) {
1860
1984
  });
1861
1985
  }
1862
1986
  const resolvedChatId = chat.id;
1863
- const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(agentName, principal, request, turnContext);
1987
+ const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(agentName, principal, request, turnContext,
1988
+ // A host may continue a subagent thread through the chat surface; keep
1989
+ // resolving it the way its spawner would.
1990
+ { parentAgent: chat.parentAgent });
1864
1991
  const prior = await storage.loadMessages(principal, resolvedChatId);
1865
1992
  // A new message while the last turn is parked on an approval is itself the
1866
1993
  // decision: the user declined to answer and wants to steer elsewhere. Close
@@ -2028,7 +2155,12 @@ export function createRuntime(config) {
2028
2155
  // reconcilable even if its (dynamic, DB-stored) agent was since deleted.
2029
2156
  // Before this ordering that throw also aborted `sweep`'s loop, so one such
2030
2157
  // run held up recovery for every other run until its attempts ran out.
2031
- const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(run.agent, principal);
2158
+ // A recovered subagent run must resolve exactly as its spawner shaped it —
2159
+ // crash recovery and `sweep` both land here.
2160
+ const chat = await storage.getChat(principal, run.chatId);
2161
+ const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(run.agent, principal, undefined, undefined, {
2162
+ parentAgent: chat?.parentAgent,
2163
+ });
2032
2164
  // A resume was admitted when the turn first ran — never refuse it (the
2033
2165
  // half-done work would strand), but its tokens still settle against the
2034
2166
  // same counters, and the mid-turn stop still applies.
@@ -2133,57 +2265,115 @@ export function createRuntime(config) {
2133
2265
  // Merge the decisions onto the persisted assistant message, then re-persist
2134
2266
  // only what changed. The agent continues from this updated history.
2135
2267
  const messages = await storage.loadMessages(principal, chatId);
2268
+ // A decision only means something for the message the turn paused on —
2269
+ // the LAST one. An approval from further back (an old tab, an old Slack
2270
+ // card, a late retry after the user moved on) was already settled: the
2271
+ // next message declined it. Continuing anyway would run the agent again
2272
+ // and append an answer nobody asked for. A real retry (a crash or a busy
2273
+ // chat after the decision was saved) still finds its approval last.
2274
+ const live = decisions.filter((d) => approvalIsLast(messages, d.approvalId));
2275
+ if (live.length === 0)
2276
+ return new Response(null, { status: 200 });
2277
+ decisions = live;
2136
2278
  const changed = applyDecisions(messages, decisions);
2137
2279
  if (changed.length > 0)
2138
2280
  await storage.appendMessages(chatId, changed);
2281
+ // Continue this chat from the decided history.
2282
+ const continueTurn = async () => {
2283
+ const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(chat.agent, principal, request, turnContext, { parentAgent: chat.parentAgent });
2284
+ const admitted = await admitContinuation(principal, chat.agent, "decision");
2285
+ const { run, lease } = await openRun(principal, chatId, chat.agent, cfg.configVersion);
2286
+ return buildTurnStream({
2287
+ resolvedChatId: chatId,
2288
+ principal,
2289
+ modelId,
2290
+ cfg,
2291
+ run,
2292
+ lease,
2293
+ // Resume: strip the paused turn's trailing text so the model call ends on
2294
+ // the tool result (a user turn), not an assistant prefill.
2295
+ uiMessages: trimTrailingAssistantPrefill(messages),
2296
+ runtimeCtx,
2297
+ // A human just decided something, so a human is present — even if this
2298
+ // turn parks again on the NEXT gate (see TurnTrigger).
2299
+ trigger: "decision",
2300
+ admitted,
2301
+ });
2302
+ };
2139
2303
  // Route any just-decided SUBAGENT approvals (bubbled up via spawn_agent):
2140
2304
  // resume the sub thread with the decision, then fold its answer back into
2141
2305
  // the parent's spawn_agent result so this turn continues with the digest.
2306
+ //
2307
+ // All of it streams in THIS response, which is the only channel back to a
2308
+ // client already showing the paused message. The resume can run for
2309
+ // minutes, pause on a SECOND gate, or fold an answer the client has never
2310
+ // seen — answered with an empty body instead, the user saw their approval
2311
+ // land and then nothing: no progress, and no card for the next gate, while
2312
+ // the chat sat awaiting a decision it never offered.
2142
2313
  const subApprovals = collectDecidedSubagentApprovals(messages);
2143
2314
  if (subApprovals.length > 0) {
2144
- for (const sa of subApprovals) {
2145
- const resumed = await resumeSubagentApproval(principal, sa.subChatId, sa.subApprovalIds.map((id) => ({
2146
- approvalId: id,
2147
- approved: sa.approved,
2148
- reason: sa.reason,
2149
- })), { parentRunId: sa.parentRunId, rootChatId: sa.rootChatId });
2150
- if (resumed.pending?.length) {
2151
- // The resumed sub paused on ANOTHER gated tool — re-bubble it as a
2152
- // fresh data-subagent-approval part on the same parent message. The
2153
- // undecided part keeps this chat awaiting_input (hasPendingApproval
2154
- // below), and the next decision routes through here again, so
2155
- // bubbling recurses per round.
2156
- const nextApprovalId = generateSubApprovalId();
2157
- appendSubagentApprovalPart(messages, sa.approvalId, {
2158
- approvalId: nextApprovalId,
2159
- subChatId: sa.subChatId,
2160
- subApprovalIds: resumed.pending.map((p) => p.approvalId),
2161
- spawnToolCallId: sa.spawnToolCallId,
2162
- summaries: resumed.pending.map((p) => p.summary),
2163
- // Carry the linkage forward so each further resume stays nested.
2164
- parentRunId: sa.parentRunId,
2165
- rootChatId: sa.rootChatId,
2166
- });
2167
- if (sa.spawnToolCallId) {
2168
- overwriteToolResult(messages, sa.spawnToolCallId, {
2169
- status: "awaiting_approval",
2170
- threadId: sa.subChatId,
2171
- pending: resumed.pending.map((p) => p.summary),
2172
- note: "Awaiting the user's approval — do not retry; stop here, the user will decide and you'll continue automatically.",
2173
- });
2315
+ const stream = createUIMessageStream({
2316
+ execute: async ({ writer }) => {
2317
+ for (const sa of subApprovals) {
2318
+ const resumed = await resumeSubagentApproval(principal, sa.subChatId, sa.subApprovalIds.map((id) => ({
2319
+ approvalId: id,
2320
+ approved: sa.approved,
2321
+ reason: sa.reason,
2322
+ })), { parentRunId: sa.parentRunId, rootChatId: sa.rootChatId }, { writer, spawnToolCallId: sa.spawnToolCallId });
2323
+ let output;
2324
+ if (resumed.pending?.length) {
2325
+ // The resumed sub paused on ANOTHER gated tool — re-bubble it as a
2326
+ // fresh data-subagent-approval part on the same parent message. The
2327
+ // undecided part keeps this chat awaiting_input (hasPendingApproval
2328
+ // below), and the next decision routes through here again, so
2329
+ // bubbling recurses per round.
2330
+ const data = {
2331
+ approvalId: generateSubApprovalId(),
2332
+ subChatId: sa.subChatId,
2333
+ subApprovalIds: resumed.pending.map((p) => p.approvalId),
2334
+ spawnToolCallId: sa.spawnToolCallId,
2335
+ summaries: resumed.pending.map((p) => p.summary),
2336
+ // Carry the linkage forward so each further resume stays nested.
2337
+ parentRunId: sa.parentRunId,
2338
+ rootChatId: sa.rootChatId,
2339
+ };
2340
+ appendSubagentApprovalPart(messages, sa.approvalId, data);
2341
+ // Same id as the stored part, so the client reconciles it into the
2342
+ // paused message exactly as a reload would show it.
2343
+ writer.write({ type: "data-subagent-approval", id: data.approvalId, data });
2344
+ output = {
2345
+ status: "awaiting_approval",
2346
+ threadId: sa.subChatId,
2347
+ pending: resumed.pending.map((p) => p.summary),
2348
+ note: "Awaiting the user's approval — do not retry; stop here, the user will decide and you'll continue automatically.",
2349
+ };
2350
+ }
2351
+ else {
2352
+ output = { threadId: sa.subChatId, answer: resumed.answer };
2353
+ }
2354
+ if (sa.spawnToolCallId) {
2355
+ overwriteToolResult(messages, sa.spawnToolCallId, output);
2356
+ writer.write({ type: "tool-output-available", toolCallId: sa.spawnToolCallId, output });
2357
+ }
2358
+ markSubagentApprovalResolved(messages, sa.approvalId);
2174
2359
  }
2175
- }
2176
- else if (sa.spawnToolCallId) {
2177
- overwriteToolResult(messages, sa.spawnToolCallId, {
2178
- threadId: sa.subChatId,
2179
- answer: resumed.answer,
2180
- });
2181
- }
2182
- markSubagentApprovalResolved(messages, sa.approvalId);
2183
- }
2184
- // Persist the folded spawn_agent result (or re-bubbled approval part)
2185
- // + resolved markers.
2186
- await storage.appendMessages(chatId, messages);
2360
+ // Persist the folded spawn_agent result (or re-bubbled approval part)
2361
+ // + resolved markers.
2362
+ await storage.appendMessages(chatId, messages);
2363
+ if (hasPendingApproval(messages)) {
2364
+ writer.write({ type: "finish" });
2365
+ return;
2366
+ }
2367
+ writer.merge(await continueTurn());
2368
+ },
2369
+ onError: (error) => clientErrorMessage(error),
2370
+ });
2371
+ return createUIMessageStreamResponse({
2372
+ stream: stream,
2373
+ // Drain server-side even if the client leaves: the resume must finish
2374
+ // and persist either way, as every turn does (see buildTurnStream).
2375
+ consumeSseStream: consumeStream,
2376
+ });
2187
2377
  }
2188
2378
  // If the step had several gated calls and some are still undecided, DON'T
2189
2379
  // re-run yet — a tool call without a result makes convertToModelMessages
@@ -2192,24 +2382,9 @@ export function createRuntime(config) {
2192
2382
  if (hasPendingApproval(messages)) {
2193
2383
  return new Response(null, { status: 200 });
2194
2384
  }
2195
- const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(chat.agent, principal, request, turnContext);
2196
- const admitted = await admitContinuation(principal, chat.agent, "decision");
2197
- const { run, lease } = await openRun(principal, chatId, chat.agent, cfg.configVersion);
2198
- return buildTurn({
2199
- resolvedChatId: chatId,
2200
- principal,
2201
- modelId,
2202
- cfg,
2203
- run,
2204
- lease,
2205
- // Resume: strip the paused turn's trailing text so the model call ends on
2206
- // the tool result (a user turn), not an assistant prefill.
2207
- uiMessages: trimTrailingAssistantPrefill(messages),
2208
- runtimeCtx,
2209
- // A human just decided something, so a human is present — even if this
2210
- // turn parks again on the NEXT gate (see TurnTrigger).
2211
- trigger: "decision",
2212
- admitted,
2385
+ return createUIMessageStreamResponse({
2386
+ stream: (await continueTurn()),
2387
+ consumeSseStream: consumeStream,
2213
2388
  });
2214
2389
  }
2215
2390
  /**
@@ -2240,7 +2415,7 @@ export function createRuntime(config) {
2240
2415
  await storage.appendMessages(chatId, changed);
2241
2416
  return new Response(null, { status: 200 });
2242
2417
  }
2243
- const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(chat.agent, principal, request, turnContext);
2418
+ const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(chat.agent, principal, request, turnContext, { parentAgent: chat.parentAgent });
2244
2419
  const admitted = await admitContinuation(principal, chat.agent, "decision");
2245
2420
  // The result is persisted WITH the run (atomically where the adapter can):
2246
2421
  // a busy chat refuses before anything is recorded, so the retry finds the
@@ -2286,7 +2461,7 @@ export function createRuntime(config) {
2286
2461
  if (hasPendingApproval(live) || hasPendingToolResult([live[live.length - 1]])) {
2287
2462
  throw new PendingApprovalError("This chat is awaiting a decision — resolve the pending request before compacting.");
2288
2463
  }
2289
- const { cfg, modelId } = await resolveAgentConfig(chat.agent, principal, request, turnContext);
2464
+ const { cfg, modelId } = await resolveAgentConfig(chat.agent, principal, request, turnContext, { parentAgent: chat.parentAgent });
2290
2465
  // Tool definitions only affect how a recorded tool RESULT is rendered for the
2291
2466
  // model (`toModelOutput`); an unknown tool falls back to its raw JSON rather
2292
2467
  // than failing. So take every definition available without I/O — the agent's
@@ -2419,11 +2594,14 @@ export function createRuntime(config) {
2419
2594
  return chat ? run : null;
2420
2595
  }
2421
2596
  /** A fresh user message from plain text (for cron-fired turns). */
2422
- function userMessageOf(text) {
2597
+ function userMessageOf(text, files) {
2423
2598
  return {
2424
2599
  id: generateMessageId(),
2425
2600
  role: "user",
2426
- parts: [{ type: "text", text }],
2601
+ parts: [
2602
+ ...(files ?? []).map((f) => ({ type: "file", url: f.url, mediaType: f.mediaType, ...(f.filename ? { filename: f.filename } : {}) })),
2603
+ { type: "text", text },
2604
+ ],
2427
2605
  metadata: { visibility: "user", createdAt: Date.now() },
2428
2606
  };
2429
2607
  }
@@ -2485,8 +2663,15 @@ export function createRuntime(config) {
2485
2663
  // Runtime-created thread: stamp it internal so a host filtering its
2486
2664
  // chat list on kind never shows it (see ChatRecord.kind).
2487
2665
  kind: "internal",
2666
+ ...(o.parentAgent ? { parentAgent: o.parentAgent } : {}),
2488
2667
  });
2489
2668
  }
2669
+ // A spawn may only continue a thread its own agent delegated. Otherwise
2670
+ // agent B could pass agent A's sub-thread id and have the child resolved
2671
+ // under A — borrowing whatever a host grants A's subagents.
2672
+ if (o.parentAgent && chat.parentAgent && chat.parentAgent !== o.parentAgent) {
2673
+ throw new Error(`Thread ${chatId} was delegated by another agent; omit threadId to start a new one.`);
2674
+ }
2490
2675
  const prior = await storage.loadMessages(o.principal, chatId);
2491
2676
  // Same rule as handleChat: a new message while the thread is parked on an
2492
2677
  // approval IS the decision — close the stale gate as declined (persisted)
@@ -2503,7 +2688,14 @@ export function createRuntime(config) {
2503
2688
  // the run first, persist the message only once it is ours to run.
2504
2689
  const admitted = await admitTurn(o.principal, o.agent, o.trigger);
2505
2690
  const { cfg, modelId, runtimeCtx, run, lease } = await withReservation(admitted, async () => {
2506
- const resolved = await resolveAgentConfig(o.agent, o.principal, undefined, o.turnContext);
2691
+ const resolved = await resolveAgentConfig(o.agent, o.principal, undefined, o.turnContext, {
2692
+ // A continued thread keeps the parent it was created under (a mismatch
2693
+ // was refused above). Without a stored one — an adapter that doesn't
2694
+ // persist the column — the spawner of THIS turn is the parent: a host
2695
+ // that restricts a subagent by its parent must never see none.
2696
+ parentAgent: chat.parentAgent ?? o.parentAgent,
2697
+ autoApprove: o.autoApprove,
2698
+ });
2507
2699
  const opened = await openTurn(o.principal, chatId, o.agent, [o.message], resolved.cfg.configVersion);
2508
2700
  return { ...resolved, ...opened };
2509
2701
  });
@@ -2520,35 +2712,20 @@ export function createRuntime(config) {
2520
2712
  admitted,
2521
2713
  autoApprove: o.autoApprove,
2522
2714
  blockGated: o.blockGated,
2715
+ memory: o.memory,
2716
+ model: o.model,
2523
2717
  promptCaching: o.promptCaching,
2524
2718
  depth: o.depth ?? 0,
2525
2719
  parentRunId: o.parentRunId,
2526
2720
  rootChatId: o.rootChatId,
2527
2721
  trigger: o.trigger,
2528
2722
  // Forward progressive UI-message snapshots to the caller (spawn_agent
2529
- // streams them into the parent turn). Observation must never break the
2530
- // run — the primary stream is drained independently below.
2531
- observeStream: onProgress
2532
- ? (observed) => {
2533
- void (async () => {
2534
- try {
2535
- const byId = new Map();
2536
- for await (const m of readUIMessageStream({
2537
- stream: observed,
2538
- })) {
2539
- const msg = m;
2540
- byId.set(msg.id, msg);
2541
- onProgress([o.message, ...byId.values()]);
2542
- }
2543
- }
2544
- catch {
2545
- // ignore — live progress is best-effort
2546
- }
2547
- })();
2548
- }
2549
- : undefined,
2723
+ // streams them into the parent turn).
2724
+ observeStream: onProgress ? progressObserver([o.message], onProgress) : undefined,
2550
2725
  });
2551
- await res.text();
2726
+ // The turn's own failure travels as an `error` chunk — the stream, not a
2727
+ // throw. Keep its text: without it a failed turn reads as a silent one.
2728
+ const error = lastStreamError(await res.text());
2552
2729
  const after = await storage.loadMessages(o.principal, chatId);
2553
2730
  // The verdict is about THIS turn, so read the FINAL message only (the same
2554
2731
  // rule hasPendingApproval applies): an undecided gate in an older message
@@ -2563,6 +2740,7 @@ export function createRuntime(config) {
2563
2740
  answer: pending.length > 0 ? "" : lastAssistantText(after),
2564
2741
  parked,
2565
2742
  ...(pending.length > 0 ? { pending } : {}),
2743
+ ...(error ? { error } : {}),
2566
2744
  };
2567
2745
  }
2568
2746
  /**
@@ -2577,7 +2755,13 @@ export function createRuntime(config) {
2577
2755
  */
2578
2756
  async function resumeSubagentApproval(principal, subChatId, decisions,
2579
2757
  /** Telemetry linkage of the original sub-turn, so the trace stays nested. */
2580
- linkage) {
2758
+ linkage,
2759
+ /**
2760
+ * Stream the resumed sub-turn into the parent's response as the same
2761
+ * `data-subagent-progress` part `spawn_agent` wrote, so the delegation the
2762
+ * user just approved visibly runs again instead of sitting silent.
2763
+ */
2764
+ live) {
2581
2765
  const subChat = await storage.getChat(principal, subChatId);
2582
2766
  if (!subChat)
2583
2767
  return { answer: "" };
@@ -2588,9 +2772,29 @@ export function createRuntime(config) {
2588
2772
  // Another gated call in the same sub step still undecided → can't finish.
2589
2773
  if (hasPendingApproval(subMessages))
2590
2774
  return { answer: "" };
2591
- const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(subChat.agent, principal);
2775
+ const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(subChat.agent, principal, undefined, undefined,
2776
+ // Same shaping as the first sub-turn: the parent is read off the persisted
2777
+ // sub-thread, and this path is only ever reached by a human decision.
2778
+ { parentAgent: subChat.parentAgent, autoApprove: false });
2592
2779
  const admitted = await admitContinuation(principal, subChat.agent, "decision");
2593
2780
  const { run, lease } = await openRun(principal, subChatId, subChat.agent, cfg.configVersion);
2781
+ const resumeFrom = trimTrailingAssistantPrefill(subMessages);
2782
+ let observeStream;
2783
+ if (live) {
2784
+ const progress = subagentProgressWriter(live.writer, {
2785
+ toolCallId: live.spawnToolCallId,
2786
+ agent: subChat.agent,
2787
+ threadId: subChatId,
2788
+ });
2789
+ // The exchange being resumed: its prompt and the paused assistant
2790
+ // message, which the resumed turn extends in place.
2791
+ const last = resumeFrom[resumeFrom.length - 1];
2792
+ const continuing = last?.role === "assistant" ? last : undefined;
2793
+ const prompt = resumeFrom.findLast((m) => m.role === "user");
2794
+ const lead = prompt ? [prompt] : [];
2795
+ progress.now(continuing ? [...lead, continuing] : lead);
2796
+ observeStream = progressObserver(lead, progress.throttled, continuing);
2797
+ }
2594
2798
  const res = await buildTurn({
2595
2799
  resolvedChatId: subChatId,
2596
2800
  principal,
@@ -2602,7 +2806,7 @@ export function createRuntime(config) {
2602
2806
  // Resume: same prefill guard as applyApproval — the paused sub turn may
2603
2807
  // have narrated after its gated call, and that text must not reach the
2604
2808
  // model as an assistant prefill.
2605
- uiMessages: trimTrailingAssistantPrefill(subMessages),
2809
+ uiMessages: resumeFrom,
2606
2810
  runtimeCtx,
2607
2811
  autoApprove: false, // keep gating any further writes in the sub
2608
2812
  depth: 1,
@@ -2611,6 +2815,7 @@ export function createRuntime(config) {
2611
2815
  rootChatId: linkage?.rootChatId,
2612
2816
  // Reached only by a human deciding the bubbled-up approval.
2613
2817
  trigger: "decision",
2818
+ observeStream,
2614
2819
  });
2615
2820
  await res.text();
2616
2821
  const after = await storage.loadMessages(principal, subChatId);
@@ -2825,24 +3030,26 @@ export function createRuntime(config) {
2825
3030
  const self = {
2826
3031
  listAgents: agentInfos,
2827
3032
  handleChat,
2828
- runAgent: async ({ principal, agent, prompt, threadId, blockGated = true, turnContext, promptCaching,
3033
+ runAgent: async ({ principal, agent, prompt, files, threadId, blockGated = true, memory, model, turnContext, promptCaching,
2829
3034
  // No client stream: assume unattended unless the host says otherwise, so
2830
3035
  // a park here can't be silently filtered out as "someone's watching".
2831
3036
  trigger = "programmatic", }) => {
2832
3037
  const r = await runToCompletion({
2833
3038
  principal,
2834
3039
  agent,
2835
- message: userMessageOf(prompt),
3040
+ message: userMessageOf(prompt, files),
2836
3041
  chatId: threadId,
2837
3042
  depth: 1,
2838
3043
  // Autonomous: no human to authorize (blockGated stubs the gated tools).
2839
3044
  autoApprove: true,
2840
3045
  blockGated,
3046
+ memory,
3047
+ model,
2841
3048
  turnContext,
2842
3049
  promptCaching,
2843
3050
  trigger,
2844
3051
  });
2845
- return { threadId: r.chatId, answer: r.answer };
3052
+ return { threadId: r.chatId, answer: r.answer, ...(r.error ? { error: r.error } : {}) };
2846
3053
  },
2847
3054
  compactChat,
2848
3055
  loadHistory,