@trigger.dev/sdk 0.0.0-prerelease-20260911153544 → 0.0.0-prerelease-windows-bundle-fix-20260928100448

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (124) hide show
  1. package/dist/commonjs/v3/ai.d.ts +37 -12
  2. package/dist/commonjs/v3/ai.js +349 -278
  3. package/dist/commonjs/v3/ai.js.map +1 -1
  4. package/dist/commonjs/v3/auth.js +8 -2
  5. package/dist/commonjs/v3/auth.js.map +1 -1
  6. package/dist/commonjs/v3/chat-client.js +7 -0
  7. package/dist/commonjs/v3/chat-client.js.map +1 -1
  8. package/dist/commonjs/v3/chat-server.d.ts +1 -0
  9. package/dist/commonjs/v3/chat-server.js +8 -0
  10. package/dist/commonjs/v3/chat-server.js.map +1 -1
  11. package/dist/commonjs/v3/chat.js +13 -9
  12. package/dist/commonjs/v3/chat.js.map +1 -1
  13. package/dist/commonjs/v3/compactionResponse.d.ts +9 -0
  14. package/dist/commonjs/v3/compactionResponse.js +37 -0
  15. package/dist/commonjs/v3/compactionResponse.js.map +1 -0
  16. package/dist/commonjs/v3/concurrency-shared.d.ts +13 -0
  17. package/dist/commonjs/v3/concurrency-shared.js +35 -0
  18. package/dist/commonjs/v3/concurrency-shared.js.map +1 -0
  19. package/dist/commonjs/v3/concurrencyLimits.d.ts +73 -0
  20. package/dist/commonjs/v3/concurrencyLimits.js +166 -0
  21. package/dist/commonjs/v3/concurrencyLimits.js.map +1 -0
  22. package/dist/commonjs/v3/index.d.ts +2 -1
  23. package/dist/commonjs/v3/index.js +3 -1
  24. package/dist/commonjs/v3/index.js.map +1 -1
  25. package/dist/commonjs/v3/managedChatResponse.d.ts +44 -0
  26. package/dist/commonjs/v3/managedChatResponse.js +233 -0
  27. package/dist/commonjs/v3/managedChatResponse.js.map +1 -0
  28. package/dist/commonjs/v3/queues.d.ts +31 -0
  29. package/dist/commonjs/v3/queues.js +31 -0
  30. package/dist/commonjs/v3/queues.js.map +1 -1
  31. package/dist/commonjs/v3/runs.d.ts +5 -0
  32. package/dist/commonjs/v3/sessionTracing.d.ts +7 -0
  33. package/dist/commonjs/v3/sessionTracing.js +44 -0
  34. package/dist/commonjs/v3/sessionTracing.js.map +1 -0
  35. package/dist/commonjs/v3/sessions.js +31 -34
  36. package/dist/commonjs/v3/sessions.js.map +1 -1
  37. package/dist/commonjs/v3/shared.d.ts +18 -1
  38. package/dist/commonjs/v3/shared.js +145 -52
  39. package/dist/commonjs/v3/shared.js.map +1 -1
  40. package/dist/commonjs/v3/steeringContext.d.ts +41 -0
  41. package/dist/commonjs/v3/steeringContext.js +118 -0
  42. package/dist/commonjs/v3/steeringContext.js.map +1 -0
  43. package/dist/commonjs/v3/transcriptStorage.d.ts +4 -1
  44. package/dist/commonjs/v3/transcriptStorage.js +55 -4
  45. package/dist/commonjs/v3/transcriptStorage.js.map +1 -1
  46. package/dist/commonjs/version.js +1 -1
  47. package/dist/esm/v3/ai.d.ts +37 -12
  48. package/dist/esm/v3/ai.js +350 -279
  49. package/dist/esm/v3/ai.js.map +1 -1
  50. package/dist/esm/v3/auth.js +8 -2
  51. package/dist/esm/v3/auth.js.map +1 -1
  52. package/dist/esm/v3/chat-client.js +7 -0
  53. package/dist/esm/v3/chat-client.js.map +1 -1
  54. package/dist/esm/v3/chat-server.d.ts +1 -0
  55. package/dist/esm/v3/chat-server.js +8 -0
  56. package/dist/esm/v3/chat-server.js.map +1 -1
  57. package/dist/esm/v3/chat.js +13 -9
  58. package/dist/esm/v3/chat.js.map +1 -1
  59. package/dist/esm/v3/compactionResponse.d.ts +9 -0
  60. package/dist/esm/v3/compactionResponse.js +34 -0
  61. package/dist/esm/v3/compactionResponse.js.map +1 -0
  62. package/dist/esm/v3/concurrency-shared.d.ts +13 -0
  63. package/dist/esm/v3/concurrency-shared.js +31 -0
  64. package/dist/esm/v3/concurrency-shared.js.map +1 -0
  65. package/dist/esm/v3/concurrencyLimits.d.ts +73 -0
  66. package/dist/esm/v3/concurrencyLimits.js +158 -0
  67. package/dist/esm/v3/concurrencyLimits.js.map +1 -0
  68. package/dist/esm/v3/index.d.ts +2 -1
  69. package/dist/esm/v3/index.js +2 -1
  70. package/dist/esm/v3/index.js.map +1 -1
  71. package/dist/esm/v3/managedChatResponse.d.ts +44 -0
  72. package/dist/esm/v3/managedChatResponse.js +228 -0
  73. package/dist/esm/v3/managedChatResponse.js.map +1 -0
  74. package/dist/esm/v3/queues.d.ts +31 -0
  75. package/dist/esm/v3/queues.js +31 -0
  76. package/dist/esm/v3/queues.js.map +1 -1
  77. package/dist/esm/v3/runs.d.ts +5 -0
  78. package/dist/esm/v3/sessionTracing.d.ts +7 -0
  79. package/dist/esm/v3/sessionTracing.js +40 -0
  80. package/dist/esm/v3/sessionTracing.js.map +1 -0
  81. package/dist/esm/v3/sessions.js +31 -34
  82. package/dist/esm/v3/sessions.js.map +1 -1
  83. package/dist/esm/v3/shared.d.ts +18 -1
  84. package/dist/esm/v3/shared.js +145 -53
  85. package/dist/esm/v3/shared.js.map +1 -1
  86. package/dist/esm/v3/steeringContext.d.ts +41 -0
  87. package/dist/esm/v3/steeringContext.js +113 -0
  88. package/dist/esm/v3/steeringContext.js.map +1 -0
  89. package/dist/esm/v3/transcriptStorage.d.ts +4 -1
  90. package/dist/esm/v3/transcriptStorage.js +55 -5
  91. package/dist/esm/v3/transcriptStorage.js.map +1 -1
  92. package/dist/esm/version.js +1 -1
  93. package/docs/ai-chat/anatomy.mdx +44 -24
  94. package/docs/ai-chat/client-protocol.mdx +2 -0
  95. package/docs/ai-chat/error-handling.mdx +44 -76
  96. package/docs/ai-chat/frontend.mdx +27 -21
  97. package/docs/ai-chat/lifecycle-hooks.mdx +4 -2
  98. package/docs/ai-chat/patterns/branching-conversations.mdx +95 -230
  99. package/docs/ai-chat/patterns/human-in-the-loop.mdx +166 -164
  100. package/docs/ai-chat/patterns/tool-result-auditing.mdx +28 -27
  101. package/docs/ai-chat/pending-messages.mdx +19 -5
  102. package/docs/ai-chat/quick-start.mdx +26 -20
  103. package/docs/ai-chat/reference.mdx +4 -2
  104. package/docs/ai-chat/testing.mdx +16 -4
  105. package/docs/cli-env-commands.mdx +114 -0
  106. package/docs/cli-projects-commands.mdx +62 -0
  107. package/docs/cli-runs-commands.mdx +94 -0
  108. package/docs/concurrency.mdx +384 -0
  109. package/docs/database-connections.mdx +3 -3
  110. package/docs/deploy-environment-variables.mdx +6 -0
  111. package/docs/idempotency.mdx +43 -5
  112. package/docs/introduction.mdx +58 -152
  113. package/docs/limits.mdx +16 -6
  114. package/docs/observability/query.mdx +25 -0
  115. package/docs/queues.mdx +248 -0
  116. package/docs/reports.mdx +1 -1
  117. package/docs/runs/priority.mdx +2 -25
  118. package/docs/self-hosting/env/webapp.mdx +7 -0
  119. package/docs/tasks/overview.mdx +3 -5
  120. package/docs/troubleshooting-alerts.mdx +124 -1
  121. package/docs/troubleshooting.mdx +12 -0
  122. package/docs/writing-tasks-introduction.mdx +2 -1
  123. package/package.json +2 -2
  124. package/docs/queue-concurrency.mdx +0 -358
@@ -15,8 +15,12 @@ const v3_1 = require("@trigger.dev/core/v3");
15
15
  // Runtime VALUES go through the ESM/CJS shim so the CJS build can `require`
16
16
  // ESM-only `ai@7` (see ../imports/ai-runtime.ts).
17
17
  const api_1 = require("@opentelemetry/api");
18
+ const sessionTracing_js_1 = require("./sessionTracing.js");
18
19
  const ai_runtime_js_1 = require("../imports/ai-runtime.js");
19
20
  const transcriptStorage_js_1 = require("./transcriptStorage.js");
21
+ const compactionResponse_js_1 = require("./compactionResponse.js");
22
+ const managedChatResponse_js_1 = require("./managedChatResponse.js");
23
+ const steeringContext_js_1 = require("./steeringContext.js");
20
24
  let transcriptStorageOverride;
21
25
  /**
22
26
  * Test-only override for the storage `chat.agent` persists through, so a
@@ -46,6 +50,7 @@ const externalDeploymentId_js_1 = require("./externalDeploymentId.js");
46
50
  const chatVersionSkew_js_1 = require("./chatVersionSkew.js");
47
51
  const sessions_js_1 = require("./sessions.js");
48
52
  const shared_js_1 = require("./shared.js");
53
+ const concurrency_shared_js_1 = require("./concurrency-shared.js");
49
54
  const streams_js_1 = require("./streams.js");
50
55
  const tracer_js_1 = require("./tracer.js");
51
56
  const METADATA_KEY = "tool.execute.options";
@@ -54,17 +59,17 @@ const METADATA_KEY = "tool.execute.options";
54
59
  * `ignoreIncompleteToolCalls: true` to prevent failures from
55
60
  * stopped/aborted conversations with partial tool parts.
56
61
  */
57
- function toModelMessages(messages) {
62
+ function toModelMessages(messages, context) {
58
63
  // Pass the resolved per-turn `tools` (if any) so the AI SDK can look up each
59
64
  // tool's `toModelOutput` and re-apply it to prior-turn tool results. Without
60
65
  // `tools` it falls back to JSON-stringifying the raw output (TRI-10149). The
61
66
  // conditional spread keeps the options object byte-identical to the no-tools
62
67
  // path when nothing was declared.
63
68
  const tools = locals_js_1.locals.get(chatResolvedToolsKey);
64
- return (0, ai_runtime_js_1.convertToModelMessages)(messages, {
69
+ return (0, steeringContext_js_1.convertSteeredMessages)(messages, async (batch) => (0, ai_runtime_js_1.convertToModelMessages)(batch, {
65
70
  ignoreIncompleteToolCalls: true,
66
71
  ...(tools ? { tools } : {}),
67
- });
72
+ }), locals_js_1.locals.get(chatSteeringInjectionsKey) ?? new Map(), context ?? locals_js_1.locals.get(chatCurrentUIMessagesKey) ?? messages, context !== undefined || locals_js_1.locals.get(chatCurrentUIMessagesKey) !== undefined);
68
73
  }
69
74
  const chatTurnContextKey = locals_js_1.locals.create("chat.turnContext");
70
75
  /**
@@ -708,6 +713,21 @@ const chatOutGateKey = locals_js_1.locals.create("chat.outGate");
708
713
  * @internal
709
714
  */
710
715
  const CHAT_OUT_GATE_TIMEOUT_MS = 10_000;
716
+ /**
717
+ * The ids to save non-final after a failed turn: the stream's partial answer,
718
+ * but only while the message under its id is still that partial by content.
719
+ * `onTurnComplete` may hand back a cloned history (same content, new objects),
720
+ * which keeps it partial, or replace it in place, which finishes it.
721
+ * @internal
722
+ */
723
+ function partialStillUnfinished(partial, fingerprint, messages) {
724
+ if (!partial || fingerprint === undefined)
725
+ return undefined;
726
+ const current = messages.find((message) => message.id === partial.id);
727
+ if (!current || (0, transcriptStorage_js_1.fingerprintMessage)(current) !== fingerprint)
728
+ return undefined;
729
+ return new Set([partial.id]);
730
+ }
711
731
  async function awaitChatOutGate() {
712
732
  const gate = locals_js_1.locals.get(chatOutGateKey);
713
733
  if (!gate || gate.open)
@@ -854,7 +874,7 @@ const chatStream = {
854
874
  * `onTurnComplete`'s `responseMessage` and `uiMessages`.
855
875
  *
856
876
  * Non-transient data chunks (`type` starts with `data-`, no `transient: true`)
857
- * are queued for accumulation into the assistant response message.
877
+ * are accumulated in emission order into the assistant response message.
858
878
  * Transient or non-data chunks are streamed only (same as `chat.stream`).
859
879
  *
860
880
  * @example
@@ -872,15 +892,7 @@ const chatResponse = {
872
892
  * response message; everything else is stream-only.
873
893
  */
874
894
  write(part) {
875
- queueResponsePart(part);
876
- const { waitUntilComplete } = chatStream.writer({
877
- spanName: "chat.response.write",
878
- collapsed: true,
879
- execute: ({ write }) => {
880
- write(part);
881
- },
882
- });
883
- waitUntilComplete().catch(() => { });
895
+ managedResponse().writeData(part);
884
896
  },
885
897
  };
886
898
  /**
@@ -889,60 +901,13 @@ const chatResponse = {
889
901
  * @internal
890
902
  */
891
903
  function createLazyChatWriter() {
892
- let writeImpl = null;
893
- let mergeImpl = null;
894
- let waitPromise = null;
895
- let resolveExecute = null;
896
- let started = false;
897
- const bufferedParts = [];
898
- const bufferedStreams = [];
899
- function ensureInitialized() {
900
- if (started)
901
- return;
902
- started = true;
903
- const executePromise = new Promise((resolve) => {
904
- resolveExecute = resolve;
905
- });
906
- const { waitUntilComplete } = chatStream.writer({
904
+ return (0, managedChatResponse_js_1.createOrderedChatWriter)(() => (locals_js_1.locals.get(chatManagedResponseActiveKey) ? managedResponse() : undefined), async (stream) => {
905
+ const { waitUntilComplete } = chatStream.pipe(stream, {
907
906
  collapsed: true,
908
907
  spanName: "callback writer",
909
- execute: ({ write, merge }) => {
910
- writeImpl = write;
911
- mergeImpl = merge;
912
- for (const part of bufferedParts.splice(0))
913
- write(part);
914
- for (const stream of bufferedStreams.splice(0))
915
- merge(stream);
916
- return executePromise;
917
- },
918
908
  });
919
- waitPromise = waitUntilComplete;
920
- }
921
- return {
922
- writer: {
923
- write(part) {
924
- ensureInitialized();
925
- queueResponsePart(part);
926
- if (writeImpl)
927
- writeImpl(part);
928
- else
929
- bufferedParts.push(part);
930
- },
931
- merge(stream) {
932
- ensureInitialized();
933
- if (mergeImpl)
934
- mergeImpl(stream);
935
- else
936
- bufferedStreams.push(stream);
937
- },
938
- },
939
- async flush() {
940
- if (resolveExecute) {
941
- resolveExecute(); // Signal execute to complete
942
- await waitPromise(); // Wait for stream to finish piping
943
- }
944
- },
945
- };
909
+ await waitUntilComplete();
910
+ });
946
911
  }
947
912
  /**
948
913
  * Runs a callback with a lazy ChatWriter, flushing the stream after completion.
@@ -1095,7 +1060,7 @@ async function waitOnChatRoute(route, options) {
1095
1060
  return tracer_js_1.tracer.startActiveSpan(options.spanName ?? `chat.${route}.wait()`, async (span) => {
1096
1061
  const idleMs = (options.idleTimeoutInSeconds ?? 0) * 1000;
1097
1062
  if (idleMs > 0) {
1098
- const warm = await router.next(route, { timeoutMs: idleMs });
1063
+ const warm = await (0, sessionTracing_js_1.traceSessionIdle)(session.id, idleMs / 1000, () => router.next(route, { timeoutMs: idleMs }));
1099
1064
  if (warm) {
1100
1065
  span.setAttribute("wait.resolved", "idle");
1101
1066
  return { ok: true, output: warm.data, record: warm };
@@ -1149,10 +1114,6 @@ async function waitOnChatRoute(route, options) {
1149
1114
  session: session.id,
1150
1115
  io: "in",
1151
1116
  route,
1152
- ...(0, v3_1.accessoryAttributes)({
1153
- items: [{ text: `${session.id}.in:${route}`, variant: "normal" }],
1154
- style: "codepath",
1155
- }),
1156
1117
  },
1157
1118
  });
1158
1119
  }
@@ -1828,6 +1789,14 @@ const chatHandoverPartialKey = locals_js_1.locals.create("chat.handoverPartial")
1828
1789
  * @internal
1829
1790
  */
1830
1791
  const chatHandoverMessageIdKey = locals_js_1.locals.create("chat.handoverMessageId");
1792
+ /**
1793
+ * The model messages a turn-0 head-start splice contributed to the model lane,
1794
+ * keyed by the UI message id it was synthesized under. When the agent's response
1795
+ * completes that message under the same id, this is the run to replace: the UI
1796
+ * form alone converts to something else (its pending tool calls drop out, the
1797
+ * approval round was never on it), so it cannot locate the run itself.
1798
+ */
1799
+ const chatHandoverSplicedRunKey = locals_js_1.locals.create("chat.handoverSplicedRun");
1831
1800
  /**
1832
1801
  * Run-scoped slot indicating that the customer's step-1 head-start
1833
1802
  * response is the FINAL turn response. When true, turn 0 runs through
@@ -1911,16 +1880,19 @@ function synthesizeHandoverUIMessage(partial, messageId) {
1911
1880
  */
1912
1881
  function spliceHandoverPartial(modelMessages, uiMessages, signal) {
1913
1882
  if (!signal.partialAssistantMessage || signal.partialAssistantMessage.length === 0) {
1914
- return;
1883
+ return undefined;
1915
1884
  }
1916
1885
  // Skip if the hydrated chain already persisted the partial under this id.
1917
1886
  const alreadyInChain = signal.messageId !== undefined && uiMessages.some((m) => m.id === signal.messageId);
1918
1887
  if (alreadyInChain)
1919
- return;
1920
- modelMessages.push(...signal.partialAssistantMessage);
1888
+ return undefined;
1889
+ const run = [...signal.partialAssistantMessage];
1890
+ modelMessages.push(...run);
1921
1891
  const partialUI = synthesizeHandoverUIMessage(signal.partialAssistantMessage, signal.messageId);
1922
- if (partialUI)
1923
- uiMessages.push(partialUI);
1892
+ if (!partialUI)
1893
+ return undefined;
1894
+ uiMessages.push(partialUI);
1895
+ return { id: partialUI.id, run };
1924
1896
  }
1925
1897
  /**
1926
1898
  * Per-turn background context queue. Messages added via `chat.backgroundWork.inject()`
@@ -2526,29 +2498,19 @@ const chatTurnNewUIMessagesKey = locals_js_1.locals.create("chat.turnNewUIMessag
2526
2498
  const chatPendingSteerKey = locals_js_1.locals.create("chat.pendingSteer");
2527
2499
  /** @internal — IDs of messages that were successfully injected via prepareStep */
2528
2500
  const chatInjectedMessageIdsKey = locals_js_1.locals.create("chat.injectedMessageIds");
2529
- /** @internal — non-transient data parts queued via chat.response or writer.write() for accumulation into the response message */
2530
- const chatResponsePartsKey = locals_js_1.locals.create("chat.responseParts");
2531
- /**
2532
- * Check if a chunk is a non-transient data part that should persist to the response message.
2533
- * @internal
2534
- */
2535
- function isNonTransientDataPart(part) {
2536
- if (typeof part !== "object" || part === null)
2537
- return false;
2538
- const p = part;
2539
- return typeof p.type === "string" && p.type.startsWith("data-") && p.transient !== true;
2540
- }
2541
- /**
2542
- * Queue a chunk for accumulation into the response message (if it's a non-transient data part).
2543
- * Called by `chat.response.write()` and `ChatWriter.write()`.
2544
- * @internal
2545
- */
2546
- function queueResponsePart(part) {
2547
- if (!isNonTransientDataPart(part))
2548
- return;
2549
- const parts = locals_js_1.locals.get(chatResponsePartsKey) ?? [];
2550
- parts.push(part);
2551
- locals_js_1.locals.set(chatResponsePartsKey, parts);
2501
+ const chatManagedResponseKey = locals_js_1.locals.create("chat.managedResponse");
2502
+ const chatManagedResponseActiveKey = locals_js_1.locals.create("chat.managedResponseActive");
2503
+ const chatSteeringInjectionsKey = locals_js_1.locals.create("chat.steeringInjections");
2504
+ function managedResponse() {
2505
+ let response = locals_js_1.locals.get(chatManagedResponseKey);
2506
+ if (!response || response.isClosed) {
2507
+ response = new managedChatResponse_js_1.ManagedChatResponse(async (stream) => {
2508
+ const { waitUntilComplete } = chatStream.pipe(stream, { spanName: "managed chat response" });
2509
+ await waitUntilComplete();
2510
+ });
2511
+ locals_js_1.locals.set(chatManagedResponseKey, response);
2512
+ }
2513
+ return response;
2552
2514
  }
2553
2515
  /**
2554
2516
  * Check that no tool calls are in-flight in a step's content.
@@ -2780,6 +2742,7 @@ async function chatCompact(messages, steps, options) {
2780
2742
  locals_js_1.locals.set(chatCompactionStateKey, {
2781
2743
  summary,
2782
2744
  baseResponseMessageCount: currentStep.response.messages.length,
2745
+ baseResponseStepCount: steps.length,
2783
2746
  });
2784
2747
  // Set model-only override — UI messages stay intact for persistence.
2785
2748
  // The summary becomes the model message history for the next turn,
@@ -2899,6 +2862,8 @@ function modelFormOf(m, batch, injected) {
2899
2862
  * @internal
2900
2863
  */
2901
2864
  async function drainSteeringQueue(config, messages, steps, queueOverride) {
2865
+ if (steps.length === 0)
2866
+ managedResponse().beginGeneration();
2902
2867
  const queue = queueOverride ?? locals_js_1.locals.get(chatSteeringQueueKey);
2903
2868
  if (!queue || queue.length === 0)
2904
2869
  return EMPTY_DRAIN;
@@ -3031,31 +2996,24 @@ async function drainSteeringQueue(config, messages, steps, queueOverride) {
3031
2996
  }
3032
2997
  locals_js_1.locals.set(chatPendingSteerKey, pendingSteer);
3033
2998
  }
3034
- // Write injection confirmation chunk to the stream so the frontend
3035
- // knows which messages were injected and where in the response.
3036
2999
  if (injected.length > 0) {
3037
- try {
3038
- const { waitUntilComplete } = chatStream.writer({
3039
- collapsed: true,
3040
- execute: ({ write }) => {
3041
- write({
3042
- type: ai_shared_js_1.PENDING_MESSAGE_INJECTED_TYPE,
3043
- id: (0, ai_runtime_js_1.generateId)(),
3044
- data: {
3045
- messageIds: claimedUIMessages.map((m) => m.id),
3046
- messages: claimedUIMessages.map((m) => ({
3047
- id: m.id,
3048
- text: textOfUIMessage(m),
3049
- })),
3050
- },
3051
- });
3052
- },
3053
- });
3054
- await waitUntilComplete();
3055
- }
3056
- catch {
3057
- /* non-fatal — stream write failed */
3058
- }
3000
+ const id = (0, ai_runtime_js_1.generateId)();
3001
+ const messageIds = claimedUIMessages.map((m) => m.id);
3002
+ const injections = locals_js_1.locals.get(chatSteeringInjectionsKey) ?? new Map();
3003
+ injections.set(id, { id, messageIds, messages: injected });
3004
+ locals_js_1.locals.set(chatSteeringInjectionsKey, injections);
3005
+ const response = managedResponse();
3006
+ // prepareStep can run ahead of the UI consumer. Admit the marker only
3007
+ // once the actual preceding finish-step chunk has entered this stream.
3008
+ response.afterStep(steps.length);
3009
+ response.writeData({
3010
+ type: ai_shared_js_1.PENDING_MESSAGE_INJECTED_TYPE,
3011
+ id,
3012
+ data: {
3013
+ messageIds,
3014
+ messages: claimedUIMessages.map((m) => ({ id: m.id, text: textOfUIMessage(m) })),
3015
+ },
3016
+ });
3059
3017
  }
3060
3018
  // Fire onInjected callback
3061
3019
  if (config.onInjected && injected.length > 0) {
@@ -3356,7 +3314,7 @@ function buildManagedStreamTextOptions(options, config) {
3356
3314
  * regenerate needs: without it a regenerated answer can call nothing.
3357
3315
  */
3358
3316
  tools: (tools ?? agentTools),
3359
- });
3317
+ }, false);
3360
3318
  const promptSystem = locals_js_1.locals.get(chatPromptKey)?.text;
3361
3319
  /**
3362
3320
  * Two managed sources conflict too, not only a caller against a managed one.
@@ -3383,6 +3341,9 @@ function buildManagedStreamTextOptions(options, config) {
3383
3341
  return { ...(first ?? {}), ...(second ?? {}) };
3384
3342
  };
3385
3343
  }
3344
+ if (typeof managed.prepareStep === "function") {
3345
+ managed.prepareStep = (0, steeringContext_js_1.retainStepMessages)(managed.prepareStep);
3346
+ }
3386
3347
  return { ...managed, ...rest };
3387
3348
  }
3388
3349
  /** @internal Test hook for {@link buildManagedStreamTextOptions}. */
@@ -3398,7 +3359,7 @@ function createBoundStreamText(registry, agentSystem, agentCacheControl, agentSy
3398
3359
  }));
3399
3360
  return bound;
3400
3361
  }
3401
- function toStreamTextOptions(options) {
3362
+ function toStreamTextOptions(options, retain = true) {
3402
3363
  const agentDefaults = locals_js_1.locals.get(chatAgentManagedConfigKey);
3403
3364
  if (agentDefaults) {
3404
3365
  options = {
@@ -3605,6 +3566,9 @@ function toStreamTextOptions(options) {
3605
3566
  return resultMessages ? { messages: resultMessages } : undefined;
3606
3567
  };
3607
3568
  }
3569
+ if (retain && typeof result.prepareStep === "function") {
3570
+ result.prepareStep = (0, steeringContext_js_1.retainStepMessages)(result.prepareStep);
3571
+ }
3608
3572
  return result;
3609
3573
  }
3610
3574
  const actionTurnBrand = Symbol.for("trigger.dev/chat/actionTurn");
@@ -3650,9 +3614,16 @@ function isActionTurn(value) {
3650
3614
  * tail does not match the old message's conversion, nothing is changed and
3651
3615
  * `false` is returned so the caller can fall back to a full reconversion.
3652
3616
  */
3653
- async function replaceModelRun(lane, oldUi, newUi, tailAfter) {
3654
- const oldRun = await toModelMessages([stripProviderMetadata(oldUi)]);
3617
+ async function replaceModelRun(lane, oldUi, newUi, tailAfter, knownOldRun) {
3618
+ const oldRun = knownOldRun ?? (await toModelMessages([stripProviderMetadata(oldUi)]));
3655
3619
  const newRun = await toModelMessages([stripProviderMetadata(newUi)]);
3620
+ // A message that converts to nothing (a pending tool call with no output yet,
3621
+ // which `ignoreIncompleteToolCalls` drops) locates no run in the lane. Matching
3622
+ // an empty slice would splice the new run in without removing what the message
3623
+ // actually contributed, such as a spliced head-start partial, and the lane would
3624
+ // then carry the same tool call twice.
3625
+ if (oldRun.length === 0)
3626
+ return false;
3656
3627
  const end = lane.length - tailAfter;
3657
3628
  const start = end - oldRun.length;
3658
3629
  if (start < 0 || end > lane.length)
@@ -3732,7 +3703,7 @@ function isReadableStream(value) {
3732
3703
  * }
3733
3704
  * ```
3734
3705
  */
3735
- async function pipeChat(source, options) {
3706
+ async function pipeChat(source, options, capture) {
3736
3707
  locals_js_1.locals.set(chatPipeCountKey, (locals_js_1.locals.get(chatPipeCountKey) ?? 0) + 1);
3737
3708
  let stream;
3738
3709
  if (isUIMessageStreamable(source)) {
@@ -3761,8 +3732,18 @@ async function pipeChat(source, options) {
3761
3732
  // accepts opaque UIMessageStreamable / raw iterables whose element
3762
3733
  // type we don't know at compile time. Cast — runtime behaviour is
3763
3734
  // identical (bytes go to session.out either way).
3764
- const { waitUntilComplete } = chatStream.pipe(stream, pipeOptions);
3765
- await waitUntilComplete();
3735
+ if (capture) {
3736
+ const response = managedResponse();
3737
+ response.seed(capture.originalMessages);
3738
+ await response.pipe(stream, options?.signal);
3739
+ }
3740
+ else {
3741
+ // Raw/manual pipes intentionally do not feed the managed capture. Never
3742
+ // leave injection events waiting for finish-step chunks it cannot see.
3743
+ managedResponse().useRawPipe();
3744
+ const { waitUntilComplete } = chatStream.pipe(stream, pipeOptions);
3745
+ await waitUntilComplete();
3746
+ }
3766
3747
  }
3767
3748
  function chatCustomAgent(options) {
3768
3749
  const { clientDataSchema, onClientDataValidationError, clientDataReportErrorAt, run: userRun, ...restOptions } = options;
@@ -4015,11 +3996,13 @@ function chatAgent(options) {
4015
3996
  if (!pending || pending.length === 0)
4016
3997
  return [];
4017
3998
  locals_js_1.locals.set(chatPendingSteerKey, []);
4018
- for (const entry of pending) {
3999
+ const inlineIds = new Set((0, steeringContext_js_1.steeringMarkers)(options?.response ? [options.response] : []).flatMap((m) => m.messageIds));
4000
+ const standalone = pending.filter((entry) => !inlineIds.has(entry.ui.id));
4001
+ for (const entry of standalone) {
4019
4002
  accumulatedMessages.push(...entry.model);
4020
4003
  options?.turnNew?.push(...entry.model);
4021
4004
  }
4022
- return pending;
4005
+ return standalone;
4023
4006
  };
4024
4007
  // Accumulated UI messages for persistence. Mirrors the model accumulator
4025
4008
  // but in frontend-friendly UIMessage format (with parts, id, etc.).
@@ -4113,7 +4096,10 @@ function chatAgent(options) {
4113
4096
  });
4114
4097
  const throughId = opts.messages.at(-1)?.id ?? "";
4115
4098
  const queued = locals_js_1.locals.get(chatBackgroundQueueKey) ?? [];
4116
- const runtimeState = laneCompacted || laneInjections.length > 0 || queued.length > 0
4099
+ const transcriptIds = new Set(opts.messages.map((message) => message.id));
4100
+ const markerIds = new Set((0, steeringContext_js_1.steeringMarkers)(opts.messages).map((marker) => marker.id));
4101
+ const steering = [...(locals_js_1.locals.get(chatSteeringInjectionsKey)?.values() ?? [])].filter((entry) => markerIds.has(entry.id) || entry.messageIds.some((id) => transcriptIds.has(id)));
4102
+ const runtimeState = laneCompacted || laneInjections.length > 0 || queued.length > 0 || steering.length > 0
4117
4103
  ? {
4118
4104
  v: 1,
4119
4105
  ...(laneCompacted
@@ -4126,6 +4112,7 @@ function chatAgent(options) {
4126
4112
  : {}),
4127
4113
  ...(laneInjections.length > 0 ? { injections: laneInjections } : {}),
4128
4114
  ...(queued.length > 0 ? { queued: [...queued] } : {}),
4115
+ ...(steering.length > 0 ? { steering } : {}),
4129
4116
  }
4130
4117
  : null;
4131
4118
  if (runtimeState !== null || persistedStateSet) {
@@ -4595,6 +4582,8 @@ function chatAgent(options) {
4595
4582
  }
4596
4583
  try {
4597
4584
  const bootRuntimeState = (0, transcriptStorage_js_1.parseTranscriptRuntimeState)(bootTranscriptState);
4585
+ locals_js_1.locals.set(chatSteeringInjectionsKey, new Map((bootRuntimeState?.steering ?? []).map((entry) => [entry.id, entry])));
4586
+ locals_js_1.locals.set(chatCurrentUIMessagesKey, accumulatedUIMessages);
4598
4587
  const restored = await (0, transcriptStorage_js_1.restoreModelLane)(accumulatedUIMessages, bootRuntimeState, (messages) => toModelMessages(messages));
4599
4588
  accumulatedMessages = restored.messages;
4600
4589
  laneCompacted = restored.compacted;
@@ -4776,7 +4765,7 @@ function chatAgent(options) {
4776
4765
  const preloadResult = await messagesInput.waitWithIdleTimeout({
4777
4766
  idleTimeoutInSeconds: effectivePreloadIdleTimeout,
4778
4767
  timeout: effectivePreloadTimeout,
4779
- spanName: "waiting for first message",
4768
+ spanName: "first message",
4780
4769
  skipSuspend: exitAfterPreloadIdle,
4781
4770
  onSuspend: onChatSuspend
4782
4771
  ? async () => {
@@ -4928,7 +4917,7 @@ function chatAgent(options) {
4928
4917
  const continuationResult = await messagesInput.waitWithIdleTimeout({
4929
4918
  idleTimeoutInSeconds: effectiveIdleTimeout,
4930
4919
  timeout: effectiveTurnTimeout,
4931
- spanName: "waiting for first message (continuation)",
4920
+ spanName: "first message (continuation)",
4932
4921
  onSuspend: onChatSuspend
4933
4922
  ? async () => {
4934
4923
  await tracer_js_1.tracer.startActiveSpan("onChatSuspend()", async () => {
@@ -5042,7 +5031,9 @@ function chatAgent(options) {
5042
5031
  locals_js_1.locals.set(chatCompactionStateKey, undefined);
5043
5032
  locals_js_1.locals.set(chatSteeringQueueKey, []);
5044
5033
  locals_js_1.locals.set(chatPendingBackgroundKey, []);
5045
- locals_js_1.locals.set(chatResponsePartsKey, []);
5034
+ await locals_js_1.locals.get(chatManagedResponseKey)?.close();
5035
+ locals_js_1.locals.set(chatManagedResponseKey, undefined);
5036
+ locals_js_1.locals.set(chatManagedResponseActiveKey, true);
5046
5037
  // NOTE: chatBackgroundQueueKey is NOT reset here — messages injected
5047
5038
  // by deferred work from the previous turn's onTurnComplete need to
5048
5039
  // survive into the next turn. The queue is drained before run().
@@ -5206,7 +5197,7 @@ function chatAgent(options) {
5206
5197
  if (actionOverride) {
5207
5198
  locals_js_1.locals.set(chatOverrideMessagesKey, undefined);
5208
5199
  accumulatedUIMessages = [...actionOverride];
5209
- accumulatedMessages = await toModelMessages(actionOverride);
5200
+ accumulatedMessages = await toModelMessages(actionOverride, actionOverride);
5210
5201
  laneCompacted = false;
5211
5202
  laneInjections = [];
5212
5203
  locals_js_1.locals.set(chatCurrentUIMessagesKey, accumulatedUIMessages);
@@ -5364,7 +5355,7 @@ function chatAgent(options) {
5364
5355
  accumulatedUIMessages[accumulatedUIMessages.length - 1].role !== "user") {
5365
5356
  accumulatedUIMessages.pop();
5366
5357
  }
5367
- accumulatedMessages = await toModelMessages(accumulatedUIMessages);
5358
+ accumulatedMessages = await toModelMessages(accumulatedUIMessages, accumulatedUIMessages);
5368
5359
  laneCompacted = false;
5369
5360
  laneInjections = [];
5370
5361
  }
@@ -5416,7 +5407,7 @@ function chatAgent(options) {
5416
5407
  }
5417
5408
  if (!inPlace) {
5418
5409
  v3_1.logger.warn("chat.agent: replaced message not found at the model lane tail; reconverting the lane");
5419
- accumulatedMessages = await toModelMessages(accumulatedUIMessages);
5410
+ accumulatedMessages = await toModelMessages(accumulatedUIMessages, accumulatedUIMessages);
5420
5411
  laneCompacted = false;
5421
5412
  laneInjections = [];
5422
5413
  }
@@ -5457,10 +5448,12 @@ function chatAgent(options) {
5457
5448
  // `UIMessageStreamError: No tool invocation found`.
5458
5449
  const pendingHandoverPartial = locals_js_1.locals.get(chatHandoverPartialKey);
5459
5450
  if (pendingHandoverPartial && pendingHandoverPartial.length > 0) {
5460
- spliceHandoverPartial(accumulatedMessages, accumulatedUIMessages, {
5451
+ const spliced = spliceHandoverPartial(accumulatedMessages, accumulatedUIMessages, {
5461
5452
  partialAssistantMessage: pendingHandoverPartial,
5462
5453
  messageId: locals_js_1.locals.get(chatHandoverMessageIdKey),
5463
5454
  });
5455
+ if (spliced)
5456
+ locals_js_1.locals.set(chatHandoverSplicedRunKey, spliced);
5464
5457
  locals_js_1.locals.set(chatHandoverPartialKey, []); // consume once
5465
5458
  splicedHandoverPartial = true;
5466
5459
  }
@@ -5631,7 +5624,7 @@ function chatAgent(options) {
5631
5624
  if (turnStartOverride) {
5632
5625
  locals_js_1.locals.set(chatOverrideMessagesKey, undefined);
5633
5626
  accumulatedUIMessages = [...turnStartOverride];
5634
- accumulatedMessages = await toModelMessages(turnStartOverride);
5627
+ accumulatedMessages = await toModelMessages(turnStartOverride, turnStartOverride);
5635
5628
  laneCompacted = false;
5636
5629
  laneInjections = [];
5637
5630
  locals_js_1.locals.set(chatCurrentUIMessagesKey, accumulatedUIMessages);
@@ -5711,6 +5704,7 @@ function chatAgent(options) {
5711
5704
  capturedResponseMessage = lastUI;
5712
5705
  capturedFinishReason = "stop";
5713
5706
  }
5707
+ managedResponse().seed(accumulatedUIMessages);
5714
5708
  // Don't call userRun. Don't pipe. Skip directly
5715
5709
  // to the post-turn flow below.
5716
5710
  }
@@ -5770,7 +5764,7 @@ function chatAgent(options) {
5770
5764
  await pipeChat(tapUIMessageChunks(uiStream, turnBufferedChunks), {
5771
5765
  signal: combinedSignal,
5772
5766
  spanName: "stream response",
5773
- });
5767
+ }, { originalMessages: isActionTurn ? undefined : accumulatedUIMessages });
5774
5768
  }
5775
5769
  }
5776
5770
  catch (error) {
@@ -5798,6 +5792,10 @@ function chatAgent(options) {
5798
5792
  new Promise((r) => setTimeout(r, 2_000)),
5799
5793
  ]);
5800
5794
  }
5795
+ capturedResponseMessage =
5796
+ (await locals_js_1.locals.get(chatManagedResponseKey)?.snapshot()) ?? capturedResponseMessage;
5797
+ if (capturedResponseMessage)
5798
+ capturedPartialResponse = capturedResponseMessage;
5801
5799
  // Capture token usage from the streamText result (if available).
5802
5800
  // totalUsage is a PromiseLike that resolves after the stream is consumed.
5803
5801
  // Race with a 2s timeout — on stop-abort the AI SDK's totalUsage
@@ -5805,6 +5803,7 @@ function chatAgent(options) {
5805
5803
  // never reports final usage), which would block the turn loop
5806
5804
  // from ever firing onTurnComplete / writeTurnComplete.
5807
5805
  let turnUsage;
5806
+ let lastStepUsage;
5808
5807
  if (runResult != null &&
5809
5808
  typeof runResult.totalUsage?.then === "function") {
5810
5809
  try {
@@ -5817,6 +5816,18 @@ function chatAgent(options) {
5817
5816
  /* non-fatal — usage capture failed */
5818
5817
  }
5819
5818
  }
5819
+ const lastStepUsagePromise = runResult != null ? runResult.usage : undefined;
5820
+ if (typeof lastStepUsagePromise?.then === "function") {
5821
+ try {
5822
+ lastStepUsage = (await Promise.race([
5823
+ lastStepUsagePromise,
5824
+ new Promise((r) => setTimeout(() => r(undefined), 2_000)),
5825
+ ]));
5826
+ }
5827
+ catch {
5828
+ /* non-fatal — usage capture failed */
5829
+ }
5830
+ }
5820
5831
  if (turnUsage) {
5821
5832
  cumulativeUsage = addUsage(cumulativeUsage, turnUsage);
5822
5833
  previousTurnUsage = turnUsage;
@@ -5867,6 +5878,13 @@ function chatAgent(options) {
5867
5878
  // Check if compaction set a model-only override (preserves UI messages).
5868
5879
  // Apply compactUIMessages/compactModelMessages callbacks if configured.
5869
5880
  const modelOnlyOverride = locals_js_1.locals.get(chatOverrideModelMessagesKey);
5881
+ const responseCompaction = modelOnlyOverride
5882
+ ? locals_js_1.locals.get(chatCompactionStateKey)
5883
+ : undefined;
5884
+ // Capture the original assistant before compactUIMessages can remove it.
5885
+ const originalResponse = responseCompaction && capturedResponseMessage
5886
+ ? accumulatedUIMessages.find((m) => m.id === capturedResponseMessage?.id)
5887
+ : undefined;
5870
5888
  if (modelOnlyOverride) {
5871
5889
  const compactionSummary = locals_js_1.locals.get(chatCompactionStateKey)?.summary ?? "";
5872
5890
  const taskCompactionConfig = locals_js_1.locals.get(chatAgentCompactionKey);
@@ -5901,6 +5919,7 @@ function chatAgent(options) {
5901
5919
  // branches below, so a turn that captured no response is covered.
5902
5920
  const steerTailThisTurn = reconcilePendingSteer({
5903
5921
  turnNew: turnNewModelMessages,
5922
+ response: capturedResponseMessage,
5904
5923
  }).reduce((n, e) => n + e.model.length, 0) + reconcilePendingBackground();
5905
5924
  // Append the assistant's response (partial or complete) to the accumulator.
5906
5925
  // The onFinish callback fires even on abort/stop, so partial responses
@@ -5923,15 +5942,6 @@ function chatAgent(options) {
5923
5942
  id: (0, ai_runtime_js_1.generateId)(),
5924
5943
  };
5925
5944
  }
5926
- // Append any non-transient data parts queued via chat.response or writer.write()
5927
- const queuedParts = locals_js_1.locals.get(chatResponsePartsKey);
5928
- if (queuedParts && queuedParts.length > 0) {
5929
- capturedResponseMessage = {
5930
- ...capturedResponseMessage,
5931
- parts: [...capturedResponseMessage.parts, ...queuedParts],
5932
- };
5933
- locals_js_1.locals.set(chatResponsePartsKey, []);
5934
- }
5935
5945
  const responseHasContent = capturedResponseMessage.parts.some((part) => part.type !== "step-start");
5936
5946
  if (responseHasContent) {
5937
5947
  // Tool-approval continuations: the AI SDK reuses the trailing
@@ -5958,15 +5968,42 @@ function chatAgent(options) {
5958
5968
  // rationale (TRI-9137).
5959
5969
  recordToolCallIdsFromMessage(capturedResponseMessage);
5960
5970
  try {
5971
+ const responseForModel = (0, compactionResponse_js_1.responseAfterCompaction)(capturedResponseMessage, responseCompaction?.baseResponseStepCount, originalResponse);
5972
+ // Preserve the complete persistence response, including same-ID
5973
+ // replacements whose old tool parts can contain new results.
5974
+ // Convert prefix and suffix separately so each tool output is
5975
+ // converted once, while only the suffix enters model context.
5976
+ const responsePrefixMessages = responseCompaction
5977
+ ? await toModelMessages([
5978
+ stripProviderMetadata({
5979
+ ...capturedResponseMessage,
5980
+ parts: capturedResponseMessage.parts.slice(0, capturedResponseMessage.parts.length -
5981
+ responseForModel.parts.length),
5982
+ }),
5983
+ ])
5984
+ : [];
5961
5985
  const responseModelMessages = await toModelMessages([
5962
- stripProviderMetadata(capturedResponseMessage),
5986
+ stripProviderMetadata(responseForModel),
5963
5987
  ]);
5964
- if (existingIdx !== -1) {
5988
+ if (responseCompaction) {
5989
+ // The summary already replaced the original response, including
5990
+ // a same-ID approval/handover prefix. Replacing its old model run
5991
+ // would miss and fall back to the full, uncompacted UI history.
5992
+ accumulatedMessages.push(...responseModelMessages);
5993
+ locals_js_1.locals.set(chatHandoverSplicedRunKey, undefined);
5994
+ }
5995
+ else if (existingIdx !== -1) {
5996
+ const spliced = locals_js_1.locals.get(chatHandoverSplicedRunKey);
5997
+ const splicedRun = spliced && previousAtIdx && spliced.id === previousAtIdx.id
5998
+ ? spliced.run
5999
+ : undefined;
5965
6000
  const ok = previousAtIdx !== undefined &&
5966
- (await replaceModelRun(accumulatedMessages, previousAtIdx, capturedResponseMessage, steerTailThisTurn));
6001
+ (await replaceModelRun(accumulatedMessages, previousAtIdx, capturedResponseMessage, steerTailThisTurn, splicedRun));
6002
+ if (splicedRun)
6003
+ locals_js_1.locals.set(chatHandoverSplicedRunKey, undefined);
5967
6004
  if (!ok) {
5968
6005
  v3_1.logger.warn("chat.agent: replaced response not found at the model lane tail; reconverting the lane");
5969
- accumulatedMessages = await toModelMessages(accumulatedUIMessages);
6006
+ accumulatedMessages = await toModelMessages(accumulatedUIMessages, accumulatedUIMessages);
5970
6007
  laneCompacted = false;
5971
6008
  laneInjections = [];
5972
6009
  }
@@ -5974,7 +6011,7 @@ function chatAgent(options) {
5974
6011
  else {
5975
6012
  accumulatedMessages.push(...responseModelMessages);
5976
6013
  }
5977
- turnNewModelMessages.push(...responseModelMessages);
6014
+ turnNewModelMessages.push(...responsePrefixMessages, ...responseModelMessages);
5978
6015
  }
5979
6016
  catch {
5980
6017
  // Conversion failed — skip accumulation for this turn
@@ -5984,22 +6021,6 @@ function chatAgent(options) {
5984
6021
  responseWasSkipped = true;
5985
6022
  }
5986
6023
  }
5987
- // If there's no captured response (manual pipe mode) but there are
5988
- // queued data parts, create a minimal response message to hold them.
5989
- if (!capturedResponseMessage) {
5990
- const remainingParts = locals_js_1.locals.get(chatResponsePartsKey);
5991
- if (remainingParts && remainingParts.length > 0) {
5992
- capturedResponseMessage = {
5993
- id: (0, ai_runtime_js_1.generateId)(),
5994
- role: "assistant",
5995
- parts: [...remainingParts],
5996
- };
5997
- locals_js_1.locals.set(chatResponsePartsKey, []);
5998
- accumulatedUIMessages.push(capturedResponseMessage);
5999
- turnNewUIMessages.push(capturedResponseMessage);
6000
- locals_js_1.locals.set(chatCurrentUIMessagesKey, accumulatedUIMessages);
6001
- }
6002
- }
6003
6024
  if (capturedResponseMessage) {
6004
6025
  responseCommitted = true;
6005
6026
  capturedPartialResponse = capturedResponseMessage;
@@ -6023,12 +6044,14 @@ function chatAgent(options) {
6023
6044
  const outerCompaction = locals_js_1.locals.get(chatAgentCompactionKey);
6024
6045
  const innerCompactionState = locals_js_1.locals.get(chatCompactionStateKey);
6025
6046
  if (outerCompaction && !innerCompactionState && turnUsage && !wasStopped) {
6047
+ const contextUsage = lastStepUsage ?? turnUsage;
6026
6048
  const shouldTrigger = await outerCompaction.shouldCompact({
6027
6049
  messages: accumulatedMessages,
6028
- totalTokens: turnUsage.totalTokens,
6029
- inputTokens: turnUsage.inputTokens,
6030
- outputTokens: turnUsage.outputTokens,
6031
- usage: turnUsage,
6050
+ totalTokens: contextUsage.totalTokens,
6051
+ inputTokens: contextUsage.inputTokens,
6052
+ outputTokens: contextUsage.outputTokens,
6053
+ usage: contextUsage,
6054
+ turnUsage,
6032
6055
  totalUsage: cumulativeUsage,
6033
6056
  chatId: currentWirePayload.chatId,
6034
6057
  turn,
@@ -6159,6 +6182,8 @@ function chatAgent(options) {
6159
6182
  totalUsage: cumulativeUsage,
6160
6183
  finishReason: capturedFinishReason,
6161
6184
  };
6185
+ const beforeHookRevision = locals_js_1.locals.get(chatManagedResponseKey)?.revision ?? 0;
6186
+ let beforeHookHistoryEdited = false;
6162
6187
  // Fire onBeforeTurnComplete — stream is still open so the hook
6163
6188
  // can write custom chunks to the frontend (e.g. compaction progress).
6164
6189
  if (onBeforeTurnComplete) {
@@ -6169,9 +6194,10 @@ function chatAgent(options) {
6169
6194
  // Check if the hook replaced messages (compaction or chat.history)
6170
6195
  const override = locals_js_1.locals.get(chatOverrideMessagesKey);
6171
6196
  if (override) {
6197
+ beforeHookHistoryEdited = true;
6172
6198
  locals_js_1.locals.set(chatOverrideMessagesKey, undefined);
6173
6199
  accumulatedUIMessages = [...override];
6174
- accumulatedMessages = await toModelMessages(override);
6200
+ accumulatedMessages = await toModelMessages(override, override);
6175
6201
  laneCompacted = false;
6176
6202
  laneInjections = [];
6177
6203
  locals_js_1.locals.set(chatCurrentUIMessagesKey, accumulatedUIMessages);
@@ -6188,35 +6214,33 @@ function chatAgent(options) {
6188
6214
  },
6189
6215
  });
6190
6216
  }
6191
- // Drain any late response parts added during onBeforeTurnComplete
6192
- const lateParts = locals_js_1.locals.get(chatResponsePartsKey);
6193
- if (lateParts && lateParts.length > 0 && capturedResponseMessage) {
6194
- const idx = accumulatedUIMessages.findIndex((m) => m.id === capturedResponseMessage.id);
6195
- if (idx !== -1) {
6196
- const msg = accumulatedUIMessages[idx];
6197
- accumulatedUIMessages[idx] = {
6198
- ...msg,
6199
- parts: [...(msg.parts ?? []), ...lateParts],
6200
- };
6201
- capturedResponseMessage = accumulatedUIMessages[idx];
6202
- capturedPartialResponse = capturedResponseMessage;
6203
- turnCompleteEvent.responseMessage = capturedResponseMessage;
6204
- turnCompleteEvent.uiMessages = accumulatedUIMessages;
6205
- locals_js_1.locals.set(chatCurrentUIMessagesKey, accumulatedUIMessages);
6206
- }
6207
- else if (responseWasSkipped) {
6208
- capturedResponseMessage = {
6209
- ...capturedResponseMessage,
6210
- parts: [...(capturedResponseMessage.parts ?? []), ...lateParts],
6211
- };
6212
- accumulatedUIMessages.push(capturedResponseMessage);
6213
- turnNewUIMessages.push(capturedResponseMessage);
6214
- capturedPartialResponse = capturedResponseMessage;
6215
- turnCompleteEvent.responseMessage = capturedResponseMessage;
6217
+ const editedResponse = beforeHookHistoryEdited && capturedResponseMessage
6218
+ ? accumulatedUIMessages.find((message) => message.id === capturedResponseMessage?.id)
6219
+ : undefined;
6220
+ const finalManagedResponse = await locals_js_1.locals
6221
+ .get(chatManagedResponseKey)
6222
+ ?.snapshot(editedResponse
6223
+ ? { message: editedResponse, from: beforeHookRevision }
6224
+ : undefined);
6225
+ if (finalManagedResponse?.parts.some((part) => part.type !== "step-start")) {
6226
+ const idx = accumulatedUIMessages.findIndex((m) => m.id === finalManagedResponse.id);
6227
+ const finalized = (wasStopped ? cleanupAbortedParts(finalManagedResponse) : finalManagedResponse);
6228
+ if (idx !== -1 || responseWasSkipped || !capturedResponseMessage) {
6229
+ if (idx !== -1)
6230
+ accumulatedUIMessages[idx] = finalized;
6231
+ else
6232
+ accumulatedUIMessages.push(finalized);
6233
+ const deltaIdx = turnNewUIMessages.findIndex((m) => m.id === finalized.id);
6234
+ if (deltaIdx !== -1)
6235
+ turnNewUIMessages[deltaIdx] = finalized;
6236
+ else
6237
+ turnNewUIMessages.push(finalized);
6238
+ capturedResponseMessage = finalized;
6239
+ capturedPartialResponse = finalized;
6240
+ turnCompleteEvent.responseMessage = finalized;
6216
6241
  turnCompleteEvent.uiMessages = accumulatedUIMessages;
6217
6242
  locals_js_1.locals.set(chatCurrentUIMessagesKey, accumulatedUIMessages);
6218
6243
  }
6219
- locals_js_1.locals.set(chatResponsePartsKey, []);
6220
6244
  }
6221
6245
  settleRecoveredTurn(currentWirePayload);
6222
6246
  // Write turn-complete control chunk — closes the frontend stream.
@@ -6233,7 +6257,7 @@ function chatAgent(options) {
6233
6257
  if (turnCompleteOverride) {
6234
6258
  locals_js_1.locals.set(chatOverrideMessagesKey, undefined);
6235
6259
  accumulatedUIMessages = [...turnCompleteOverride];
6236
- accumulatedMessages = await toModelMessages(turnCompleteOverride);
6260
+ accumulatedMessages = await toModelMessages(turnCompleteOverride, turnCompleteOverride);
6237
6261
  laneCompacted = false;
6238
6262
  laneInjections = [];
6239
6263
  locals_js_1.locals.set(chatCurrentUIMessagesKey, accumulatedUIMessages);
@@ -6353,7 +6377,7 @@ function chatAgent(options) {
6353
6377
  const next = await messagesInput.waitWithIdleTimeout({
6354
6378
  idleTimeoutInSeconds: effectiveIdleTimeout,
6355
6379
  timeout: effectiveTurnTimeout,
6356
- spanName: "waiting for next message",
6380
+ spanName: "next message",
6357
6381
  onSuspend: onChatSuspend
6358
6382
  ? async () => {
6359
6383
  await tracer_js_1.tracer.startActiveSpan("onChatSuspend()", async () => {
@@ -6469,7 +6493,8 @@ function chatAgent(options) {
6469
6493
  !accumulatedUIMessages.some((m) => m.id === erroredWireMessage.id)
6470
6494
  ? [...accumulatedUIMessages, erroredWireMessage]
6471
6495
  : accumulatedUIMessages;
6472
- let partialResponse = capturedPartialResponse ??
6496
+ let partialResponse = (await locals_js_1.locals.get(chatManagedResponseKey)?.snapshot()) ??
6497
+ capturedPartialResponse ??
6473
6498
  (await assemblePartialFromChunks(turnBufferedChunks));
6474
6499
  if (partialResponse) {
6475
6500
  partialResponse = cleanupAbortedParts(partialResponse);
@@ -6477,24 +6502,21 @@ function chatAgent(options) {
6477
6502
  let partialIdx = partialResponse?.id
6478
6503
  ? erroredUIMessages.findIndex((m) => m.id === partialResponse.id)
6479
6504
  : -1;
6480
- if (partialResponse && capturedPartialResponse === undefined && partialIdx !== -1) {
6505
+ if (partialResponse &&
6506
+ capturedPartialResponse === undefined &&
6507
+ !locals_js_1.locals.get(chatManagedResponseKey)?.continues(partialResponse.id) &&
6508
+ partialIdx !== -1) {
6481
6509
  partialResponse = undefined;
6482
6510
  partialIdx = -1;
6483
6511
  }
6484
6512
  if (partialResponse && !partialResponse.id) {
6485
6513
  partialResponse = { ...partialResponse, id: (0, ai_runtime_js_1.generateId)() };
6486
6514
  }
6487
- if (partialResponse && !responseCommitted) {
6488
- const queuedParts = locals_js_1.locals.get(chatResponsePartsKey);
6489
- if (queuedParts && queuedParts.length > 0) {
6490
- partialResponse = {
6491
- ...partialResponse,
6492
- parts: [...partialResponse.parts, ...queuedParts],
6493
- };
6494
- locals_js_1.locals.set(chatResponsePartsKey, []);
6495
- }
6496
- }
6497
6515
  const includePartial = partialResponse != null && !responseCommitted;
6516
+ // What the stream left behind, by content. After `onTurnComplete` the
6517
+ // partial is still unfinished only if the message under its id is
6518
+ // byte-for-byte this: a clone keeps it partial, an edit finishes it.
6519
+ const partialFingerprint = includePartial && partialResponse ? (0, transcriptStorage_js_1.fingerprintMessage)(partialResponse) : undefined;
6498
6520
  let erroredUIMessagesWithPartial = !includePartial
6499
6521
  ? erroredUIMessages
6500
6522
  : partialIdx === -1
@@ -6522,7 +6544,7 @@ function chatAgent(options) {
6522
6544
  };
6523
6545
  let erroredNewUIMessages = buildErroredNew();
6524
6546
  let erroredNewModelMessages = [];
6525
- const reconciledSteer = reconcilePendingSteer();
6547
+ const reconciledSteer = reconcilePendingSteer({ response: partialResponse });
6526
6548
  const backgroundTailThisTurn = reconcilePendingBackground();
6527
6549
  if (!responseCommitted) {
6528
6550
  try {
@@ -6533,14 +6555,7 @@ function chatAgent(options) {
6533
6555
  * the model received it (what `prepare` produced), matching the
6534
6556
  * lane. The wire message and partial are converted as before.
6535
6557
  */
6536
- const steerModelById = new Map(reconciledSteer.map((e) => [e.ui.id, e.model]));
6537
- for (const m of erroredNewUIMessages) {
6538
- const recorded = steerModelById.get(m.id);
6539
- if (recorded)
6540
- erroredNewModelMessages.push(...recorded);
6541
- else
6542
- erroredNewModelMessages.push(...(await toModelMessages([stripProviderMetadata(m)])));
6543
- }
6558
+ erroredNewModelMessages = await toModelMessages(erroredNewUIMessages.map(stripProviderMetadata));
6544
6559
  }
6545
6560
  if (erroredUIMessagesWithPartial !== accumulatedUIMessages) {
6546
6561
  if (partialIdx === -1) {
@@ -6552,7 +6567,7 @@ function chatAgent(options) {
6552
6567
  backgroundTailThisTurn);
6553
6568
  if (!ok) {
6554
6569
  v3_1.logger.warn("chat.agent: replaced partial not found at the model lane tail; reconverting the lane");
6555
- accumulatedMessages = await toModelMessages(erroredUIMessagesWithPartial);
6570
+ accumulatedMessages = await toModelMessages(erroredUIMessagesWithPartial, erroredUIMessagesWithPartial);
6556
6571
  laneCompacted = false;
6557
6572
  laneInjections = [];
6558
6573
  }
@@ -6567,6 +6582,11 @@ function chatAgent(options) {
6567
6582
  erroredNewUIMessages = buildErroredNew().filter((m) => m !== partialResponse);
6568
6583
  }
6569
6584
  }
6585
+ // An earlier hook that set the history and then threw (which is one way
6586
+ // to get here) left its abandoned edit pending. Discard it before the
6587
+ // failed turn continues, so neither the error-path `onTurnComplete`
6588
+ // below nor the next turn's history reads mistake it for a real edit.
6589
+ locals_js_1.locals.set(chatOverrideMessagesKey, undefined);
6570
6590
  if (onTurnComplete) {
6571
6591
  try {
6572
6592
  await tracer_js_1.tracer.startActiveSpan("onTurnComplete()", async () => {
@@ -6595,6 +6615,23 @@ function chatAgent(options) {
6595
6615
  error: turnError,
6596
6616
  lastEventId: errorTurnCompleteResult?.lastEventId,
6597
6617
  });
6618
+ // The hook may edit the history here too (a failure record, a
6619
+ // card the turn left open). Honour it the way the success path
6620
+ // does, so the edit reaches the accumulator and the save below.
6621
+ const errorTurnOverride = locals_js_1.locals.get(chatOverrideMessagesKey);
6622
+ if (errorTurnOverride) {
6623
+ locals_js_1.locals.set(chatOverrideMessagesKey, undefined);
6624
+ // Convert first: a rejected conversion (a tool's `toModelOutput`
6625
+ // can throw) must leave every lane on the history it had.
6626
+ const overrideUIMessages = [...errorTurnOverride];
6627
+ const overrideModelMessages = await toModelMessages(errorTurnOverride, errorTurnOverride);
6628
+ erroredUIMessagesWithPartial = overrideUIMessages;
6629
+ accumulatedUIMessages = overrideUIMessages;
6630
+ accumulatedMessages = overrideModelMessages;
6631
+ laneCompacted = false;
6632
+ laneInjections = [];
6633
+ locals_js_1.locals.set(chatCurrentUIMessagesKey, accumulatedUIMessages);
6634
+ }
6598
6635
  }, {
6599
6636
  attributes: {
6600
6637
  [v3_1.SemanticInternalAttributes.STYLE_ICON]: "task-hook-onComplete",
@@ -6608,6 +6645,7 @@ function chatAgent(options) {
6608
6645
  catch {
6609
6646
  // A throwing onTurnComplete on the error path must not crash
6610
6647
  // the run — keep the conversation alive for the next message.
6648
+ locals_js_1.locals.set(chatOverrideMessagesKey, undefined);
6611
6649
  }
6612
6650
  }
6613
6651
  // Persist a snapshot so the failed turn's user message isn't
@@ -6625,7 +6663,10 @@ function chatAgent(options) {
6625
6663
  trigger: storageTrigger(currentWirePayload.trigger),
6626
6664
  clientData: turnClientData,
6627
6665
  lastOutEventId: lastSnapshotOutEventId,
6628
- nonFinalIds: includePartial && partialResponse ? new Set([partialResponse.id]) : undefined,
6666
+ // The partial is non-final only while the message under its id is
6667
+ // still what the stream left behind. A hook that replaced it (a
6668
+ // closed card, a finished body) produced a final message.
6669
+ nonFinalIds: partialStillUnfinished(partialResponse, partialFingerprint, erroredUIMessagesWithPartial),
6629
6670
  });
6630
6671
  }
6631
6672
  catch (error) {
@@ -6660,7 +6701,7 @@ function chatAgent(options) {
6660
6701
  const next = await messagesInput.waitWithIdleTimeout({
6661
6702
  idleTimeoutInSeconds: effectiveIdleTimeout,
6662
6703
  timeout: effectiveTurnTimeout,
6663
- spanName: "waiting for next message (after error)",
6704
+ spanName: "next message (after error)",
6664
6705
  });
6665
6706
  if (!next.ok) {
6666
6707
  return; // Timed out — end run gracefully
@@ -6676,6 +6717,11 @@ function chatAgent(options) {
6676
6717
  }
6677
6718
  finally {
6678
6719
  turnMsgSub?.off();
6720
+ locals_js_1.locals.set(chatManagedResponseActiveKey, false);
6721
+ await locals_js_1.locals
6722
+ .get(chatManagedResponseKey)
6723
+ ?.close()
6724
+ .catch(() => { });
6679
6725
  }
6680
6726
  }
6681
6727
  }
@@ -7645,7 +7691,7 @@ async function pipeChatAndCapture(source, options) {
7645
7691
  await pipeChat(tappedStream, {
7646
7692
  signal: options?.signal,
7647
7693
  spanName: options?.spanName ?? "stream response",
7648
- });
7694
+ }, { originalMessages: options?.originalMessages });
7649
7695
  // The pipe can drain cleanly on a stop — the source stream just ends
7650
7696
  // early — so classify by the signal rather than relying on a throw.
7651
7697
  if (options?.signal?.aborted) {
@@ -7671,6 +7717,9 @@ async function pipeChatAndCapture(source, options) {
7671
7717
  if (!captured && bufferedChunks.length > 0) {
7672
7718
  captured = await assemblePartialFromChunks(bufferedChunks);
7673
7719
  }
7720
+ captured = (await locals_js_1.locals.get(chatManagedResponseKey)?.snapshot()) ?? captured;
7721
+ if (!locals_js_1.locals.get(chatTurnContextKey))
7722
+ await locals_js_1.locals.get(chatManagedResponseKey)?.close();
7674
7723
  return {
7675
7724
  message: captured,
7676
7725
  status,
@@ -7700,8 +7749,11 @@ class ChatMessageAccumulator {
7700
7749
  modelMessages = [];
7701
7750
  uiMessages = [];
7702
7751
  _compaction;
7752
+ /** The run a spliced head-start partial contributed, until its response replaces it. */
7753
+ _handoverRun;
7703
7754
  _pendingMessages;
7704
7755
  _steeringQueue = [];
7756
+ _pendingSteer = [];
7705
7757
  constructor(options) {
7706
7758
  this._compaction = options?.compaction;
7707
7759
  this._pendingMessages = options?.pendingMessages;
@@ -7743,7 +7795,7 @@ class ChatMessageAccumulator {
7743
7795
  * `consumeHandover` for the wait+seed+apply convenience.
7744
7796
  */
7745
7797
  applyHandover(signal) {
7746
- spliceHandoverPartial(this.modelMessages, this.uiMessages, signal);
7798
+ this._handoverRun = spliceHandoverPartial(this.modelMessages, this.uiMessages, signal);
7747
7799
  }
7748
7800
  /**
7749
7801
  * One-call `chat.headStart` handover for a custom-agent loop: waits for the
@@ -7774,6 +7826,19 @@ class ChatMessageAccumulator {
7774
7826
  return { isFinal: signal.isFinal, skipped: false };
7775
7827
  }
7776
7828
  async addResponse(response) {
7829
+ const inlineIds = new Set((0, steeringContext_js_1.steeringMarkers)([response]).flatMap((m) => m.messageIds));
7830
+ for (const entry of this._pendingSteer.splice(0)) {
7831
+ if (!inlineIds.has(entry.ui.id))
7832
+ continue;
7833
+ // absorbSteering is public and updates context immediately. Move just
7834
+ // those exact appended objects into the response's chronological run;
7835
+ // never rebuild a possibly compacted lane from the UI transcript.
7836
+ for (const message of entry.model) {
7837
+ const index = this.modelMessages.lastIndexOf(message);
7838
+ if (index !== -1)
7839
+ this.modelMessages.splice(index, 1);
7840
+ }
7841
+ }
7777
7842
  if (!response.id) {
7778
7843
  response = { ...response, id: (0, ai_runtime_js_1.generateId)() };
7779
7844
  }
@@ -7786,8 +7851,13 @@ class ChatMessageAccumulator {
7786
7851
  if (existingIdx !== -1) {
7787
7852
  const previous = this.uiMessages[existingIdx];
7788
7853
  this.uiMessages[existingIdx] = response;
7854
+ const handoverRun = this._handoverRun && this._handoverRun.id === previous.id
7855
+ ? this._handoverRun.run
7856
+ : undefined;
7857
+ if (handoverRun)
7858
+ this._handoverRun = undefined;
7789
7859
  try {
7790
- if (!(await replaceModelRun(this.modelMessages, previous, response, 0))) {
7860
+ if (!(await replaceModelRun(this.modelMessages, previous, response, 0, handoverRun))) {
7791
7861
  this.modelMessages = await toModelMessages(this.uiMessages.map((m) => stripProviderMetadata(m)));
7792
7862
  }
7793
7863
  }
@@ -7845,7 +7915,10 @@ class ChatMessageAccumulator {
7845
7915
  this.uiMessages.push(...fresh);
7846
7916
  // Record what the model received. Only when the whole batch is new is
7847
7917
  // `injected` known to describe exactly these messages.
7848
- this.modelMessages.push(...(injected && fresh.length === claimed.length ? injected : await toModelMessages(fresh)));
7918
+ const model = injected && fresh.length === claimed.length ? injected : await toModelMessages(fresh);
7919
+ this.modelMessages.push(...model);
7920
+ for (const ui of fresh)
7921
+ this._pendingSteer.push({ ui, model: modelFormOf(ui, fresh, model) });
7849
7922
  }
7850
7923
  /**
7851
7924
  * Get and clear unconsumed steering messages.
@@ -7865,7 +7938,7 @@ class ChatMessageAccumulator {
7865
7938
  const comp = this._compaction;
7866
7939
  const pm = this._pendingMessages;
7867
7940
  const queue = this._steeringQueue;
7868
- return async ({ messages, steps }) => {
7941
+ return (0, steeringContext_js_1.retainStepMessages)(async ({ messages, steps }) => {
7869
7942
  let resultMessages;
7870
7943
  // 1. Compaction
7871
7944
  if (comp) {
@@ -7878,7 +7951,7 @@ class ChatMessageAccumulator {
7878
7951
  }
7879
7952
  }
7880
7953
  // 2. Pending message injection
7881
- if (pm && queue.length > 0) {
7954
+ if (pm) {
7882
7955
  const { injected, claimed } = await drainSteeringQueue(pm, resultMessages ?? messages, steps, queue);
7883
7956
  await this.absorbSteering(claimed, injected);
7884
7957
  if (injected.length > 0) {
@@ -7886,12 +7959,14 @@ class ChatMessageAccumulator {
7886
7959
  }
7887
7960
  }
7888
7961
  return resultMessages ? { messages: resultMessages } : undefined;
7889
- };
7962
+ });
7890
7963
  }
7891
7964
  /**
7892
7965
  * Run outer-loop compaction if needed. Call after adding the response
7893
- * and capturing usage. Applies `compactModelMessages` and `compactUIMessages`
7894
- * callbacks if configured.
7966
+ * and capturing usage. Pass the LAST step's usage (`result.usage`), which is
7967
+ * the context the model held on its final call; `result.totalUsage` sums every
7968
+ * step of a tool-using turn and belongs in `context.turnUsage`. Applies
7969
+ * `compactModelMessages` and `compactUIMessages` callbacks if configured.
7895
7970
  *
7896
7971
  * @returns `true` if compaction was performed, `false` otherwise.
7897
7972
  */
@@ -7904,6 +7979,7 @@ class ChatMessageAccumulator {
7904
7979
  inputTokens: usage.inputTokens,
7905
7980
  outputTokens: usage.outputTokens,
7906
7981
  usage,
7982
+ turnUsage: context?.turnUsage,
7907
7983
  totalUsage: context?.totalUsage,
7908
7984
  chatId: context?.chatId,
7909
7985
  turn: context?.turn,
@@ -8136,8 +8212,8 @@ function createChatSession(payload, options) {
8136
8212
  idleTimeoutInSeconds: sessionIdleTimeoutOpt ?? currentPayload.idleTimeoutInSeconds ?? 30,
8137
8213
  timeout,
8138
8214
  spanName: currentPayload.trigger === "preload"
8139
- ? "waiting for first message"
8140
- : "waiting for first message (continuation)",
8215
+ ? "first message"
8216
+ : "first message (continuation)",
8141
8217
  });
8142
8218
  if (!result.ok || runSignal.aborted) {
8143
8219
  stop.cleanup();
@@ -8174,7 +8250,7 @@ function createChatSession(payload, options) {
8174
8250
  const next = await messagesInput.waitWithIdleTimeout({
8175
8251
  idleTimeoutInSeconds,
8176
8252
  timeout,
8177
- spanName: "waiting for next message",
8253
+ spanName: "next message",
8178
8254
  });
8179
8255
  if (!next.ok || runSignal.aborted) {
8180
8256
  stop.cleanup();
@@ -8195,7 +8271,9 @@ function createChatSession(payload, options) {
8195
8271
  // Reset stop signal for this turn
8196
8272
  stop.reset();
8197
8273
  // Reset per-turn state
8198
- locals_js_1.locals.set(chatResponsePartsKey, []);
8274
+ await locals_js_1.locals.get(chatManagedResponseKey)?.close();
8275
+ locals_js_1.locals.set(chatManagedResponseKey, undefined);
8276
+ locals_js_1.locals.set(chatManagedResponseActiveKey, true);
8199
8277
  // Set up steering queue and pending messages config in locals
8200
8278
  // so toStreamTextOptions() auto-injects prepareStep for steering
8201
8279
  const turnSteeringQueue = [];
@@ -8340,11 +8418,6 @@ function createChatSession(payload, options) {
8340
8418
  if (captured.status === "error") {
8341
8419
  if (captured.message) {
8342
8420
  const partial = cleanupAbortedParts(captured.message);
8343
- const queuedParts = locals_js_1.locals.get(chatResponsePartsKey);
8344
- if (queuedParts && queuedParts.length > 0) {
8345
- partial.parts = [...(partial.parts ?? []), ...queuedParts];
8346
- locals_js_1.locals.set(chatResponsePartsKey, []);
8347
- }
8348
8421
  await accumulator.addResponse(partial);
8349
8422
  }
8350
8423
  throw captured.error;
@@ -8360,31 +8433,14 @@ function createChatSession(payload, options) {
8360
8433
  const cleaned = stop.signal.aborted && !runSignal.aborted
8361
8434
  ? cleanupAbortedParts(response)
8362
8435
  : response;
8363
- // Append any non-transient data parts queued via chat.response or writer.write()
8364
- const queuedParts = locals_js_1.locals.get(chatResponsePartsKey);
8365
- if (queuedParts && queuedParts.length > 0) {
8366
- cleaned.parts = [...(cleaned.parts ?? []), ...queuedParts];
8367
- locals_js_1.locals.set(chatResponsePartsKey, []);
8368
- }
8369
8436
  await accumulator.addResponse(cleaned);
8370
8437
  }
8371
- else {
8372
- // No response (manual pipe mode) but there are queued data parts
8373
- const queuedParts = locals_js_1.locals.get(chatResponsePartsKey);
8374
- if (queuedParts && queuedParts.length > 0) {
8375
- await accumulator.addResponse({
8376
- id: (0, ai_runtime_js_1.generateId)(),
8377
- role: "assistant",
8378
- parts: queuedParts,
8379
- });
8380
- locals_js_1.locals.set(chatResponsePartsKey, []);
8381
- }
8382
- }
8383
8438
  // Capture token usage from the streamText result. Race with a 2s
8384
8439
  // timeout — on stop-abort the AI SDK's totalUsage promise can hang
8385
8440
  // indefinitely, which would wedge the turn loop (same guard as
8386
8441
  // chat.agent's turn loop).
8387
8442
  let turnUsage;
8443
+ let lastStepUsage;
8388
8444
  if (typeof source.totalUsage?.then === "function") {
8389
8445
  try {
8390
8446
  const usage = (await Promise.race([
@@ -8401,14 +8457,28 @@ function createChatSession(payload, options) {
8401
8457
  /* non-fatal */
8402
8458
  }
8403
8459
  }
8460
+ const lastStepUsagePromise = source.usage;
8461
+ if (typeof lastStepUsagePromise?.then === "function") {
8462
+ try {
8463
+ lastStepUsage = (await Promise.race([
8464
+ lastStepUsagePromise,
8465
+ new Promise((r) => setTimeout(() => r(undefined), 2_000)),
8466
+ ]));
8467
+ }
8468
+ catch {
8469
+ /* non-fatal */
8470
+ }
8471
+ }
8404
8472
  // Outer-loop compaction (same logic as chat.agent)
8405
8473
  if (sessionCompaction && turnUsage && !turnObj.stopped) {
8474
+ const contextUsage = lastStepUsage ?? turnUsage;
8406
8475
  const shouldTrigger = await sessionCompaction.shouldCompact({
8407
8476
  messages: accumulator.modelMessages,
8408
- totalTokens: turnUsage.totalTokens,
8409
- inputTokens: turnUsage.inputTokens,
8410
- outputTokens: turnUsage.outputTokens,
8411
- usage: turnUsage,
8477
+ totalTokens: contextUsage.totalTokens,
8478
+ inputTokens: contextUsage.inputTokens,
8479
+ outputTokens: contextUsage.outputTokens,
8480
+ usage: contextUsage,
8481
+ turnUsage,
8412
8482
  totalUsage: cumulativeUsage,
8413
8483
  chatId: currentPayload.chatId,
8414
8484
  turn,
@@ -8455,15 +8525,6 @@ function createChatSession(payload, options) {
8455
8525
  return response;
8456
8526
  },
8457
8527
  async addResponse(response) {
8458
- // Append any non-transient data parts queued via chat.response or writer.write()
8459
- const queuedParts = locals_js_1.locals.get(chatResponsePartsKey);
8460
- if (queuedParts && queuedParts.length > 0) {
8461
- response = {
8462
- ...response,
8463
- parts: [...(response.parts ?? []), ...queuedParts],
8464
- };
8465
- locals_js_1.locals.set(chatResponsePartsKey, []);
8466
- }
8467
8528
  await accumulator.addResponse(response);
8468
8529
  },
8469
8530
  async done() {
@@ -8475,7 +8536,7 @@ function createChatSession(payload, options) {
8475
8536
  const hasPending = !!sessionPendingMessages;
8476
8537
  if (!hasCompaction && !hasPending)
8477
8538
  return undefined;
8478
- return async ({ messages: stepMsgs, steps, }) => {
8539
+ return (0, steeringContext_js_1.retainStepMessages)(async ({ messages: stepMsgs, steps, }) => {
8479
8540
  let resultMessages;
8480
8541
  if (sessionCompaction) {
8481
8542
  const compactResult = await chatCompact(stepMsgs, steps, {
@@ -8494,7 +8555,7 @@ function createChatSession(payload, options) {
8494
8555
  }
8495
8556
  }
8496
8557
  return resultMessages ? { messages: resultMessages } : undefined;
8497
- };
8558
+ });
8498
8559
  },
8499
8560
  };
8500
8561
  return { done: false, value: turnObj };
@@ -8762,6 +8823,10 @@ function createChatStartSessionAction(taskId, options) {
8762
8823
  const clientDataMetadata = params.clientData !== undefined ? { metadata: params.clientData } : {};
8763
8824
  const maxAttempts = params.triggerConfig?.maxAttempts ?? options?.triggerConfig?.maxAttempts;
8764
8825
  const maxDuration = params.triggerConfig?.maxDuration ?? options?.triggerConfig?.maxDuration;
8826
+ const concurrency = params.triggerConfig?.concurrency !== undefined
8827
+ ? params.triggerConfig.concurrency
8828
+ : options?.triggerConfig?.concurrency;
8829
+ const concurrencyKey = params.triggerConfig?.concurrencyKey ?? options?.triggerConfig?.concurrencyKey;
8765
8830
  const idleTimeoutInSeconds = params.triggerConfig?.idleTimeoutInSeconds ?? options?.triggerConfig?.idleTimeoutInSeconds;
8766
8831
  // Only `undefined` means "not supplied": a per-call `null` (opt out) has to beat a pinning
8767
8832
  // action default, which neither truthiness nor `??` would allow.
@@ -8783,6 +8848,8 @@ function createChatStartSessionAction(taskId, options) {
8783
8848
  ...(options?.triggerConfig?.queue || params.triggerConfig?.queue
8784
8849
  ? { queue: params.triggerConfig?.queue ?? options?.triggerConfig?.queue }
8785
8850
  : {}),
8851
+ ...(concurrency !== undefined ? (0, concurrency_shared_js_1.triggerConcurrencyBody)(concurrency) : {}),
8852
+ ...(concurrencyKey !== undefined ? { concurrencyKey } : {}),
8786
8853
  tags,
8787
8854
  ...(maxAttempts !== undefined ? { maxAttempts } : {}),
8788
8855
  ...(maxDuration !== undefined ? { maxDuration } : {}),
@@ -9130,6 +9197,10 @@ exports.chat = {
9130
9197
  * @internal
9131
9198
  */
9132
9199
  async function writeTurnCompleteChunk(_chatId, publicAccessToken) {
9200
+ locals_js_1.locals.set(chatManagedResponseActiveKey, false);
9201
+ const response = locals_js_1.locals.get(chatManagedResponseKey);
9202
+ if (response)
9203
+ await response.close();
9133
9204
  const session = getChatSession();
9134
9205
  // A handover-prepare boot claims the handover kinds so a signal arriving
9135
9206
  // before `waitForHandover` attaches is not drained. Released here rather than