@trigger.dev/sdk 4.6.1 → 4.6.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/dist/commonjs/v3/ai.d.ts +32 -10
  2. package/dist/commonjs/v3/ai.js +122 -36
  3. package/dist/commonjs/v3/ai.js.map +1 -1
  4. package/dist/commonjs/v3/auth.js +8 -2
  5. package/dist/commonjs/v3/auth.js.map +1 -1
  6. package/dist/commonjs/v3/compactionResponse.d.ts +9 -0
  7. package/dist/commonjs/v3/compactionResponse.js +32 -0
  8. package/dist/commonjs/v3/compactionResponse.js.map +1 -0
  9. package/dist/commonjs/v3/runs.d.ts +5 -0
  10. package/dist/commonjs/v3/sessionTracing.d.ts +7 -0
  11. package/dist/commonjs/v3/sessionTracing.js +44 -0
  12. package/dist/commonjs/v3/sessionTracing.js.map +1 -0
  13. package/dist/commonjs/v3/sessions.js +31 -34
  14. package/dist/commonjs/v3/sessions.js.map +1 -1
  15. package/dist/commonjs/v3/shared.js +8 -5
  16. package/dist/commonjs/v3/shared.js.map +1 -1
  17. package/dist/commonjs/version.js +1 -1
  18. package/dist/esm/v3/ai.d.ts +32 -10
  19. package/dist/esm/v3/ai.js +122 -36
  20. package/dist/esm/v3/ai.js.map +1 -1
  21. package/dist/esm/v3/auth.js +8 -2
  22. package/dist/esm/v3/auth.js.map +1 -1
  23. package/dist/esm/v3/compactionResponse.d.ts +9 -0
  24. package/dist/esm/v3/compactionResponse.js +29 -0
  25. package/dist/esm/v3/compactionResponse.js.map +1 -0
  26. package/dist/esm/v3/runs.d.ts +5 -0
  27. package/dist/esm/v3/sessionTracing.d.ts +7 -0
  28. package/dist/esm/v3/sessionTracing.js +40 -0
  29. package/dist/esm/v3/sessionTracing.js.map +1 -0
  30. package/dist/esm/v3/sessions.js +31 -34
  31. package/dist/esm/v3/sessions.js.map +1 -1
  32. package/dist/esm/v3/shared.js +9 -6
  33. package/dist/esm/v3/shared.js.map +1 -1
  34. package/dist/esm/version.js +1 -1
  35. package/docs/ai-chat/lifecycle-hooks.mdx +4 -2
  36. package/docs/ai-chat/reference.mdx +4 -2
  37. package/package.json +2 -2
@@ -415,6 +415,8 @@ export type ChatNewToolResult = {
415
415
  interface CompactionState {
416
416
  summary: string;
417
417
  baseResponseMessageCount: number;
418
+ /** Completed steps summarized this turn; unlike message counts, matches UI step markers. */
419
+ baseResponseStepCount: number;
418
420
  }
419
421
  /**
420
422
  * Event passed to `summarize` callbacks.
@@ -667,14 +669,24 @@ export type CompactedEvent = {
667
669
  export type ShouldCompactEvent = {
668
670
  /** The current model messages (full conversation). */
669
671
  messages: ModelMessage[];
670
- /** Total token count from the triggering step/turn. */
672
+ /**
673
+ * Total token count of the triggering model call: the step that just finished
674
+ * (`"inner"`), or the turn's LAST step (`"outer"`). This is the size of the
675
+ * context the model held on that call, which is what a compaction decision is
676
+ * about. It is never the sum over a multi-step turn; see `turnUsage` for that.
677
+ */
671
678
  totalTokens: number | undefined;
672
- /** Input token count from the triggering step/turn. */
679
+ /** Input token count of the triggering model call (see `totalTokens`). */
673
680
  inputTokens: number | undefined;
674
- /** Output token count from the triggering step/turn. */
681
+ /** Output token count of the triggering model call (see `totalTokens`). */
675
682
  outputTokens: number | undefined;
676
- /** Full usage object from the triggering step/turn. */
683
+ /** Full usage object of the triggering model call (see `totalTokens`). */
677
684
  usage?: LanguageModelUsage;
685
+ /**
686
+ * The whole turn's usage summed over every step, as the provider billed it.
687
+ * Only present when `source` is `"outer"`.
688
+ */
689
+ turnUsage?: LanguageModelUsage;
678
690
  /** Cumulative token usage across all completed turns. Present in chat.agent contexts. */
679
691
  totalUsage?: LanguageModelUsage;
680
692
  /** The chat session ID (if running inside a chat.agent). */
@@ -1444,13 +1456,17 @@ export type TurnCompleteEvent<TClientData = unknown, TUIM extends UIMessage = UI
1444
1456
  */
1445
1457
  uiMessages: TUIM[];
1446
1458
  /**
1447
- * Only the new model messages from this turn (user message(s) + assistant response).
1448
- * Useful for appending to an existing conversation record.
1459
+ * Model messages for this turn's user message(s) and complete assistant response,
1460
+ * including steps summarized during the turn. Same-ID approval and handover
1461
+ * continuations include the full replacement response, so these are not always
1462
+ * an append-only delta. Persist `messages` for future model context, or upsert
1463
+ * `newUIMessages` by ID for the visible conversation.
1449
1464
  */
1450
1465
  newMessages: ModelMessage[];
1451
1466
  /**
1452
- * Only the new UI messages from this turn (user message(s) + assistant response).
1453
- * Useful for inserting individual message records instead of overwriting the full history.
1467
+ * New or updated UI messages from this turn (user message(s) + assistant response).
1468
+ * Upsert by message ID: approval and handover continuations can replace an
1469
+ * existing assistant message.
1454
1470
  */
1455
1471
  newUIMessages: TUIM[];
1456
1472
  /** The assistant's response for this turn, with aborted parts cleaned up when `stopped` is true. Undefined if `pipeChat` was used manually. */
@@ -2899,6 +2915,8 @@ declare class ChatMessageAccumulator {
2899
2915
  modelMessages: ModelMessage[];
2900
2916
  uiMessages: UIMessage[];
2901
2917
  private _compaction?;
2918
+ /** The run a spliced head-start partial contributed, until its response replaces it. */
2919
+ private _handoverRun?;
2902
2920
  private _pendingMessages?;
2903
2921
  private _steeringQueue;
2904
2922
  constructor(options?: {
@@ -2995,8 +3013,10 @@ declare class ChatMessageAccumulator {
2995
3013
  } | undefined>) | undefined;
2996
3014
  /**
2997
3015
  * Run outer-loop compaction if needed. Call after adding the response
2998
- * and capturing usage. Applies `compactModelMessages` and `compactUIMessages`
2999
- * callbacks if configured.
3016
+ * and capturing usage. Pass the LAST step's usage (`result.usage`), which is
3017
+ * the context the model held on its final call; `result.totalUsage` sums every
3018
+ * step of a tool-using turn and belongs in `context.turnUsage`. Applies
3019
+ * `compactModelMessages` and `compactUIMessages` callbacks if configured.
3000
3020
  *
3001
3021
  * @returns `true` if compaction was performed, `false` otherwise.
3002
3022
  */
@@ -3004,6 +3024,8 @@ declare class ChatMessageAccumulator {
3004
3024
  chatId?: string;
3005
3025
  turn?: number;
3006
3026
  clientData?: unknown;
3027
+ /** The whole turn summed over its steps (`result.totalUsage`). */
3028
+ turnUsage?: LanguageModelUsage;
3007
3029
  totalUsage?: LanguageModelUsage;
3008
3030
  }): Promise<boolean>;
3009
3031
  }
@@ -15,8 +15,10 @@ const v3_1 = require("@trigger.dev/core/v3");
15
15
  // Runtime VALUES go through the ESM/CJS shim so the CJS build can `require`
16
16
  // ESM-only `ai@7` (see ../imports/ai-runtime.ts).
17
17
  const api_1 = require("@opentelemetry/api");
18
+ const sessionTracing_js_1 = require("./sessionTracing.js");
18
19
  const ai_runtime_js_1 = require("../imports/ai-runtime.js");
19
20
  const transcriptStorage_js_1 = require("./transcriptStorage.js");
21
+ const compactionResponse_js_1 = require("./compactionResponse.js");
20
22
  let transcriptStorageOverride;
21
23
  /**
22
24
  * Test-only override for the storage `chat.agent` persists through, so a
@@ -1110,7 +1112,7 @@ async function waitOnChatRoute(route, options) {
1110
1112
  return tracer_js_1.tracer.startActiveSpan(options.spanName ?? `chat.${route}.wait()`, async (span) => {
1111
1113
  const idleMs = (options.idleTimeoutInSeconds ?? 0) * 1000;
1112
1114
  if (idleMs > 0) {
1113
- const warm = await router.next(route, { timeoutMs: idleMs });
1115
+ const warm = await (0, sessionTracing_js_1.traceSessionIdle)(session.id, idleMs / 1000, () => router.next(route, { timeoutMs: idleMs }));
1114
1116
  if (warm) {
1115
1117
  span.setAttribute("wait.resolved", "idle");
1116
1118
  return { ok: true, output: warm.data, record: warm };
@@ -1164,10 +1166,6 @@ async function waitOnChatRoute(route, options) {
1164
1166
  session: session.id,
1165
1167
  io: "in",
1166
1168
  route,
1167
- ...(0, v3_1.accessoryAttributes)({
1168
- items: [{ text: `${session.id}.in:${route}`, variant: "normal" }],
1169
- style: "codepath",
1170
- }),
1171
1169
  },
1172
1170
  });
1173
1171
  }
@@ -1843,6 +1841,14 @@ const chatHandoverPartialKey = locals_js_1.locals.create("chat.handoverPartial")
1843
1841
  * @internal
1844
1842
  */
1845
1843
  const chatHandoverMessageIdKey = locals_js_1.locals.create("chat.handoverMessageId");
1844
+ /**
1845
+ * The model messages a turn-0 head-start splice contributed to the model lane,
1846
+ * keyed by the UI message id it was synthesized under. When the agent's response
1847
+ * completes that message under the same id, this is the run to replace: the UI
1848
+ * form alone converts to something else (its pending tool calls drop out, the
1849
+ * approval round was never on it), so it cannot locate the run itself.
1850
+ */
1851
+ const chatHandoverSplicedRunKey = locals_js_1.locals.create("chat.handoverSplicedRun");
1846
1852
  /**
1847
1853
  * Run-scoped slot indicating that the customer's step-1 head-start
1848
1854
  * response is the FINAL turn response. When true, turn 0 runs through
@@ -1926,16 +1932,19 @@ function synthesizeHandoverUIMessage(partial, messageId) {
1926
1932
  */
1927
1933
  function spliceHandoverPartial(modelMessages, uiMessages, signal) {
1928
1934
  if (!signal.partialAssistantMessage || signal.partialAssistantMessage.length === 0) {
1929
- return;
1935
+ return undefined;
1930
1936
  }
1931
1937
  // Skip if the hydrated chain already persisted the partial under this id.
1932
1938
  const alreadyInChain = signal.messageId !== undefined && uiMessages.some((m) => m.id === signal.messageId);
1933
1939
  if (alreadyInChain)
1934
- return;
1935
- modelMessages.push(...signal.partialAssistantMessage);
1940
+ return undefined;
1941
+ const run = [...signal.partialAssistantMessage];
1942
+ modelMessages.push(...run);
1936
1943
  const partialUI = synthesizeHandoverUIMessage(signal.partialAssistantMessage, signal.messageId);
1937
- if (partialUI)
1938
- uiMessages.push(partialUI);
1944
+ if (!partialUI)
1945
+ return undefined;
1946
+ uiMessages.push(partialUI);
1947
+ return { id: partialUI.id, run };
1939
1948
  }
1940
1949
  /**
1941
1950
  * Per-turn background context queue. Messages added via `chat.backgroundWork.inject()`
@@ -2795,6 +2804,7 @@ async function chatCompact(messages, steps, options) {
2795
2804
  locals_js_1.locals.set(chatCompactionStateKey, {
2796
2805
  summary,
2797
2806
  baseResponseMessageCount: currentStep.response.messages.length,
2807
+ baseResponseStepCount: steps.length,
2798
2808
  });
2799
2809
  // Set model-only override — UI messages stay intact for persistence.
2800
2810
  // The summary becomes the model message history for the next turn,
@@ -3665,8 +3675,8 @@ function isActionTurn(value) {
3665
3675
  * tail does not match the old message's conversion, nothing is changed and
3666
3676
  * `false` is returned so the caller can fall back to a full reconversion.
3667
3677
  */
3668
- async function replaceModelRun(lane, oldUi, newUi, tailAfter) {
3669
- const oldRun = await toModelMessages([stripProviderMetadata(oldUi)]);
3678
+ async function replaceModelRun(lane, oldUi, newUi, tailAfter, knownOldRun) {
3679
+ const oldRun = knownOldRun ?? (await toModelMessages([stripProviderMetadata(oldUi)]));
3670
3680
  const newRun = await toModelMessages([stripProviderMetadata(newUi)]);
3671
3681
  // A message that converts to nothing (a pending tool call with no output yet,
3672
3682
  // which `ignoreIncompleteToolCalls` drops) locates no run in the lane. Matching
@@ -4798,7 +4808,7 @@ function chatAgent(options) {
4798
4808
  const preloadResult = await messagesInput.waitWithIdleTimeout({
4799
4809
  idleTimeoutInSeconds: effectivePreloadIdleTimeout,
4800
4810
  timeout: effectivePreloadTimeout,
4801
- spanName: "waiting for first message",
4811
+ spanName: "first message",
4802
4812
  skipSuspend: exitAfterPreloadIdle,
4803
4813
  onSuspend: onChatSuspend
4804
4814
  ? async () => {
@@ -4950,7 +4960,7 @@ function chatAgent(options) {
4950
4960
  const continuationResult = await messagesInput.waitWithIdleTimeout({
4951
4961
  idleTimeoutInSeconds: effectiveIdleTimeout,
4952
4962
  timeout: effectiveTurnTimeout,
4953
- spanName: "waiting for first message (continuation)",
4963
+ spanName: "first message (continuation)",
4954
4964
  onSuspend: onChatSuspend
4955
4965
  ? async () => {
4956
4966
  await tracer_js_1.tracer.startActiveSpan("onChatSuspend()", async () => {
@@ -5479,10 +5489,12 @@ function chatAgent(options) {
5479
5489
  // `UIMessageStreamError: No tool invocation found`.
5480
5490
  const pendingHandoverPartial = locals_js_1.locals.get(chatHandoverPartialKey);
5481
5491
  if (pendingHandoverPartial && pendingHandoverPartial.length > 0) {
5482
- spliceHandoverPartial(accumulatedMessages, accumulatedUIMessages, {
5492
+ const spliced = spliceHandoverPartial(accumulatedMessages, accumulatedUIMessages, {
5483
5493
  partialAssistantMessage: pendingHandoverPartial,
5484
5494
  messageId: locals_js_1.locals.get(chatHandoverMessageIdKey),
5485
5495
  });
5496
+ if (spliced)
5497
+ locals_js_1.locals.set(chatHandoverSplicedRunKey, spliced);
5486
5498
  locals_js_1.locals.set(chatHandoverPartialKey, []); // consume once
5487
5499
  splicedHandoverPartial = true;
5488
5500
  }
@@ -5827,6 +5839,7 @@ function chatAgent(options) {
5827
5839
  // never reports final usage), which would block the turn loop
5828
5840
  // from ever firing onTurnComplete / writeTurnComplete.
5829
5841
  let turnUsage;
5842
+ let lastStepUsage;
5830
5843
  if (runResult != null &&
5831
5844
  typeof runResult.totalUsage?.then === "function") {
5832
5845
  try {
@@ -5839,6 +5852,18 @@ function chatAgent(options) {
5839
5852
  /* non-fatal — usage capture failed */
5840
5853
  }
5841
5854
  }
5855
+ const lastStepUsagePromise = runResult != null ? runResult.usage : undefined;
5856
+ if (typeof lastStepUsagePromise?.then === "function") {
5857
+ try {
5858
+ lastStepUsage = (await Promise.race([
5859
+ lastStepUsagePromise,
5860
+ new Promise((r) => setTimeout(() => r(undefined), 2_000)),
5861
+ ]));
5862
+ }
5863
+ catch {
5864
+ /* non-fatal — usage capture failed */
5865
+ }
5866
+ }
5842
5867
  if (turnUsage) {
5843
5868
  cumulativeUsage = addUsage(cumulativeUsage, turnUsage);
5844
5869
  previousTurnUsage = turnUsage;
@@ -5889,6 +5914,13 @@ function chatAgent(options) {
5889
5914
  // Check if compaction set a model-only override (preserves UI messages).
5890
5915
  // Apply compactUIMessages/compactModelMessages callbacks if configured.
5891
5916
  const modelOnlyOverride = locals_js_1.locals.get(chatOverrideModelMessagesKey);
5917
+ const responseCompaction = modelOnlyOverride
5918
+ ? locals_js_1.locals.get(chatCompactionStateKey)
5919
+ : undefined;
5920
+ // Capture the original assistant before compactUIMessages can remove it.
5921
+ const originalResponse = responseCompaction && capturedResponseMessage
5922
+ ? accumulatedUIMessages.find((m) => m.id === capturedResponseMessage?.id)
5923
+ : undefined;
5892
5924
  if (modelOnlyOverride) {
5893
5925
  const compactionSummary = locals_js_1.locals.get(chatCompactionStateKey)?.summary ?? "";
5894
5926
  const taskCompactionConfig = locals_js_1.locals.get(chatAgentCompactionKey);
@@ -5980,12 +6012,39 @@ function chatAgent(options) {
5980
6012
  // rationale (TRI-9137).
5981
6013
  recordToolCallIdsFromMessage(capturedResponseMessage);
5982
6014
  try {
6015
+ const responseForModel = (0, compactionResponse_js_1.responseAfterCompaction)(capturedResponseMessage, responseCompaction?.baseResponseStepCount, originalResponse);
6016
+ // Preserve the complete persistence response, including same-ID
6017
+ // replacements whose old tool parts can contain new results.
6018
+ // Convert prefix and suffix separately so each tool output is
6019
+ // converted once, while only the suffix enters model context.
6020
+ const responsePrefixMessages = responseCompaction
6021
+ ? await toModelMessages([
6022
+ stripProviderMetadata({
6023
+ ...capturedResponseMessage,
6024
+ parts: capturedResponseMessage.parts.slice(0, capturedResponseMessage.parts.length -
6025
+ responseForModel.parts.length),
6026
+ }),
6027
+ ])
6028
+ : [];
5983
6029
  const responseModelMessages = await toModelMessages([
5984
- stripProviderMetadata(capturedResponseMessage),
6030
+ stripProviderMetadata(responseForModel),
5985
6031
  ]);
5986
- if (existingIdx !== -1) {
6032
+ if (responseCompaction) {
6033
+ // The summary already replaced the original response, including
6034
+ // a same-ID approval/handover prefix. Replacing its old model run
6035
+ // would miss and fall back to the full, uncompacted UI history.
6036
+ accumulatedMessages.push(...responseModelMessages);
6037
+ locals_js_1.locals.set(chatHandoverSplicedRunKey, undefined);
6038
+ }
6039
+ else if (existingIdx !== -1) {
6040
+ const spliced = locals_js_1.locals.get(chatHandoverSplicedRunKey);
6041
+ const splicedRun = spliced && previousAtIdx && spliced.id === previousAtIdx.id
6042
+ ? spliced.run
6043
+ : undefined;
5987
6044
  const ok = previousAtIdx !== undefined &&
5988
- (await replaceModelRun(accumulatedMessages, previousAtIdx, capturedResponseMessage, steerTailThisTurn));
6045
+ (await replaceModelRun(accumulatedMessages, previousAtIdx, capturedResponseMessage, steerTailThisTurn, splicedRun));
6046
+ if (splicedRun)
6047
+ locals_js_1.locals.set(chatHandoverSplicedRunKey, undefined);
5989
6048
  if (!ok) {
5990
6049
  v3_1.logger.warn("chat.agent: replaced response not found at the model lane tail; reconverting the lane");
5991
6050
  accumulatedMessages = await toModelMessages(accumulatedUIMessages);
@@ -5996,7 +6055,7 @@ function chatAgent(options) {
5996
6055
  else {
5997
6056
  accumulatedMessages.push(...responseModelMessages);
5998
6057
  }
5999
- turnNewModelMessages.push(...responseModelMessages);
6058
+ turnNewModelMessages.push(...responsePrefixMessages, ...responseModelMessages);
6000
6059
  }
6001
6060
  catch {
6002
6061
  // Conversion failed — skip accumulation for this turn
@@ -6045,12 +6104,14 @@ function chatAgent(options) {
6045
6104
  const outerCompaction = locals_js_1.locals.get(chatAgentCompactionKey);
6046
6105
  const innerCompactionState = locals_js_1.locals.get(chatCompactionStateKey);
6047
6106
  if (outerCompaction && !innerCompactionState && turnUsage && !wasStopped) {
6107
+ const contextUsage = lastStepUsage ?? turnUsage;
6048
6108
  const shouldTrigger = await outerCompaction.shouldCompact({
6049
6109
  messages: accumulatedMessages,
6050
- totalTokens: turnUsage.totalTokens,
6051
- inputTokens: turnUsage.inputTokens,
6052
- outputTokens: turnUsage.outputTokens,
6053
- usage: turnUsage,
6110
+ totalTokens: contextUsage.totalTokens,
6111
+ inputTokens: contextUsage.inputTokens,
6112
+ outputTokens: contextUsage.outputTokens,
6113
+ usage: contextUsage,
6114
+ turnUsage,
6054
6115
  totalUsage: cumulativeUsage,
6055
6116
  chatId: currentWirePayload.chatId,
6056
6117
  turn,
@@ -6375,7 +6436,7 @@ function chatAgent(options) {
6375
6436
  const next = await messagesInput.waitWithIdleTimeout({
6376
6437
  idleTimeoutInSeconds: effectiveIdleTimeout,
6377
6438
  timeout: effectiveTurnTimeout,
6378
- spanName: "waiting for next message",
6439
+ spanName: "next message",
6379
6440
  onSuspend: onChatSuspend
6380
6441
  ? async () => {
6381
6442
  await tracer_js_1.tracer.startActiveSpan("onChatSuspend()", async () => {
@@ -6712,7 +6773,7 @@ function chatAgent(options) {
6712
6773
  const next = await messagesInput.waitWithIdleTimeout({
6713
6774
  idleTimeoutInSeconds: effectiveIdleTimeout,
6714
6775
  timeout: effectiveTurnTimeout,
6715
- spanName: "waiting for next message (after error)",
6776
+ spanName: "next message (after error)",
6716
6777
  });
6717
6778
  if (!next.ok) {
6718
6779
  return; // Timed out — end run gracefully
@@ -7752,6 +7813,8 @@ class ChatMessageAccumulator {
7752
7813
  modelMessages = [];
7753
7814
  uiMessages = [];
7754
7815
  _compaction;
7816
+ /** The run a spliced head-start partial contributed, until its response replaces it. */
7817
+ _handoverRun;
7755
7818
  _pendingMessages;
7756
7819
  _steeringQueue = [];
7757
7820
  constructor(options) {
@@ -7795,7 +7858,7 @@ class ChatMessageAccumulator {
7795
7858
  * `consumeHandover` for the wait+seed+apply convenience.
7796
7859
  */
7797
7860
  applyHandover(signal) {
7798
- spliceHandoverPartial(this.modelMessages, this.uiMessages, signal);
7861
+ this._handoverRun = spliceHandoverPartial(this.modelMessages, this.uiMessages, signal);
7799
7862
  }
7800
7863
  /**
7801
7864
  * One-call `chat.headStart` handover for a custom-agent loop: waits for the
@@ -7838,8 +7901,13 @@ class ChatMessageAccumulator {
7838
7901
  if (existingIdx !== -1) {
7839
7902
  const previous = this.uiMessages[existingIdx];
7840
7903
  this.uiMessages[existingIdx] = response;
7904
+ const handoverRun = this._handoverRun && this._handoverRun.id === previous.id
7905
+ ? this._handoverRun.run
7906
+ : undefined;
7907
+ if (handoverRun)
7908
+ this._handoverRun = undefined;
7841
7909
  try {
7842
- if (!(await replaceModelRun(this.modelMessages, previous, response, 0))) {
7910
+ if (!(await replaceModelRun(this.modelMessages, previous, response, 0, handoverRun))) {
7843
7911
  this.modelMessages = await toModelMessages(this.uiMessages.map((m) => stripProviderMetadata(m)));
7844
7912
  }
7845
7913
  }
@@ -7942,8 +8010,10 @@ class ChatMessageAccumulator {
7942
8010
  }
7943
8011
  /**
7944
8012
  * Run outer-loop compaction if needed. Call after adding the response
7945
- * and capturing usage. Applies `compactModelMessages` and `compactUIMessages`
7946
- * callbacks if configured.
8013
+ * and capturing usage. Pass the LAST step's usage (`result.usage`), which is
8014
+ * the context the model held on its final call; `result.totalUsage` sums every
8015
+ * step of a tool-using turn and belongs in `context.turnUsage`. Applies
8016
+ * `compactModelMessages` and `compactUIMessages` callbacks if configured.
7947
8017
  *
7948
8018
  * @returns `true` if compaction was performed, `false` otherwise.
7949
8019
  */
@@ -7956,6 +8026,7 @@ class ChatMessageAccumulator {
7956
8026
  inputTokens: usage.inputTokens,
7957
8027
  outputTokens: usage.outputTokens,
7958
8028
  usage,
8029
+ turnUsage: context?.turnUsage,
7959
8030
  totalUsage: context?.totalUsage,
7960
8031
  chatId: context?.chatId,
7961
8032
  turn: context?.turn,
@@ -8188,8 +8259,8 @@ function createChatSession(payload, options) {
8188
8259
  idleTimeoutInSeconds: sessionIdleTimeoutOpt ?? currentPayload.idleTimeoutInSeconds ?? 30,
8189
8260
  timeout,
8190
8261
  spanName: currentPayload.trigger === "preload"
8191
- ? "waiting for first message"
8192
- : "waiting for first message (continuation)",
8262
+ ? "first message"
8263
+ : "first message (continuation)",
8193
8264
  });
8194
8265
  if (!result.ok || runSignal.aborted) {
8195
8266
  stop.cleanup();
@@ -8226,7 +8297,7 @@ function createChatSession(payload, options) {
8226
8297
  const next = await messagesInput.waitWithIdleTimeout({
8227
8298
  idleTimeoutInSeconds,
8228
8299
  timeout,
8229
- spanName: "waiting for next message",
8300
+ spanName: "next message",
8230
8301
  });
8231
8302
  if (!next.ok || runSignal.aborted) {
8232
8303
  stop.cleanup();
@@ -8437,6 +8508,7 @@ function createChatSession(payload, options) {
8437
8508
  // indefinitely, which would wedge the turn loop (same guard as
8438
8509
  // chat.agent's turn loop).
8439
8510
  let turnUsage;
8511
+ let lastStepUsage;
8440
8512
  if (typeof source.totalUsage?.then === "function") {
8441
8513
  try {
8442
8514
  const usage = (await Promise.race([
@@ -8453,14 +8525,28 @@ function createChatSession(payload, options) {
8453
8525
  /* non-fatal */
8454
8526
  }
8455
8527
  }
8528
+ const lastStepUsagePromise = source.usage;
8529
+ if (typeof lastStepUsagePromise?.then === "function") {
8530
+ try {
8531
+ lastStepUsage = (await Promise.race([
8532
+ lastStepUsagePromise,
8533
+ new Promise((r) => setTimeout(() => r(undefined), 2_000)),
8534
+ ]));
8535
+ }
8536
+ catch {
8537
+ /* non-fatal */
8538
+ }
8539
+ }
8456
8540
  // Outer-loop compaction (same logic as chat.agent)
8457
8541
  if (sessionCompaction && turnUsage && !turnObj.stopped) {
8542
+ const contextUsage = lastStepUsage ?? turnUsage;
8458
8543
  const shouldTrigger = await sessionCompaction.shouldCompact({
8459
8544
  messages: accumulator.modelMessages,
8460
- totalTokens: turnUsage.totalTokens,
8461
- inputTokens: turnUsage.inputTokens,
8462
- outputTokens: turnUsage.outputTokens,
8463
- usage: turnUsage,
8545
+ totalTokens: contextUsage.totalTokens,
8546
+ inputTokens: contextUsage.inputTokens,
8547
+ outputTokens: contextUsage.outputTokens,
8548
+ usage: contextUsage,
8549
+ turnUsage,
8464
8550
  totalUsage: cumulativeUsage,
8465
8551
  chatId: currentPayload.chatId,
8466
8552
  turn,