@memberjunction/ai-agents 5.48.0 → 5.50.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/README.md +14 -0
  2. package/dist/AgentRunner.d.ts.map +1 -1
  3. package/dist/AgentRunner.js +5 -0
  4. package/dist/AgentRunner.js.map +1 -1
  5. package/dist/ConversationCompactionManager.d.ts +202 -0
  6. package/dist/ConversationCompactionManager.d.ts.map +1 -0
  7. package/dist/ConversationCompactionManager.js +399 -0
  8. package/dist/ConversationCompactionManager.js.map +1 -0
  9. package/dist/ConversationToolManager.d.ts +154 -0
  10. package/dist/ConversationToolManager.d.ts.map +1 -0
  11. package/dist/ConversationToolManager.js +336 -0
  12. package/dist/ConversationToolManager.js.map +1 -0
  13. package/dist/agent-types/loop-agent-prompt-params.d.ts +8 -0
  14. package/dist/agent-types/loop-agent-prompt-params.d.ts.map +1 -1
  15. package/dist/agent-types/loop-agent-prompt-params.js +1 -0
  16. package/dist/agent-types/loop-agent-prompt-params.js.map +1 -1
  17. package/dist/agent-types/loop-agent-response-type.d.ts +10 -1
  18. package/dist/agent-types/loop-agent-response-type.d.ts.map +1 -1
  19. package/dist/agent-types/loop-agent-response-type.js.map +1 -1
  20. package/dist/agent-types/loop-agent-type.d.ts +23 -34
  21. package/dist/agent-types/loop-agent-type.d.ts.map +1 -1
  22. package/dist/agent-types/loop-agent-type.js +75 -2
  23. package/dist/agent-types/loop-agent-type.js.map +1 -1
  24. package/dist/base-agent.d.ts +314 -4
  25. package/dist/base-agent.d.ts.map +1 -1
  26. package/dist/base-agent.js +863 -95
  27. package/dist/base-agent.js.map +1 -1
  28. package/dist/conversation-history-format.d.ts +19 -0
  29. package/dist/conversation-history-format.d.ts.map +1 -0
  30. package/dist/conversation-history-format.js +40 -0
  31. package/dist/conversation-history-format.js.map +1 -0
  32. package/dist/index.d.ts +4 -0
  33. package/dist/index.d.ts.map +1 -1
  34. package/dist/index.js +4 -0
  35. package/dist/index.js.map +1 -1
  36. package/dist/prior-turn-tool-result-cache.d.ts +71 -0
  37. package/dist/prior-turn-tool-result-cache.d.ts.map +1 -0
  38. package/dist/prior-turn-tool-result-cache.js +85 -0
  39. package/dist/prior-turn-tool-result-cache.js.map +1 -0
  40. package/dist/realtime/realtime-client-session-service.d.ts +8 -1
  41. package/dist/realtime/realtime-client-session-service.d.ts.map +1 -1
  42. package/dist/realtime/realtime-client-session-service.js +44 -14
  43. package/dist/realtime/realtime-client-session-service.js.map +1 -1
  44. package/dist/realtime/realtime-coagent-config.d.ts +41 -0
  45. package/dist/realtime/realtime-coagent-config.d.ts.map +1 -1
  46. package/dist/realtime/realtime-coagent-config.js +69 -0
  47. package/dist/realtime/realtime-coagent-config.js.map +1 -1
  48. package/dist/realtime/realtime-session-runner.d.ts +72 -1
  49. package/dist/realtime/realtime-session-runner.d.ts.map +1 -1
  50. package/dist/realtime/realtime-session-runner.js +215 -13
  51. package/dist/realtime/realtime-session-runner.js.map +1 -1
  52. package/dist/realtime/realtime-tool-broker.d.ts +7 -2
  53. package/dist/realtime/realtime-tool-broker.d.ts.map +1 -1
  54. package/dist/realtime/realtime-tool-broker.js +13 -10
  55. package/dist/realtime/realtime-tool-broker.js.map +1 -1
  56. package/dist/realtime/realtime-turn-moderator.d.ts.map +1 -1
  57. package/dist/realtime/realtime-turn-moderator.js +0 -1
  58. package/dist/realtime/realtime-turn-moderator.js.map +1 -1
  59. package/dist/tool-result-format.d.ts +78 -0
  60. package/dist/tool-result-format.d.ts.map +1 -0
  61. package/dist/tool-result-format.js +46 -0
  62. package/dist/tool-result-format.js.map +1 -0
  63. package/package.json +18 -18
@@ -23,7 +23,7 @@ import { CrushJSON, DescribeCrush, PartitionStablePrefix } from '@memberjunction
23
23
  import { CrushCode } from '@memberjunction/context-crush/code';
24
24
  import { RealtimeSessionRunner } from './realtime/realtime-session-runner.js';
25
25
  import { ResolveNarrationInstructionsTemplate } from './realtime/realtime-narration.js';
26
- import { BuildRealtimeOverridesJson, BuildVoiceMannerSection, GetNarrationPaceMs, GetProviderVoiceSettings, ResolveEffectiveRealtimeConfig } from './realtime/realtime-coagent-config.js';
26
+ import { BuildRealtimeOverridesJson, BuildVoiceMannerSection, GetNarrationPaceMs, GetProviderVoiceSettings, GetSessionTuningSettings, DeepMergeConfigs, ResolveEffectiveRealtimeConfig } from './realtime/realtime-coagent-config.js';
27
27
  import { RealtimeClientSessionService } from './realtime/realtime-client-session-service.js';
28
28
  import { BuildRealtimeAgentFraming } from './realtime/realtime-tool-broker.js';
29
29
  import { RealtimeRecordingController } from './realtime/realtime-recording-capture.js';
@@ -32,9 +32,13 @@ import { AIEngine } from '@memberjunction/aiengine';
32
32
  import { ActionEngineServer } from '@memberjunction/actions';
33
33
  import { AIAgentPermissionHelper } from '@memberjunction/ai-engine-base';
34
34
  import { AgentMemoryContextBuilder } from './agent-memory-context-builder.js';
35
+ import { ConversationCompactionManager } from './ConversationCompactionManager.js';
36
+ import { ConversationToolManager, ConversationToolNames, MAX_CONVERSATION_TOOL_CALLS_PER_TURN } from './ConversationToolManager.js';
37
+ import { FormatToolResultSection, FormatToolErrorSection, RenderToolResultData, CarryForwardToolFamily } from './tool-result-format.js';
38
+ import { PriorTurnToolResultCache } from './prior-turn-tool-result-cache.js';
35
39
  import { PromptComponentResolver, InjectScopedPromptParts } from './prompt-component-resolver.js';
36
40
  import { ScopedPromptConfigResolver, ApplyScopedPromptConfig } from './scoped-prompt-config-resolver.js';
37
- import { AIPromptParams, ChildPromptParam, ConversationUtility, ParseFileOutputRef, parseAssignmentStrategy, ResolveClientTools, initAgentRunStep, finalizeAgentRunStep, AgentRunStepSaveQueue } from '@memberjunction/ai-core-plus';
41
+ import { AIPromptParams, ChildPromptParam, ConversationUtility, ParseFileOutputRef, parseAssignmentStrategy, ResolveClientTools, initAgentRunStep, finalizeAgentRunStep, AgentRunStepSaveQueue, ExtractPromptResultText } from '@memberjunction/ai-core-plus';
38
42
  import { AgentRunner } from './AgentRunner.js';
39
43
  import { PayloadManager } from './PayloadManager.js';
40
44
  import { ScratchpadManager } from './ScratchpadManager.js';
@@ -215,6 +219,11 @@ export class BaseAgent {
215
219
  * Allows agents to explore input artifacts on demand.
216
220
  */
217
221
  this._artifactToolManager = new ArtifactToolManager();
222
+ /**
223
+ * Manages conversation-history retrieval tools for the current agent run.
224
+ * Armed only when the run has a conversationId (the cross-turn context gate).
225
+ */
226
+ this._conversationToolManager = new ConversationToolManager();
218
227
  /**
219
228
  * Manages in-flight durable memory writes for the current agent run.
220
229
  * Only consulted when the agent has AllowMemoryWrite enabled.
@@ -332,11 +341,30 @@ export class BaseAgent {
332
341
  */
333
342
  // ── Realtime per-session capture state (scoped to one executeRealtimeSession run) ──────────
334
343
  /**
335
- * In-flight realtime turn rows keyed by transcript role (`'user'`/`'assistant'`), driving the
336
- * create-on-start / update-on-complete persistence lifecycle. Reset at the start of every
337
- * realtime session so a prior run can never leak an in-flight id into the next.
344
+ * The current realtime turn row per transcript role (`'user'`/`'assistant'`), driving the
345
+ * create-on-start / update-on-complete persistence lifecycle. `open` is true while the row is an
346
+ * unfinalized In-Progress interim (so subsequent interim deltas fold into it and a following final
347
+ * finalizes it in place); it flips false once finalized, but the entry is KEPT so a streamed
348
+ * `ReplacesPrevious` re-final can still update the same row. A new turn is detected when the next
349
+ * interim (or non-replacing final) arrives with the current entry already closed. Reset at the
350
+ * start of every realtime session so a prior run can never leak a row id into the next.
338
351
  */
339
352
  this.realtimeInFlightTurns = new Map();
353
+ /**
354
+ * Per-role serialization queue for transcript persistence.
355
+ *
356
+ * The runner dispatches provider transcript frames FIRE-AND-FORGET (`void this.handleTranscript(t)`),
357
+ * so frames for the same role can be in flight CONCURRENTLY. {@link persistRealtimeTranscript} does a
358
+ * check-then-act on {@link realtimeInFlightTurns} that spans `await`s (GetEntityObject / Load / Save):
359
+ * without serialization, two captions arriving a few ms apart both observe "no tracked row yet", both
360
+ * take the create branch, and the turn is persisted TWICE. Observed in production against a streamed
361
+ * Grok session (two byte-identical rows, the second created 17ms before the first's final update).
362
+ *
363
+ * Each role's calls are therefore chained through this map so the read-modify-write is atomic with
364
+ * respect to other frames of the SAME role. Roles are independent (separate `realtimeInFlightTurns`
365
+ * entries), so they are not serialized against each other. Reset per session alongside the turn map.
366
+ */
367
+ this.realtimePersistQueues = new Map();
340
368
  /** Active audio recording controller for the current realtime session, or `null` when recording is off. */
341
369
  this.realtimeRecording = null;
342
370
  /** Storage account id the active recording stores to (RecordingStorageProviderID ?? AttachmentStorageProviderID). */
@@ -1089,6 +1117,11 @@ export class BaseAgent {
1089
1117
  ...params,
1090
1118
  onProgress: this.wrapProgressCallback(params.onProgress)
1091
1119
  };
1120
+ // Capture for lifecycle hooks that don't receive params (post-turn compaction
1121
+ // inside finalizeAgentRun reads conversationId / verbose / provider from here).
1122
+ this._executeParams = wrappedParams;
1123
+ this._agentConfig = undefined;
1124
+ this._lastModelSelectionInfo = undefined;
1092
1125
  // Convert UI markup in conversation messages to plain text if requested (default: true)
1093
1126
  if (params.convertUIMarkupToPlainText !== false) {
1094
1127
  this.convertUIMarkupInMessages(wrappedParams.conversationMessages);
@@ -1097,6 +1130,10 @@ export class BaseAgent {
1097
1130
  this._scratchpadManager.Clear();
1098
1131
  this._artifactToolManager.Clear();
1099
1132
  this._memoryWriteManager.Clear();
1133
+ // Arm conversation-history retrieval tools — available only when the run has a
1134
+ // conversation to page against (the same gate as all cross-turn context features).
1135
+ this._conversationToolManager.Initialize(wrappedParams.conversationId || null, params.contextUser);
1136
+ this._conversationToolManager.SetSummaryHost(this.buildConversationSummaryHost(wrappedParams));
1100
1137
  // Initialize artifact tools with any input artifacts attached to the run.
1101
1138
  // Artifacts arrive as a typed first-class field on ExecuteAgentParams —
1102
1139
  // they are NOT routed through `data` because prompt-template rendering
@@ -1224,7 +1261,11 @@ export class BaseAgent {
1224
1261
  this.loadAgentConfiguration(params.agent),
1225
1262
  this.preloadAgentData(wrappedParams),
1226
1263
  this.InjectContextMemory(typeof inputText === 'string' ? inputText : '', params.agent, userId, companyId, params.contextUser, wrappedParams.conversationMessages, primaryScopeEntityId, primaryScopeRecordId, secondaryScopes, scopeConfig),
1227
- this.InjectPreExecutionRAG(typeof inputText === 'string' ? inputText : '', params.agent, params.contextUser, wrappedParams.conversationMessages, params.conversationMessages, primaryScopeEntityId, primaryScopeRecordId, secondaryScopes, params.payload)
1264
+ this.InjectPreExecutionRAG(typeof inputText === 'string' ? inputText : '', params.agent, params.contextUser, wrappedParams.conversationMessages, params.conversationMessages, primaryScopeEntityId, primaryScopeRecordId, secondaryScopes, params.payload),
1265
+ // Carry the previous turn's tool results forward (no-op without a
1266
+ // conversationId). Runs here so the results are in the messages before
1267
+ // the pre-turn compaction check and the first prompt.
1268
+ this.injectPriorTurnToolResults(wrappedParams)
1228
1269
  ]);
1229
1270
  // Inject scope-resolved prompt parts (role-faithful) for this agent's prompt, alongside
1230
1271
  // memory/RAG. Synchronous — parts are cached on AIEngine. Uses the same run scope.
@@ -1239,6 +1280,7 @@ export class BaseAgent {
1239
1280
  // --- PHASE 3: Agent type initialization (sequential) ---
1240
1281
  // Must wait for config from Phase 2 because it needs the resolved agent type and
1241
1282
  // prompt configuration to initialize the type-specific state machine.
1283
+ this._agentConfig = config;
1242
1284
  await this.initializeAgentType(wrappedParams, config);
1243
1285
  // =====================================================================================
1244
1286
  // SESSION-DRIVEN BRANCH (Realtime agent type)
@@ -1255,6 +1297,10 @@ export class BaseAgent {
1255
1297
  this.logStatus(`🎙️ Agent '${params.agent.Name}' is session-driven — routing to RealtimeSessionRunner`, true, params);
1256
1298
  return await this.executeRealtimeSession(wrappedParams, config);
1257
1299
  }
1300
+ // Cross-turn compaction PRE-TURN fallback: only when the assembled window is
1301
+ // ALREADY over the trigger budget before the first prompt (the normal path is
1302
+ // the post-turn fire-and-forget in finalizeAgentRun, which hides the latency).
1303
+ await this.checkPreTurnCompaction(wrappedParams, config);
1258
1304
  // Execute the agent's internal logic with wrapped parameters
1259
1305
  this.logStatus(`🚀 Executing agent '${params.agent.Name}' internal logic`, true, params);
1260
1306
  const executionResult = await this.executeAgentInternal(wrappedParams, config);
@@ -1362,6 +1408,7 @@ export class BaseAgent {
1362
1408
  // 3) Resolve recording (OFF by default; runtime > agent > off; consent + storage gated) and reset
1363
1409
  // the per-session turn-lifecycle state, then build the injected deps and run the session.
1364
1410
  this.realtimeInFlightTurns = new Map();
1411
+ this.realtimePersistQueues = new Map();
1365
1412
  const recording = await this.resolveRealtimeRecording(params);
1366
1413
  this.realtimeRecording = recording?.controller ?? null;
1367
1414
  this.realtimeRecordingAccountId = recording?.storageAccountId ?? null;
@@ -1616,7 +1663,6 @@ export class BaseAgent {
1616
1663
  promptRun.ModelID = modelResolution.modelID;
1617
1664
  promptRun.VendorID = modelResolution.vendorID || null;
1618
1665
  promptRun.AgentID = params.agent.ID;
1619
- promptRun.AgentRunID = this._agentRun?.ID ?? null;
1620
1666
  promptRun.Status = 'Running';
1621
1667
  promptRun.RunAt = new Date();
1622
1668
  promptRun.StreamingEnabled = true;
@@ -1660,9 +1706,14 @@ export class BaseAgent {
1660
1706
  DelegateToTarget: (request) => this.delegateRealtimeToTarget(params, config, request),
1661
1707
  ExecuteTool: (call) => this.executeRealtimeTool(params, call),
1662
1708
  PersistTranscript: (transcript) => this.persistRealtimeTranscript(params, transcript),
1709
+ FlushTranscripts: () => this.flushRealtimeTranscriptQueues(),
1663
1710
  Recording: this.realtimeRecording ?? undefined,
1664
1711
  FinalizeRecording: () => this.finalizeRealtimeRecording(params),
1665
1712
  CheckpointUsage: (usage) => this.checkpointRealtimeUsage(promptRun, usage),
1713
+ // The chained agent cancellation signal (caller token + agent timeout) — the runner
1714
+ // observes it so a realtime session honors the same wall-clock/cancel semantics as
1715
+ // every other agent run instead of living until the janitor sweeps it.
1716
+ AbortSignal: params.cancellationToken,
1666
1717
  // DB-driven spoken-progress wording (shared lookup with the client-direct path);
1667
1718
  // null → the runner's documented built-in first-person fallback.
1668
1719
  NarrationInstructionsTemplate: ResolveNarrationInstructionsTemplate(),
@@ -1699,15 +1750,20 @@ export class BaseAgent {
1699
1750
  const systemPrompt = [framing, basePrompt, voiceManner, memoryContext]
1700
1751
  .filter(part => part && part.trim().length > 0)
1701
1752
  .join('\n\n');
1702
- // Provider-matched voice settings (realtime.voice.providers.<provider>) flow into the
1703
- // driver's open Config bag — the same pact every other config entry rides.
1753
+ // Provider-matched voice settings (realtime.voice.providers.<provider>) AND session-tuning
1754
+ // knobs (realtime.session) flow into the driver's open Config bag — the same pact every
1755
+ // other config entry rides, mirroring the client-direct builder's cascade exactly.
1704
1756
  const providerVoice = GetProviderVoiceSettings(effectiveConfig, driverClass ?? null);
1757
+ const sessionTuning = GetSessionTuningSettings(effectiveConfig);
1758
+ const configBag = (sessionTuning || providerVoice)
1759
+ ? DeepMergeConfigs(sessionTuning, providerVoice)
1760
+ : undefined;
1705
1761
  return {
1706
1762
  Model: modelApiName,
1707
1763
  SystemPrompt: systemPrompt,
1708
1764
  InitialContext: memoryContext || undefined,
1709
- // JSONObjectLike -> JSONObject: safe — the settings object came from JSON.parse.
1710
- Config: providerVoice ? providerVoice : undefined
1765
+ // JSONObjectLike -> JSONObject: safe — the settings objects came from JSON.parse.
1766
+ Config: configBag
1711
1767
  };
1712
1768
  }
1713
1769
  /**
@@ -1922,7 +1978,45 @@ export class BaseAgent {
1922
1978
  * @param transcript The transcript turn (interim delta or final) emitted by the model.
1923
1979
  * @returns The created row id on first creation of a turn, else `null`.
1924
1980
  */
1925
- async persistRealtimeTranscript(params, transcript) {
1981
+ persistRealtimeTranscript(params, transcript) {
1982
+ // Serialize per role — see realtimePersistQueues. Transcript frames arrive fire-and-forget, so
1983
+ // without this chain two concurrent captions can both pass the "is there a tracked row?" check
1984
+ // before either has written one back, and the turn is persisted twice.
1985
+ const roleKey = transcript.Role;
1986
+ const run = () => this.persistRealtimeTranscriptSerialized(params, transcript);
1987
+ const prior = this.realtimePersistQueues.get(roleKey) ?? Promise.resolve();
1988
+ // `.then(run, run)` (not `.then(run)`) so a rejected predecessor never strands the rest of the
1989
+ // queue — each frame runs regardless of how the previous one settled.
1990
+ const result = prior.then(run, run);
1991
+ // The stored link swallows outcomes: the queue only needs ordering, and an unhandled rejection
1992
+ // parked in the map would surface as an unhandled promise rejection.
1993
+ this.realtimePersistQueues.set(roleKey, result.then(() => undefined, () => undefined));
1994
+ return result;
1995
+ }
1996
+ /**
1997
+ * Waits for every role's queued transcript writes to settle.
1998
+ *
1999
+ * Transcript frames are dispatched fire-and-forget, so writes for the last turns of a session can
2000
+ * still be in flight at teardown. The session runner calls this during `Stop()` — after the provider
2001
+ * session is closed, so no new frames can arrive — under its own hard timeout, which is why this
2002
+ * method itself is unbounded and simply awaits what is queued.
2003
+ *
2004
+ * Awaits the STORED queue links, which are outcome-swallowing by construction, so a failed write
2005
+ * can never reject here and abort the drain for other roles.
2006
+ */
2007
+ async flushRealtimeTranscriptQueues() {
2008
+ const pending = [...this.realtimePersistQueues.values()];
2009
+ if (pending.length === 0) {
2010
+ return;
2011
+ }
2012
+ await Promise.all(pending);
2013
+ }
2014
+ /**
2015
+ * The actual persistence work for one transcript frame. Runs under the per-role queue established by
2016
+ * {@link persistRealtimeTranscript}, so it may safely read-modify-write {@link realtimeInFlightTurns}
2017
+ * across its `await`s without another frame of the same role interleaving.
2018
+ */
2019
+ async persistRealtimeTranscriptSerialized(params, transcript) {
1926
2020
  if (!transcript.Text?.trim()) {
1927
2021
  return null;
1928
2022
  }
@@ -1935,9 +2029,12 @@ export class BaseAgent {
1935
2029
  const mjRole = transcript.Role === 'user' ? 'User' : 'AI';
1936
2030
  // ── INTERIM: create the In-Progress row once per turn (first delta) ───────────────────────
1937
2031
  if (!transcript.IsFinal) {
1938
- if (this.realtimeInFlightTurns.has(roleKey)) {
1939
- return null; // already created for this turn; ignore subsequent deltas
2032
+ if (this.realtimeInFlightTurns.get(roleKey)?.open) {
2033
+ return null; // an In-Progress row for THIS turn already exists; fold this delta into it
1940
2034
  }
2035
+ // A closed entry (a prior turn's finalized row still tracked for streamed re-finals) means
2036
+ // THIS delta begins a NEW turn — fall through and create a fresh In-Progress row, replacing
2037
+ // the tracked entry below.
1941
2038
  const detail = await md.GetEntityObject('MJ: Conversation Details', params.contextUser);
1942
2039
  detail.NewRecord();
1943
2040
  detail.ConversationID = conversationID;
@@ -1954,18 +2051,35 @@ export class BaseAgent {
1954
2051
  });
1955
2052
  return null;
1956
2053
  }
1957
- this.realtimeInFlightTurns.set(roleKey, detail.ID);
2054
+ this.realtimeInFlightTurns.set(roleKey, { id: detail.ID, open: true });
1958
2055
  return detail.ID;
1959
2056
  }
1960
- // ── FINAL: update the in-flight row (or create+finalize when no interim was seen) ─────────
1961
- const inFlightId = this.realtimeInFlightTurns.get(roleKey);
1962
- this.realtimeInFlightTurns.delete(roleKey);
1963
- let detail = await md.GetEntityObject('MJ: Conversation Details', params.contextUser);
1964
- let created = false;
1965
- if (inFlightId && await detail.Load(inFlightId)) {
1966
- // updating the existing streaming row → not a new turn
2057
+ // ── FINAL: update the in-flight row, or create+finalize a fresh turn ──────────────────────
2058
+ // Every real provider shape must yield exactly ONE row per turn:
2059
+ // 1. interim-based (OpenAI): delta(s) open the In-Progress row → final finalizes it;
2060
+ // 2. streamed re-finals (Grok user captions): the SAME turn emits repeated finals, each the
2061
+ // full growing text — the 2nd+ carry ReplacesPrevious=true (stamped by the driver) and
2062
+ // REPLACE the turn's row, not append;
2063
+ // 3. finals-only single (+ ElevenLabs corrections): one non-replacing final, optionally
2064
+ // followed by a ReplacesPrevious correction;
2065
+ // 4. (robustness) a provider that emits BOTH interim deltas AND repeated completeds.
2066
+ //
2067
+ // Reuse the tracked row iff this final REPLACES the turn (ReplacesPrevious) OR the tracked row
2068
+ // is still an OPEN interim (this final finalizes it). A non-replacing final whose tracked entry
2069
+ // is already CLOSED (a prior turn's finalized row) starts a NEW turn. The entry is then KEPT
2070
+ // (closed) rather than deleted, so a later streamed re-final can still update this same row and
2071
+ // the next interim/non-replacing-final correctly detects the turn boundary via `open`.
2072
+ const inFlight = this.realtimeInFlightTurns.get(roleKey);
2073
+ let detail = null;
2074
+ if (inFlight && (transcript.ReplacesPrevious || inFlight.open)) {
2075
+ const candidate = await md.GetEntityObject('MJ: Conversation Details', params.contextUser);
2076
+ if (await candidate.Load(inFlight.id)) {
2077
+ detail = candidate; // update the existing row in place → not a new turn
2078
+ }
1967
2079
  }
1968
- else {
2080
+ let created = false;
2081
+ if (!detail) {
2082
+ detail = await md.GetEntityObject('MJ: Conversation Details', params.contextUser);
1969
2083
  detail.NewRecord();
1970
2084
  detail.ConversationID = conversationID;
1971
2085
  detail.Role = mjRole;
@@ -1981,11 +2095,19 @@ export class BaseAgent {
1981
2095
  if (this.realtimeRecording) {
1982
2096
  detail.UtteranceEndMs = this.realtimeRecording.NowOffsetMs();
1983
2097
  }
1984
- if (!await detail.Save()) {
2098
+ const saved = await detail.Save();
2099
+ if (!saved) {
1985
2100
  this.logError(`Failed to finalize realtime transcript turn: ${detail.LatestResult?.CompleteMessage ?? 'unknown error'}`, {
1986
2101
  agent: params.agent, category: 'RealtimeSession'
1987
2102
  });
1988
2103
  }
2104
+ // Track this turn's now-finalized (closed) row so a subsequent ReplacesPrevious re-final updates
2105
+ // it in place, and so the next interim / non-replacing final detects the new-turn boundary via
2106
+ // `open === false`. Only bind a real id — a failed create leaves an empty id that would poison
2107
+ // the next lookup, so leave the prior entry untouched in that case.
2108
+ if (saved && detail.ID) {
2109
+ this.realtimeInFlightTurns.set(roleKey, { id: detail.ID, open: false });
2110
+ }
1989
2111
  return created ? detail.ID : null;
1990
2112
  }
1991
2113
  /**
@@ -2125,6 +2247,18 @@ export class BaseAgent {
2125
2247
  promptRun.TokensPrompt = usage.InputTokens;
2126
2248
  promptRun.TokensCompletion = usage.OutputTokens;
2127
2249
  promptRun.TokensUsed = usage.InputTokens + usage.OutputTokens;
2250
+ // Per-modality detail (audio vs text vs cached) — REQUIRED for correct multi-channel cost
2251
+ // attribution (audio-in bills ~8x text-in on GPT Realtime 2.1). The realtime prompt run's
2252
+ // Result column is otherwise unused (a live session has no single prompt output), so the
2253
+ // detail rides there as JSON for the cost pipeline / dashboards to consume.
2254
+ if (usage.InputTokenDetails || usage.OutputTokenDetails) {
2255
+ promptRun.Result = JSON.stringify({
2256
+ realtimeUsageDetails: {
2257
+ input: usage.InputTokenDetails ?? null,
2258
+ output: usage.OutputTokenDetails ?? null,
2259
+ },
2260
+ });
2261
+ }
2128
2262
  if (!await promptRun.Save()) {
2129
2263
  this.logError(`Failed to checkpoint realtime usage: ${promptRun.LatestResult?.CompleteMessage ?? 'unknown error'}`, {
2130
2264
  category: 'RealtimeSession'
@@ -2718,7 +2852,6 @@ export class BaseAgent {
2718
2852
  promptParams.templateMessageRole = 'user';
2719
2853
  }
2720
2854
  promptParams.data = promptTemplateData;
2721
- promptParams.agentRunId = this.AgentRun?.ID;
2722
2855
  promptParams.contextUser = params.contextUser;
2723
2856
  promptParams.conversationMessages = params.conversationMessages;
2724
2857
  promptParams.verbose = params.verbose; // Pass through verbose flag
@@ -2779,6 +2912,13 @@ export class BaseAgent {
2779
2912
  else if (this._artifactToolManager.HasArtifacts()) {
2780
2913
  this.logStatus(`[ArtifactTools] Artifacts present but tools disabled by agent config (includeArtifactToolsDocs=false)`, true, params);
2781
2914
  }
2915
+ // Inject conversation-history retrieval tool docs when the run has a
2916
+ // conversation to page against. Like artifact tools, results are pushed as
2917
+ // one-shot conversation messages, never re-rendered per turn.
2918
+ const conversationToolsEnabled = agentTypePromptParams?.includeConversationToolsDocs !== false;
2919
+ if (conversationToolsEnabled && this._conversationToolManager.IsAvailable) {
2920
+ promptParams.data['_CONVERSATION_TOOLS'] = this._conversationToolManager.GetToolDocumentation();
2921
+ }
2782
2922
  // Enable the memory-writes response field + docs only for agents that opted in
2783
2923
  // via AllowMemoryWrite. Disabled agents never see the docs, so a well-behaved
2784
2924
  // LLM never emits the field (the turn loop still guards against drift).
@@ -2820,7 +2960,6 @@ export class BaseAgent {
2820
2960
  conversationMessages: params.conversationMessages,
2821
2961
  templateMessageRole: 'user',
2822
2962
  verbose: params.verbose,
2823
- agentRunId: this.AgentRun?.ID
2824
2963
  };
2825
2964
  // Pass through effortLevel to child prompt (same precedence hierarchy)
2826
2965
  if (params.effortLevel !== undefined && params.effortLevel !== null) {
@@ -4420,17 +4559,382 @@ The context is now within limits. Please retry your request with the recovered c
4420
4559
  },
4421
4560
  });
4422
4561
  const stored = await this._artifactToolManager.ExecuteSingleToolCall(call);
4423
- await this.finalizeStepEntity(toolStep, stored.result.success, stored.result.success ? undefined : stored.result.errorMessage, {
4562
+ // Typed cross-turn contract — see CarryForwardToolStepOutput (read back by
4563
+ // BuildPriorTurnToolResultsMessage on the next run; StepName is display-only).
4564
+ const carryForwardOutput = {
4565
+ toolFamily: CarryForwardToolFamily.Artifact,
4424
4566
  artifactId: stored.artifactId,
4425
4567
  tool: stored.tool,
4426
4568
  input: stored.input,
4427
4569
  result: stored.result,
4428
4570
  durationMs: stored.durationMs,
4429
- });
4571
+ };
4572
+ await this.finalizeStepEntity(toolStep, stored.result.success, stored.result.success ? undefined : stored.result.errorMessage, carryForwardOutput);
4430
4573
  return stored;
4431
4574
  }));
4432
4575
  return results;
4433
4576
  }
4577
+ /**
4578
+ * Carries the PREVIOUS turn's tool results forward into this run's context.
4579
+ *
4580
+ * Inline tool results (artifact + conversation tools) are injected into the run's
4581
+ * in-memory messages only — the next turn rebuilds messages from the conversation
4582
+ * window, so a result paged in on turn N is gone on turn N+1 and the agent must
4583
+ * re-call the tool. The results already persist in each Tool step's OutputData;
4584
+ * this re-injects the immediately previous completed run's successful results as
4585
+ * one transient message. One-turn memory by construction: each run carries only
4586
+ * its direct predecessor's results, so context never compounds.
4587
+ *
4588
+ * Gated on conversationId + root depth — programmatic runs and sub-agents skip it.
4589
+ * @protected
4590
+ */
4591
+ async injectPriorTurnToolResults(params) {
4592
+ if (!params.conversationId || this._depth !== 0) {
4593
+ return;
4594
+ }
4595
+ try {
4596
+ const steps = await this.loadPriorTurnToolResultSteps(params);
4597
+ const body = BaseAgent.BuildPriorTurnToolResultsMessage(steps, this.maxStandaloneToolResultChars);
4598
+ if (!body) {
4599
+ return;
4600
+ }
4601
+ const message = {
4602
+ role: 'user',
4603
+ content: body,
4604
+ metadata: {
4605
+ turnAdded: 0,
4606
+ messageType: BaseAgent.toolResultMessageType,
4607
+ expirationTurns: 2,
4608
+ expirationMode: 'Compact',
4609
+ compactMode: 'First N Chars',
4610
+ compactLength: 500,
4611
+ compactPromptId: '',
4612
+ },
4613
+ };
4614
+ params.conversationMessages.push(message);
4615
+ this.logStatus(`[PriorTurnToolResults] Carried ${steps.length} tool result(s) forward from the previous run`, true, params);
4616
+ }
4617
+ catch (error) {
4618
+ // Carry-forward is an optimization — never let it break the run.
4619
+ this.logStatus(`[PriorTurnToolResults] Skipped (contained error): ${error instanceof Error ? error.message : error}`, true, params);
4620
+ }
4621
+ }
4622
+ /**
4623
+ * Loads this agent's previous settled root run's Tool steps for this conversation
4624
+ * (settled = {@link settledRunStatuses}: Completed or AwaitingFeedback).
4625
+ * Deliberately loads ALL completed Tool steps — eligibility for carry-forward is
4626
+ * decided structurally by {@link BuildPriorTurnToolResultsMessage} via the
4627
+ * `toolFamily` field the executors stamp into OutputData, never by StepName
4628
+ * (which is a display label and free to change).
4629
+ *
4630
+ * Consults {@link PriorTurnToolResultCache} first — the completing run populates it
4631
+ * in {@link finalizeAgentRun} from its in-memory steps, so on this node the common
4632
+ * case (including "prior run made no tool calls") costs zero DB queries; the
4633
+ * RunView pair below is the cache-miss fallback (first turn, restart, other node).
4634
+ * @private
4635
+ */
4636
+ async loadPriorTurnToolResultSteps(params) {
4637
+ const cached = PriorTurnToolResultCache.Instance.Get(params.conversationId, params.agent.ID);
4638
+ if (cached) {
4639
+ this.logStatus(`[PriorTurnToolResults] Prior-run tool results served from cache (${cached.length} step(s), no DB lookup)`, true, params);
4640
+ return cached;
4641
+ }
4642
+ const predicate = BaseAgent.carryForwardPredicate;
4643
+ const statusList = predicate.runStatuses.map(s => `'${s}'`).join(', ');
4644
+ const rv = RunView.FromMetadataProvider(this.ProviderToUse);
4645
+ // AgentID scopes provenance: in a multi-agent conversation, agent B must never
4646
+ // inherit agent A's results labeled "your previous turn".
4647
+ const priorRun = await rv.RunView({
4648
+ EntityName: 'MJ: AI Agent Runs',
4649
+ ExtraFilter: `ConversationID='${params.conversationId}' AND Status IN (${statusList}) AND ParentRunID IS NULL AND AgentID='${params.agent.ID}'`,
4650
+ OrderBy: '__mj_CreatedAt DESC',
4651
+ MaxRows: 1,
4652
+ Fields: ['ID'],
4653
+ ResultType: 'simple',
4654
+ }, params.contextUser);
4655
+ const priorRunId = priorRun.Success ? priorRun.Results?.[0]?.ID : undefined;
4656
+ if (!priorRunId) {
4657
+ return [];
4658
+ }
4659
+ const steps = await rv.RunView({
4660
+ EntityName: 'MJ: AI Agent Run Steps',
4661
+ ExtraFilter: `AgentRunID='${priorRunId}' AND StepType='${predicate.stepType}' AND Status='${predicate.stepStatus}'`,
4662
+ OrderBy: 'StartedAt ASC',
4663
+ Fields: ['OutputData'],
4664
+ ResultType: 'simple',
4665
+ }, params.contextUser);
4666
+ return steps.Success ? (steps.Results || []) : [];
4667
+ }
4668
+ /**
4669
+ * Publishes this run's completed Tool-step results to {@link PriorTurnToolResultCache}
4670
+ * so the conversation's next turn skips the prior-run DB lookups. Applies the same
4671
+ * row predicate as the DB path ({@link carryForwardPredicate}): root runs only, and
4672
+ * only when the run row settled as {@link settledRunStatuses} (Completed OR
4673
+ * AwaitingFeedback — the normal chat-turn ending) — a failed run leaves the previous
4674
+ * settled run's entry standing, just as the RunView filter would. Scoped to this
4675
+ * run's agent (cache key = conversation + agent) so parallel agents in one
4676
+ * conversation never cross-pollinate. An empty projection is cached too (the
4677
+ * negative-cache case that spares tool-free conversations the queries every turn).
4678
+ * Same-node edge semantics (failed step INSERTs, concurrent completions) are
4679
+ * documented on the cache class. Called from {@link finalizeAgentRun}.
4680
+ * @private
4681
+ */
4682
+ cachePriorTurnToolResults() {
4683
+ const predicate = BaseAgent.carryForwardPredicate;
4684
+ const conversationId = this._executeParams?.conversationId;
4685
+ if (!conversationId || this._depth !== 0 || !this._agentRun
4686
+ || !predicate.runStatuses.includes(this._agentRun.Status)) {
4687
+ return;
4688
+ }
4689
+ const records = (this._agentRun.Steps || [])
4690
+ .filter(s => s.StepType === predicate.stepType && s.Status === predicate.stepStatus)
4691
+ .map(s => ({ OutputData: s.OutputData || null }));
4692
+ PriorTurnToolResultCache.Instance.Set(conversationId, this._agentRun.AgentID, records);
4693
+ }
4694
+ /**
4695
+ * Tool families whose step results are eligible for prior-turn carry-forward —
4696
+ * derived from {@link CarryForwardToolFamily} (the single source the stamp sites use).
4697
+ * Read-tool families only: memory writes, pipelines, and client tools also record
4698
+ * `StepType='Tool'` steps but must never be replayed as reusable results.
4699
+ */
4700
+ static { this.CarryForwardToolFamilies = Object.values(CarryForwardToolFamily); }
4701
+ /**
4702
+ * Run statuses that count as a successfully settled root turn. 'AwaitingFeedback' is
4703
+ * included because a Chat final step is the NORMAL per-turn completion for
4704
+ * conversational agents — {@link finalizeAgentRun} maps `step === 'Chat'` to
4705
+ * `Status='AwaitingFeedback'` with `Success=true`, so gating on 'Completed' alone
4706
+ * silently disables post-turn compaction and carry-forward for the most common
4707
+ * agent shape (a chat agent in a long conversation).
4708
+ *
4709
+ * Deliberately `ReadonlyArray<Union>` rather than an `as const` tuple: a narrowed
4710
+ * tuple type would make `.includes(status)` fail to typecheck against the wider
4711
+ * entity union, while this form keeps the compile-time check that each literal is a
4712
+ * valid status (a CHECK-constraint change still surfaces here). Do not "tighten" it.
4713
+ *
4714
+ * Single source for the consumers that must agree: the carry-forward predicate
4715
+ * ({@link carryForwardPredicate} → DB filter + cache-population gate) and the
4716
+ * post-turn compaction gate ({@link startPostTurnCompaction}).
4717
+ */
4718
+ static { this.settledRunStatuses = ['Completed', 'AwaitingFeedback']; }
4719
+ /**
4720
+ * The carry-forward row predicate — the SINGLE source shared by the two places that
4721
+ * must select the same rows or the cache diverges from the DB path: the RunView
4722
+ * `ExtraFilter`s in {@link loadPriorTurnToolResultSteps} (DB fallback) and the
4723
+ * in-memory gate/projection in {@link cachePriorTurnToolResults} (cache population).
4724
+ * Values are typed from the entity unions so a CHECK-constraint change surfaces here
4725
+ * at compile time instead of silently desynchronizing the two loaders.
4726
+ *
4727
+ * The executing agent's ID also scopes both paths (SQL `AgentID=` clause + cache
4728
+ * key) but is per-run data, not a literal contract — it lives at the call sites,
4729
+ * not here.
4730
+ */
4731
+ static { this.carryForwardPredicate = {
4732
+ stepType: 'Tool',
4733
+ stepStatus: 'Completed',
4734
+ runStatuses: BaseAgent.settledRunStatuses,
4735
+ }; }
4736
+ /**
4737
+ * Display name of the seeded system prompt behind summarizeRange's recursive
4738
+ * sub-call (see metadata/prompts/.summarize-range-prompt.json). Resolved with a
4739
+ * trimmed, case-insensitive compare — never an exact-case inline literal.
4740
+ */
4741
+ static { this.SummarizeRangePromptName = 'Summarize Conversation Range'; }
4742
+ /**
4743
+ * The `messageType` marker stamped on injected tool-result messages and matched by
4744
+ * the compaction/pruning eligibility checks — single-sourced so writers and matchers
4745
+ * cannot drift. (Value participates in the AgentChatMessageMetadata union.)
4746
+ */
4747
+ static { this.toolResultMessageType = 'tool-result'; }
4748
+ /**
4749
+ * Header stems for injected tool-result messages. These exact headers are a contract:
4750
+ * the loop-agent system template (loop-agent-type-system-prompt.template.md, "header
4751
+ * `Conversation history tool result:`" / "`Artifact tool result:`") teaches the model
4752
+ * to recognize them — change the template in lockstep.
4753
+ */
4754
+ static conversationToolResultsHeader(count) {
4755
+ return count === 1 ? 'Conversation history tool result:' : `Conversation history tool results (${count} calls):`;
4756
+ }
4757
+ /** Artifact analog of {@link conversationToolResultsHeader} — same template contract. */
4758
+ static artifactToolResultsHeader(count) {
4759
+ return count === 1 ? 'Artifact tool result:' : `Artifact tool results (${count} calls):`;
4760
+ }
4761
+ /**
4762
+ * Renders prior-turn tool-result steps into the carried-forward message body.
4763
+ * Pure and static for testability: keeps only steps whose OutputData satisfies the
4764
+ * structured contract stamped by the tool executors — a carry-forward-eligible
4765
+ * `toolFamily` (see {@link CarryForwardToolFamilies}) AND a non-empty `tool` name.
4766
+ * Tolerant of missing/invalid OutputData JSON, keeps only successful results,
4767
+ * caps each result and the total under `maxChars` (adding an explicit truncation
4768
+ * note when results are dropped). Returns null when nothing usable remains.
4769
+ */
4770
+ static BuildPriorTurnToolResultsMessage(steps, maxChars) {
4771
+ const sections = [];
4772
+ let usedChars = 0;
4773
+ let dropped = 0;
4774
+ for (const step of steps) {
4775
+ if (!step.OutputData)
4776
+ continue;
4777
+ let parsed;
4778
+ try {
4779
+ parsed = JSON.parse(step.OutputData);
4780
+ }
4781
+ catch {
4782
+ continue;
4783
+ }
4784
+ if (!parsed.toolFamily || !BaseAgent.CarryForwardToolFamilies.includes(parsed.toolFamily))
4785
+ continue;
4786
+ if (typeof parsed.tool !== 'string' || parsed.tool.length === 0)
4787
+ continue;
4788
+ if (parsed.result?.success !== true)
4789
+ continue;
4790
+ const section = FormatToolResultSection({ tool: parsed.tool, input: parsed.input }, RenderToolResultData(parsed.result.data));
4791
+ if (usedChars + section.length > maxChars && sections.length > 0) {
4792
+ dropped++;
4793
+ continue;
4794
+ }
4795
+ const capped = section.length > maxChars
4796
+ ? `${section.slice(0, maxChars)}\n[truncated]`
4797
+ : section;
4798
+ usedChars += capped.length;
4799
+ sections.push(capped);
4800
+ }
4801
+ if (sections.length === 0) {
4802
+ return null;
4803
+ }
4804
+ const header = 'Tool results from your previous turn (still valid — reuse instead of re-calling):';
4805
+ const droppedNote = dropped > 0 ? `\n\n[${dropped} additional result(s) omitted for size — re-call those tools if needed]` : '';
4806
+ return `${header}\n${sections.join('\n\n')}${droppedNote}`;
4807
+ }
4808
+ /**
4809
+ * Builds the summarizeRange recursive-sub-call host: resolves the seeded
4810
+ * 'Summarize Conversation Range' prompt (priority-ordered cheap models — the RLM
4811
+ * "strong root model, cheap sub-call model" split) and runs it via the standard
4812
+ * prompt runner so the AIPromptRun records itself.
4813
+ * @protected
4814
+ */
4815
+ buildConversationSummaryHost(params) {
4816
+ return {
4817
+ RunSummaryPrompt: async (rangeText, lens) => {
4818
+ // Trimmed, case-insensitive name lookup (the AIPromptRunner 'Repair JSON'
4819
+ // style) so cosmetic re-casing of the seeded prompt can't break the tool.
4820
+ const targetName = BaseAgent.SummarizeRangePromptName.toLowerCase();
4821
+ const prompt = AIEngine.Instance.Prompts.find(p => p.Name.trim().toLowerCase() === targetName);
4822
+ if (!prompt) {
4823
+ throw new Error(`The '${BaseAgent.SummarizeRangePromptName}' system prompt is not present in this environment`);
4824
+ }
4825
+ const promptParams = new AIPromptParams();
4826
+ promptParams.prompt = prompt;
4827
+ // Keys are the summarize-range.template.md contract ({{ lens }}, {{ messages }})
4828
+ promptParams.data = { lens, messages: rangeText };
4829
+ promptParams.contextUser = params.contextUser;
4830
+ const result = await this._promptRunner.ExecutePrompt(promptParams);
4831
+ const text = ExtractPromptResultText(result);
4832
+ if (!result.success || text.length === 0) {
4833
+ throw new Error(result.errorMessage || 'summarizeRange sub-call returned no content');
4834
+ }
4835
+ return { text, promptRunId: result.promptRun?.ID };
4836
+ }
4837
+ };
4838
+ }
4839
+ /**
4840
+ * Executes conversation-history retrieval tool calls, wrapping each invocation in
4841
+ * its own AIAgentRunStep (StepType='Tool', "Conversation Tool: {tool}") — the same
4842
+ * per-call observability shape as artifact tools. Reads are served from the
4843
+ * ConversationEngine cache; per-call failures are contained in the result.
4844
+ *
4845
+ * At most {@link MAX_CONVERSATION_TOOL_CALLS_PER_TURN} calls execute per response;
4846
+ * the excess come back as skipped failure-shaped results (no run steps recorded)
4847
+ * telling the model to re-request them next turn.
4848
+ *
4849
+ * @protected
4850
+ */
4851
+ async executeConversationToolCallsAsSteps(calls, params) {
4852
+ // Per-turn fan-out cap: each call is a run step (summarizeRange a full LLM
4853
+ // sub-call) — excess calls are reported back as skipped failure-shaped results
4854
+ // through the normal rendering path so the model can re-request them next turn.
4855
+ // Deliberately no DB rows for skipped calls (zero I/O for work not done; the
4856
+ // Status union has no 'Skipped' and Failed steps would pollute failure metrics).
4857
+ const callsToExecute = calls.slice(0, MAX_CONVERSATION_TOOL_CALLS_PER_TURN);
4858
+ const skippedCalls = calls.slice(MAX_CONVERSATION_TOOL_CALLS_PER_TURN);
4859
+ if (skippedCalls.length > 0) {
4860
+ this.logStatus(`[ConversationTools] ${calls.length} calls requested — executing first ${callsToExecute.length}, skipping ${skippedCalls.length} (per-turn cap)`, true, params);
4861
+ }
4862
+ const executedResults = await Promise.all(callsToExecute.map(async (call) => {
4863
+ const toolStep = await this.createStepEntity({
4864
+ stepType: 'Tool',
4865
+ stepName: `Conversation Tool: ${call.tool}`,
4866
+ contextUser: params.contextUser,
4867
+ inputData: {
4868
+ tool: call.tool,
4869
+ input: call.input,
4870
+ conversationId: params.conversationId,
4871
+ },
4872
+ });
4873
+ const executed = await this._conversationToolManager.ExecuteSingleToolCall(call);
4874
+ // summarizeRange's recursive LLM sub-call records an AIPromptRun — link it
4875
+ // through this Tool step's TargetLogID (one step + one prompt run: full
4876
+ // lineage without a duplicate Prompt step for the same call).
4877
+ if (executed.promptRunId) {
4878
+ toolStep.TargetLogID = executed.promptRunId;
4879
+ }
4880
+ // Typed cross-turn contract — see CarryForwardToolStepOutput (read back by
4881
+ // BuildPriorTurnToolResultsMessage on the next run; StepName is display-only).
4882
+ const carryForwardOutput = {
4883
+ toolFamily: CarryForwardToolFamily.Conversation,
4884
+ tool: executed.tool,
4885
+ input: executed.input,
4886
+ result: executed.result,
4887
+ durationMs: executed.durationMs,
4888
+ ...(executed.promptRunId && { promptRunId: executed.promptRunId }),
4889
+ };
4890
+ await this.finalizeStepEntity(toolStep, executed.result.success, executed.result.success ? undefined : executed.result.errorMessage, carryForwardOutput);
4891
+ return executed;
4892
+ }));
4893
+ const skippedResults = skippedCalls.map(call => ({
4894
+ tool: call.tool,
4895
+ input: call.input,
4896
+ result: {
4897
+ success: false,
4898
+ errorMessage: `Skipped — per-turn cap of ${MAX_CONVERSATION_TOOL_CALLS_PER_TURN} conversation tool calls reached. Re-request this call on your next turn.`,
4899
+ },
4900
+ durationMs: 0,
4901
+ }));
4902
+ return [...executedResults, ...skippedResults];
4903
+ }
4904
+ /**
4905
+ * Pushes a single user-role message containing rendered conversation-tool results
4906
+ * into the conversation — the same inject-once-then-expire lifecycle as artifact
4907
+ * tool results.
4908
+ *
4909
+ * @protected
4910
+ */
4911
+ injectConversationToolResultsMessage(params, toolResults) {
4912
+ if (toolResults.length === 0)
4913
+ return;
4914
+ const header = BaseAgent.conversationToolResultsHeader(toolResults.length);
4915
+ const body = toolResults.map((r, i) => {
4916
+ const parts = { tool: r.tool, input: r.input, ordinal: i + 1 };
4917
+ if (r.result.success) {
4918
+ const data = this.capStandaloneToolResultText(RenderToolResultData(r.result.data));
4919
+ return FormatToolResultSection(parts, data);
4920
+ }
4921
+ return FormatToolErrorSection(parts, r.result.errorMessage);
4922
+ }).join('\n\n');
4923
+ const message = {
4924
+ role: 'user',
4925
+ content: `${header}\n${body}`,
4926
+ metadata: {
4927
+ turnAdded: this._promptTurnCount,
4928
+ messageType: BaseAgent.toolResultMessageType,
4929
+ expirationTurns: 3,
4930
+ expirationMode: 'Compact',
4931
+ compactMode: 'First N Chars',
4932
+ compactLength: 500,
4933
+ compactPromptId: '',
4934
+ },
4935
+ };
4936
+ params.conversationMessages.push(message);
4937
+ }
4434
4938
  /**
4435
4939
  * Pushes a single user-role message containing rendered artifact-tool
4436
4940
  * results into the conversation. This mirrors the action-result
@@ -4444,19 +4948,14 @@ The context is now within limits. Please retry your request with the recovered c
4444
4948
  injectArtifactToolResultsMessage(params, toolResults) {
4445
4949
  if (toolResults.length === 0)
4446
4950
  return;
4447
- const header = toolResults.length === 1
4448
- ? 'Artifact tool result:'
4449
- : `Artifact tool results (${toolResults.length} calls):`;
4951
+ const header = BaseAgent.artifactToolResultsHeader(toolResults.length);
4450
4952
  const body = toolResults.map((r, i) => {
4451
- const heading = `### ${i + 1}. ${r.artifactId}.${r.tool}(${JSON.stringify(r.input)})`;
4953
+ const parts = { tool: r.tool, input: r.input, ordinal: i + 1, signaturePrefix: r.artifactId };
4452
4954
  if (r.result.success) {
4453
- const raw = typeof r.result.data === 'string'
4454
- ? r.result.data
4455
- : JSON.stringify(r.result.data, null, 2);
4456
- const data = this.capStandaloneToolResultText(raw);
4457
- return `${heading}\n\`\`\`json\n${data}\n\`\`\``;
4955
+ const data = this.capStandaloneToolResultText(RenderToolResultData(r.result.data));
4956
+ return FormatToolResultSection(parts, data);
4458
4957
  }
4459
- return `${heading}\n**Error:** ${r.result.errorMessage}`;
4958
+ return FormatToolErrorSection(parts, r.result.errorMessage);
4460
4959
  }).join('\n\n');
4461
4960
  const message = {
4462
4961
  role: 'user',
@@ -5082,6 +5581,7 @@ The context is now within limits. Please retry your request with the recovered c
5082
5581
  { docsFlag: 'includeWhileDocs', responseTypeKey: 'while' },
5083
5582
  { docsFlag: 'includeScratchpadDocs', responseTypeKey: 'scratchpad' },
5084
5583
  { docsFlag: 'includeArtifactToolsDocs', responseTypeKey: 'artifactToolCalls' },
5584
+ { docsFlag: 'includeConversationToolsDocs', responseTypeKey: 'conversationToolCalls' },
5085
5585
  { docsFlag: 'includePipelineDocs', responseTypeKey: 'pipeline' },
5086
5586
  { docsFlag: 'includeMemoryWritesDocs', responseTypeKey: 'memoryWrites' }
5087
5587
  ];
@@ -6331,6 +6831,16 @@ The context is now within limits. Please retry your request with the recovered c
6331
6831
  if (skillsForStep && skillsForStep.length > 0) {
6332
6832
  stepEntity.Skills = JSON.stringify(skillsForStep);
6333
6833
  }
6834
+ // Completed-at-creation steps: stamp the terminal state NOW so the INSERT below is the
6835
+ // step's ONLY write (same shared helper + OutputData treatment finalizeStepEntity uses).
6836
+ if (params.completed) {
6837
+ finalizeAgentRunStep(stepEntity, {
6838
+ success: params.completed.success,
6839
+ errorMessage: params.completed.errorMessage,
6840
+ outputData: params.completed.outputData ? CopyScalarsAndArrays(params.completed.outputData, true) : undefined,
6841
+ completedAt: new Date()
6842
+ });
6843
+ }
6334
6844
  // Fire-and-forget the 'started' INSERT — the agent flow never blocks on a step save. The queue
6335
6845
  // tracks the INSERT so every later UPDATE (queueStepSave) chains AFTER it commits.
6336
6846
  // When the step has a parent, chain the INSERT AFTER the parent's INSERT to satisfy the
@@ -6607,7 +7117,19 @@ The context is now within limits. Please retry your request with the recovered c
6607
7117
  // Check if this is a message expansion request
6608
7118
  if (previousDecision.messageIndex !== undefined) {
6609
7119
  // Handle message expansion before retrying
6610
- this.executeExpandMessageStep(previousDecision, params, this._promptTurnCount);
7120
+ const expandFailure = this.executeExpandMessageStep(previousDecision, params, this._promptTurnCount);
7121
+ if (expandFailure) {
7122
+ // A failed expansion MUST NOT leave the loop state unchanged: the model
7123
+ // re-requests the identical expansion forever (observed live when a
7124
+ // spliced cross-turn summary message — which has no expanded form — was
7125
+ // requested for expansion; the silent no-op produced an unbounded Retry
7126
+ // loop that exhausted the process heap). Surface the failure into the
7127
+ // conversation so the next prompt steers the model away.
7128
+ params.conversationMessages.push({
7129
+ role: 'user',
7130
+ content: `Message expansion failed: ${expandFailure}`
7131
+ });
7132
+ }
6611
7133
  }
6612
7134
  return await this.executePromptStep(params, config, previousDecision, stepCount);
6613
7135
  case 'Sub-Agent':
@@ -6839,6 +7361,12 @@ The context is now within limits. Please retry your request with the recovered c
6839
7361
  stepEntity.PromptRun = promptResult.promptRun; // transient related object (not a persisted field)
6840
7362
  // don't save here, we save when we call finalizeStepEntity()
6841
7363
  }
7364
+ // Remember the most recent model selection — cross-turn compaction resolves its
7365
+ // effective budget against "the model about to run", and the last prompt's
7366
+ // selection is the best available proxy for the next turn's model.
7367
+ if (promptResult.modelSelectionInfo) {
7368
+ this._lastModelSelectionInfo = promptResult.modelSelectionInfo;
7369
+ }
6842
7370
  // Check if prompt execution failed
6843
7371
  if (!promptResult.success) {
6844
7372
  // CRITICAL FIX: Preserve payload before finalizing step
@@ -6953,6 +7481,19 @@ The context is now within limits. Please retry your request with the recovered c
6953
7481
  else if (this._artifactToolManager.HasArtifacts()) {
6954
7482
  this.logStatus(`[ArtifactTools] LLM did not use artifact tools this turn (artifacts available but not accessed)`, true, params);
6955
7483
  }
7484
+ // Execute conversation-history retrieval tool calls if provided (zero turn cost —
7485
+ // processed inline, results delivered as a conversation message next turn)
7486
+ const conversationToolCalls = initialNextStep.conversationToolCalls;
7487
+ if (conversationToolCalls?.length) {
7488
+ if (this._conversationToolManager.IsAvailable) {
7489
+ this.logStatus(`[ConversationTools] LLM requested ${conversationToolCalls.length} tool call(s): ${conversationToolCalls.map(c => c.tool).join(', ')}`, true, params);
7490
+ const conversationToolResults = await this.executeConversationToolCallsAsSteps(conversationToolCalls, params);
7491
+ this.injectConversationToolResultsMessage(params, conversationToolResults);
7492
+ }
7493
+ else {
7494
+ this.logStatus(`[ConversationTools] LLM requested conversation tools but the run has no conversationId — ignored`, true, params);
7495
+ }
7496
+ }
6956
7497
  // Execute in-flight memory writes if provided (zero turn cost — processed inline)
6957
7498
  const memoryWrites = initialNextStep.memoryWrites;
6958
7499
  if (memoryWrites?.length) {
@@ -10168,13 +10709,7 @@ The context is now within limits. Please retry your request with the recovered c
10168
10709
  this._agentRun.Success = false;
10169
10710
  this._agentRun.ErrorMessage = errorMessage;
10170
10711
  // Calculate total tokens even for failed runs
10171
- const tokenStats = this.calculateTokenStats();
10172
- this._agentRun.TotalTokensUsed = tokenStats.totalTokens;
10173
- this._agentRun.TotalPromptTokensUsed = tokenStats.promptTokens;
10174
- this._agentRun.TotalCompletionTokensUsed = tokenStats.completionTokens;
10175
- this._agentRun.TotalCacheReadTokensUsed = tokenStats.cacheReadTokens;
10176
- this._agentRun.TotalCacheWriteTokensUsed = tokenStats.cacheWriteTokens;
10177
- this._agentRun.TotalCost = tokenStats.totalCost;
10712
+ this.applyTokenStatsToRun(this._agentRun, this.calculateTokenStats());
10178
10713
  await this._agentRun.Save();
10179
10714
  }
10180
10715
  return {
@@ -10196,13 +10731,7 @@ The context is now within limits. Please retry your request with the recovered c
10196
10731
  this._agentRun.Success = false;
10197
10732
  this._agentRun.ErrorMessage = message;
10198
10733
  // Calculate total tokens even for cancelled runs
10199
- const tokenStats = this.calculateTokenStats();
10200
- this._agentRun.TotalTokensUsed = tokenStats.totalTokens;
10201
- this._agentRun.TotalPromptTokensUsed = tokenStats.promptTokens;
10202
- this._agentRun.TotalCompletionTokensUsed = tokenStats.completionTokens;
10203
- this._agentRun.TotalCacheReadTokensUsed = tokenStats.cacheReadTokens;
10204
- this._agentRun.TotalCacheWriteTokensUsed = tokenStats.cacheWriteTokens;
10205
- this._agentRun.TotalCost = tokenStats.totalCost;
10734
+ this.applyTokenStatsToRun(this._agentRun, this.calculateTokenStats());
10206
10735
  await this._agentRun.Save();
10207
10736
  }
10208
10737
  return {
@@ -10272,17 +10801,21 @@ The context is now within limits. Please retry your request with the recovered c
10272
10801
  this._agentRun.FinalPayloadObject = resolvedPayload;
10273
10802
  this._agentRun.FinalPayload = finalPayloadJson;
10274
10803
  // Calculate total tokens from all prompts and sub-agents
10275
- const tokenStats = this.calculateTokenStats();
10276
- this._agentRun.TotalTokensUsed = tokenStats.totalTokens;
10277
- this._agentRun.TotalPromptTokensUsed = tokenStats.promptTokens;
10278
- this._agentRun.TotalCompletionTokensUsed = tokenStats.completionTokens;
10279
- this._agentRun.TotalCacheReadTokensUsed = tokenStats.cacheReadTokens;
10280
- this._agentRun.TotalCacheWriteTokensUsed = tokenStats.cacheWriteTokens;
10281
- this._agentRun.TotalCost = tokenStats.totalCost;
10804
+ this.applyTokenStatsToRun(this._agentRun, this.calculateTokenStats());
10282
10805
  const ok = await this._agentRun.Save();
10283
10806
  if (!ok) {
10284
10807
  LogError(`Failed to finalize agent run ${this._agentRun.ID}`);
10285
10808
  }
10809
+ else {
10810
+ // Hand the NEXT turn's carry-forward check this run's tool results straight
10811
+ // from memory, so it can skip its DB lookups (see PriorTurnToolResultCache).
10812
+ this.cachePriorTurnToolResults();
10813
+ }
10814
+ // Cross-turn compaction (post-turn, the primary path): fire-and-forget AFTER the
10815
+ // run row is final so the summary-LLM latency never delays the caller's
10816
+ // completion event. Errors are contained — a failed pass leaves the conversation
10817
+ // untouched and simply re-triggers on a later turn.
10818
+ this.startPostTurnCompaction();
10286
10819
  }
10287
10820
  // Also promote any media from the final step's promoteMediaOutputs
10288
10821
  if (finalStep.promoteMediaOutputs && finalStep.promoteMediaOutputs.length > 0) {
@@ -10322,7 +10855,7 @@ The context is now within limits. Please retry your request with the recovered c
10322
10855
  // Iterate through the agent run's steps to sum up tokens
10323
10856
  if (this._agentRun?.Steps) {
10324
10857
  for (const step of this._agentRun.Steps) {
10325
- if (step.StepType === 'Prompt' && step.PromptRun) {
10858
+ if ((step.StepType === 'Prompt' || step.StepType === 'Compaction') && step.PromptRun) {
10326
10859
  // Add tokens from prompt runs (rollup fields include any nested child prompt runs)
10327
10860
  totalTokens += step.PromptRun.TokensUsedRollup || 0;
10328
10861
  promptTokens += step.PromptRun.TokensPromptRollup || 0;
@@ -10344,6 +10877,21 @@ The context is now within limits. Please retry your request with the recovered c
10344
10877
  }
10345
10878
  return { totalTokens, promptTokens, completionTokens, cacheReadTokens, cacheWriteTokens, totalCost };
10346
10879
  }
10880
+ /**
10881
+ * Applies a {@link calculateTokenStats} result to a run entity's six denormalized
10882
+ * token/cost columns — the single source for the assignment shape shared by the
10883
+ * failure/cancel/finalize paths AND the post-turn compaction top-up
10884
+ * ({@link recordCompactionRunStep}).
10885
+ * @private
10886
+ */
10887
+ applyTokenStatsToRun(run, tokenStats) {
10888
+ run.TotalTokensUsed = tokenStats.totalTokens;
10889
+ run.TotalPromptTokensUsed = tokenStats.promptTokens;
10890
+ run.TotalCompletionTokensUsed = tokenStats.completionTokens;
10891
+ run.TotalCacheReadTokensUsed = tokenStats.cacheReadTokens;
10892
+ run.TotalCacheWriteTokensUsed = tokenStats.cacheWriteTokens;
10893
+ run.TotalCost = tokenStats.totalCost;
10894
+ }
10347
10895
  /**
10348
10896
  * Gets the count of how many times a specific action has been executed in this agent run.
10349
10897
  *
@@ -10549,6 +11097,224 @@ The context is now within limits. Please retry your request with the recovered c
10549
11097
  `${messagesToRemove.length} removed`);
10550
11098
  }
10551
11099
  }
11100
+ // =====================================================================================
11101
+ // CROSS-TURN (TIER A) CONVERSATION COMPACTION HOOKS
11102
+ // Durable summary layer per plans/agent-conversation-compaction.md. All hooks are
11103
+ // gated on params.conversationId + root depth — programmatic runs, sub-agents, and
11104
+ // tests without a conversation are untouched. Trigger math / boundary selection /
11105
+ // the boundary-row write live in ConversationCompactionManager; BaseAgent owns
11106
+ // budget resolution (it knows the model) and run-step recording.
11107
+ // =====================================================================================
11108
+ /**
11109
+ * Resolves the effective context budget for cross-turn compaction, validated against
11110
+ * the most recent prompt's model when available. Logs the clamp warning once when a
11111
+ * configured budget exceeded the model's MaxInputTokens.
11112
+ * @protected
11113
+ */
11114
+ resolveCompactionBudget(params, config) {
11115
+ const modelMax = this._lastModelSelectionInfo
11116
+ ? this.tryGetModelMaxInputTokens(this._lastModelSelectionInfo)
11117
+ : null;
11118
+ const budget = ConversationCompactionManager.ResolveEffectiveBudget(params.agent, config?.agentType || null, modelMax);
11119
+ if (budget.ClampedToModel) {
11120
+ // Verbose-only: this re-evaluates every turn while the budget stays mis-set, and
11121
+ // the clamp is already captured structurally in CompactionOutcome.Warnings → the
11122
+ // Compaction step's OutputData (§8: keep debug detail, don't spam info logs).
11123
+ this.logStatus(`⚠️ [CrossTurnCompaction] Configured ContextWindowMaxTokens exceeds the model's MaxInputTokens — clamped to ${budget.MaxTokens}`, true, params);
11124
+ }
11125
+ return budget;
11126
+ }
11127
+ /**
11128
+ * Pre-turn fallback: compacts synchronously when the assembled window is already over
11129
+ * the trigger budget BEFORE the first prompt of this run, then splices the fresh
11130
+ * summary into the live message array. Only runs with an EXPLICIT configured budget
11131
+ * (agent or type ContextWindowMaxTokens) — before the first prompt the model is
11132
+ * unknown, and compacting against the conservative default would over-trigger on
11133
+ * large-context models. The post-turn hook (real model known) covers those.
11134
+ * @protected
11135
+ */
11136
+ async checkPreTurnCompaction(params, config) {
11137
+ if (!params.conversationId || this._depth !== 0) {
11138
+ return;
11139
+ }
11140
+ const budget = this.resolveCompactionBudget(params, config);
11141
+ if (budget.BoundedBy !== 'Agent' && budget.BoundedBy !== 'AgentType') {
11142
+ return;
11143
+ }
11144
+ const estimatedTokens = this.estimateConversationTokens(params.conversationMessages);
11145
+ if (estimatedTokens < budget.TriggerTokens) {
11146
+ return;
11147
+ }
11148
+ this.logStatus(`🗜️ [CrossTurnCompaction] Pre-turn window ~${estimatedTokens} tokens ≥ trigger ${budget.TriggerTokens} — compacting before first prompt`, true, params);
11149
+ const outcome = await this.runCrossTurnCompaction('pre-turn', params, config, budget);
11150
+ if (outcome?.Fired && outcome.BoundarySequence !== undefined && outcome.SummaryText) {
11151
+ this.applyCompactionToLiveMessages(params.conversationMessages, outcome.BoundarySequence, outcome.SummaryText);
11152
+ }
11153
+ }
11154
+ /**
11155
+ * Post-turn hook (the primary path), called from {@link finalizeAgentRun} after the
11156
+ * run row is saved. Fire-and-forget by design: the caller's completion event never
11157
+ * waits on the summary LLM call. Fires for settled root runs with a conversation —
11158
+ * {@link settledRunStatuses}: 'Completed' AND 'AwaitingFeedback', because a Chat
11159
+ * final step (→ AwaitingFeedback) is the NORMAL ending of a conversational turn;
11160
+ * gating on 'Completed' alone silently disabled post-turn compaction for exactly
11161
+ * the long-chat scenario this feature targets.
11162
+ * @protected
11163
+ */
11164
+ startPostTurnCompaction() {
11165
+ const params = this._executeParams;
11166
+ if (!params?.conversationId || this._depth !== 0 || !this._agentRun
11167
+ || !BaseAgent.settledRunStatuses.includes(this._agentRun.Status)) {
11168
+ return;
11169
+ }
11170
+ const config = this._agentConfig;
11171
+ const budget = this.resolveCompactionBudget(params, config);
11172
+ void this.runCrossTurnCompaction('post-turn', params, config, budget).catch(error => {
11173
+ LogError(`Post-turn cross-turn compaction error (contained): ${error instanceof Error ? error.message : error}`);
11174
+ });
11175
+ }
11176
+ /**
11177
+ * Runs one compaction pass and records it as a `StepType='Compaction'` run step —
11178
+ * TargetID = the summary prompt, TargetLogID = the summary AIPromptRun (the same ID
11179
+ * written to `ConversationDetail.SummaryPromptRunID`, closing the lineage chain).
11180
+ * Quiet no-ops (window under trigger) record no step; fired passes and failures do.
11181
+ * @protected
11182
+ */
11183
+ async runCrossTurnCompaction(phase, params, config, budget) {
11184
+ if (!params.conversationId || !this._agentRun) {
11185
+ return undefined;
11186
+ }
11187
+ const outcome = await ConversationCompactionManager.CompactIfNeeded({
11188
+ ConversationId: params.conversationId,
11189
+ Agent: params.agent,
11190
+ AgentType: config?.agentType || null,
11191
+ Budget: budget,
11192
+ ContextUser: params.contextUser,
11193
+ Provider: this.ProviderToUse,
11194
+ EstimateTokens: (messages) => this.estimateConversationTokens(messages),
11195
+ Verbose: params.verbose,
11196
+ // The in-flight agent-response placeholder row: a post-turn pass runs while
11197
+ // the resolver may still be writing its Message — keep it out of the window
11198
+ // so the boundary can never land on it.
11199
+ ExcludeDetailIds: params.conversationDetailId ? [params.conversationDetailId] : undefined,
11200
+ });
11201
+ if (outcome.Fired || outcome.ErrorMessage) {
11202
+ await this.recordCompactionRunStep(phase, params, budget, outcome);
11203
+ }
11204
+ return outcome;
11205
+ }
11206
+ /**
11207
+ * Persists the Compaction run step for a fired or failed pass — as a SINGLE INSERT:
11208
+ * the pass is already over when this is called, so the step is created pre-finalized
11209
+ * via `createStepEntity`'s `completed` option instead of paying a second UPDATE
11210
+ * round trip. The summary AIPromptRun rides on the step's transient `PromptRun` so
11211
+ * {@link calculateTokenStats}'s Compaction branch counts it: pre-turn fires are
11212
+ * picked up by finalizeAgentRun's normal rollup for free; post-turn fires happen
11213
+ * AFTER that rollup ran, so this method tops the run's token columns up itself.
11214
+ * @private
11215
+ */
11216
+ async recordCompactionRunStep(phase, params, budget, outcome) {
11217
+ try {
11218
+ const stepEntity = await this.createStepEntity({
11219
+ stepType: 'Compaction',
11220
+ stepName: `Cross-Turn Conversation Compaction (${phase})`,
11221
+ contextUser: params.contextUser,
11222
+ targetId: outcome.PromptId,
11223
+ targetLogId: outcome.PromptRunId,
11224
+ inputData: {
11225
+ phase,
11226
+ conversationId: params.conversationId,
11227
+ budget
11228
+ },
11229
+ completed: {
11230
+ success: !outcome.ErrorMessage,
11231
+ errorMessage: outcome.ErrorMessage,
11232
+ outputData: {
11233
+ fired: outcome.Fired,
11234
+ boundarySequence: outcome.BoundarySequence,
11235
+ tokensBefore: outcome.TokensBefore,
11236
+ tokensAfter: outcome.TokensAfter,
11237
+ summaryLength: outcome.SummaryText?.length,
11238
+ promptRunId: outcome.PromptRunId,
11239
+ warnings: outcome.Warnings
11240
+ }
11241
+ }
11242
+ });
11243
+ if (outcome.PromptRun) {
11244
+ stepEntity.PromptRun = outcome.PromptRun;
11245
+ }
11246
+ if (phase === 'post-turn') {
11247
+ // finalizeAgentRun's flush has already run — drain this step's INSERT now.
11248
+ await this._stepSaveQueue.Flush();
11249
+ await this.topUpRunTokenTotalsAfterPostTurnCompaction(outcome, params.contextUser);
11250
+ }
11251
+ }
11252
+ catch (error) {
11253
+ LogError(`Failed to record Compaction run step (compaction itself ${outcome.Fired ? 'succeeded' : 'failed'}): ${error instanceof Error ? error.message : error}`);
11254
+ }
11255
+ }
11256
+ /**
11257
+ * After a fired POST-turn compaction, folds the summary prompt's tokens/cost into
11258
+ * the run row — finalizeAgentRun's rollup ran before the pass, so without this the
11259
+ * recursive summary spend would be missing from the run's denormalized totals.
11260
+ * Uses a FRESH-loaded run entity for the write: the persisted run may be
11261
+ * 'AwaitingFeedback' and a quick user reply could have resumed it — re-Saving the
11262
+ * stale in-memory `_agentRun` would clobber the resumed row's Status. Residual: the
11263
+ * token columns are last-writer-wins in the tiny Load→Save window (self-healing at
11264
+ * the resumed run's own finalize). Failures are contained (LogError only).
11265
+ * @private
11266
+ */
11267
+ async topUpRunTokenTotalsAfterPostTurnCompaction(outcome, contextUser) {
11268
+ if (!outcome.Fired || !outcome.PromptRun || !this._agentRun) {
11269
+ return;
11270
+ }
11271
+ const tokenStats = this.calculateTokenStats();
11272
+ const runUpdate = await this._activeProvider.GetEntityObject('MJ: AI Agent Runs', contextUser);
11273
+ if (!(await runUpdate.Load(this._agentRun.ID))) {
11274
+ LogError(`Post-turn compaction token top-up: failed to load run ${this._agentRun.ID}`);
11275
+ return;
11276
+ }
11277
+ this.applyTokenStatsToRun(runUpdate, tokenStats);
11278
+ if (!(await runUpdate.Save())) {
11279
+ LogError(`Post-turn compaction token top-up: save failed for run ${this._agentRun.ID}: ${runUpdate.LatestResult?.CompleteMessage || 'unknown error'}`);
11280
+ }
11281
+ }
11282
+ /**
11283
+ * Splices a freshly generated summary into the live message array in place: every
11284
+ * message covered by the new boundary (sequence below it, or a prior summary
11285
+ * message) collapses into one summary message; enrichment-bearing tail messages and
11286
+ * injected messages without sequence metadata are preserved untouched.
11287
+ * @private
11288
+ */
11289
+ applyCompactionToLiveMessages(messages, boundarySequence, summaryText) {
11290
+ const retained = [];
11291
+ let summaryInserted = false;
11292
+ for (const message of messages) {
11293
+ const metadata = message.metadata;
11294
+ const covered = metadata?.isConversationSummary === true
11295
+ || (metadata?.sequence !== undefined && metadata.sequence < boundarySequence);
11296
+ if (covered) {
11297
+ if (!summaryInserted) {
11298
+ const summaryMessage = {
11299
+ role: 'user',
11300
+ content: summaryText,
11301
+ metadata: {
11302
+ isConversationSummary: true,
11303
+ summaryBoundarySequence: boundarySequence,
11304
+ sequence: boundarySequence
11305
+ }
11306
+ };
11307
+ retained.push(summaryMessage);
11308
+ summaryInserted = true;
11309
+ }
11310
+ }
11311
+ else {
11312
+ retained.push(message);
11313
+ }
11314
+ }
11315
+ messages.length = 0;
11316
+ messages.push(...retained);
11317
+ }
10552
11318
  /**
10553
11319
  * Creates an AIAgentRunStep for message compaction operations.
10554
11320
  * Records the compaction attempt with context about the message being compacted.
@@ -10566,7 +11332,7 @@ The context is now within limits. Please retry your request with the recovered c
10566
11332
  const step = await (params.provider || this._activeProvider).GetEntityObject('MJ: AI Agent Run Steps', params.contextUser);
10567
11333
  step.NewRecord();
10568
11334
  step.AgentRunID = this._agentRun.ID;
10569
- step.StepType = 'Prompt';
11335
+ step.StepType = 'Compaction';
10570
11336
  step.Status = 'Running';
10571
11337
  step.InputData = JSON.stringify({
10572
11338
  stepName: 'Message Compaction',
@@ -10708,7 +11474,7 @@ The context is now within limits. Please retry your request with the recovered c
10708
11474
  const messageType = msg.metadata?.messageType;
10709
11475
  return messageType === 'action-result'
10710
11476
  || messageType === 'client-tool-result'
10711
- || messageType === 'tool-result';
11477
+ || messageType === BaseAgent.toolResultMessageType;
10712
11478
  }
10713
11479
  /**
10714
11480
  * Returns true if the message is a turn-generated result (action, tool, client tool,
@@ -10721,7 +11487,7 @@ The context is now within limits. Please retry your request with the recovered c
10721
11487
  const messageType = msg.metadata?.messageType;
10722
11488
  return messageType === 'action-result'
10723
11489
  || messageType === 'client-tool-result'
10724
- || messageType === 'tool-result'
11490
+ || messageType === BaseAgent.toolResultMessageType
10725
11491
  || messageType === 'sub-agent-result'
10726
11492
  || messageType === 'loop-result';
10727
11493
  }
@@ -10763,47 +11529,42 @@ The context is now within limits. Please retry your request with the recovered c
10763
11529
  getModelContextLimit(modelSelectionInfo) {
10764
11530
  // Default conservative limit if we can't determine the actual limit
10765
11531
  const DEFAULT_LIMIT = 8000;
10766
- if (!modelSelectionInfo) {
10767
- this.logStatus(`No model selection info available, using default limit: ${DEFAULT_LIMIT}`, true);
10768
- return DEFAULT_LIMIT;
11532
+ const known = modelSelectionInfo ? this.tryGetModelMaxInputTokens(modelSelectionInfo) : null;
11533
+ if (known === null) {
11534
+ this.logStatus(`Could not determine model context limit, using default limit: ${DEFAULT_LIMIT}`, true);
10769
11535
  }
11536
+ return known || DEFAULT_LIMIT;
11537
+ }
11538
+ /**
11539
+ * Extracts the vendor-specific MaxInputTokens from model selection info, returning
11540
+ * null when it genuinely cannot be determined. Callers that need a hard number use
11541
+ * {@link getModelContextLimit} (which falls back to a conservative default); callers
11542
+ * for whom a guessed default would be WRONG — e.g. cross-turn compaction budget
11543
+ * clamping, where a bogus 8000 would clamp a configured 200k budget — use this and
11544
+ * handle null explicitly.
11545
+ * @protected
11546
+ */
11547
+ tryGetModelMaxInputTokens(modelSelectionInfo) {
10770
11548
  try {
10771
- // Get the selected model and vendor from the model selection info
10772
11549
  const modelSelected = modelSelectionInfo.modelSelected;
10773
11550
  const vendorSelected = modelSelectionInfo.vendorSelected;
10774
- if (!modelSelected) {
10775
- this.logStatus(`No model selected in model selection info, using default limit: ${DEFAULT_LIMIT}`, true);
10776
- return DEFAULT_LIMIT;
10777
- }
10778
- // If no vendor selected, can't determine model-specific limit
10779
- if (!vendorSelected) {
10780
- this.logStatus(`No vendor selected, using default limit: ${DEFAULT_LIMIT}`, true);
10781
- return DEFAULT_LIMIT;
11551
+ if (!modelSelected || !vendorSelected) {
11552
+ return null;
10782
11553
  }
10783
- // Find the ModelVendor entry that matches the selected vendor
10784
11554
  const modelVendors = modelSelected.ModelVendors;
10785
11555
  if (!modelVendors || modelVendors.length === 0) {
10786
- this.logStatus(`No ModelVendors array found on model, using default limit: ${DEFAULT_LIMIT}`, true);
10787
- return DEFAULT_LIMIT;
11556
+ return null;
10788
11557
  }
10789
- // Find the vendor-specific entry
10790
11558
  const vendorEntry = modelVendors.find((mv) => UUIDsEqual(mv.VendorID, vendorSelected.ID));
10791
- if (!vendorEntry) {
10792
- this.logStatus(`No matching vendor entry found in ModelVendors, using default limit: ${DEFAULT_LIMIT}`, true);
10793
- return DEFAULT_LIMIT;
10794
- }
10795
- // Get MaxInputTokens from the vendor-specific entry
10796
- const maxInputTokens = vendorEntry.MaxInputTokens;
10797
- if (!maxInputTokens || maxInputTokens <= 0) {
10798
- this.logStatus(`MaxInputTokens not set or invalid on vendor entry, using default limit: ${DEFAULT_LIMIT}`, true);
10799
- return DEFAULT_LIMIT;
11559
+ if (!vendorEntry || !vendorEntry.MaxInputTokens || vendorEntry.MaxInputTokens <= 0) {
11560
+ return null;
10800
11561
  }
10801
- this.logStatus(`Using vendor-specific MaxInputTokens: ${maxInputTokens} (Model: ${modelSelected.Name}, Vendor: ${vendorSelected.Name})`, true);
10802
- return maxInputTokens;
11562
+ this.logStatus(`Using vendor-specific MaxInputTokens: ${vendorEntry.MaxInputTokens} (Model: ${modelSelected.Name}, Vendor: ${vendorSelected.Name})`, true);
11563
+ return vendorEntry.MaxInputTokens;
10803
11564
  }
10804
11565
  catch (error) {
10805
- this.logStatus(`Error extracting model context limit: ${error}, using default limit: ${DEFAULT_LIMIT}`, true);
10806
- return DEFAULT_LIMIT;
11566
+ this.logStatus(`Error extracting model context limit: ${error}`, true);
11567
+ return null;
10807
11568
  }
10808
11569
  }
10809
11570
  /**
@@ -10832,6 +11593,10 @@ The context is now within limits. Please retry your request with the recovered c
10832
11593
  * @param request - The expand message request
10833
11594
  * @param params - Agent execution parameters
10834
11595
  * @param currentTurn - Current turn number
11596
+ * @returns null when the expansion succeeded; otherwise a model-facing reason the
11597
+ * expansion is impossible. Callers must surface a non-null reason into the next
11598
+ * prompt's context — a silent no-op leaves the loop state identical and the model
11599
+ * re-requests the same expansion indefinitely.
10835
11600
  * @protected
10836
11601
  */
10837
11602
  executeExpandMessageStep(request, params, currentTurn) {
@@ -10839,12 +11604,14 @@ The context is now within limits. Please retry your request with the recovered c
10839
11604
  const reason = request.expandReason;
10840
11605
  if (messageIndex === undefined || messageIndex < 0 || messageIndex >= params.conversationMessages.length) {
10841
11606
  console.warn(`Cannot expand message: index ${messageIndex} out of bounds`);
10842
- return;
11607
+ return `message index ${messageIndex} is out of bounds — do not request this expansion again.`;
10843
11608
  }
10844
11609
  const message = params.conversationMessages[messageIndex];
10845
11610
  if (!message.metadata?.canExpand || !message.metadata?.originalContent) {
10846
11611
  console.warn(`Cannot expand message at index ${messageIndex}: not expandable or no original content`);
10847
- return;
11612
+ return message.metadata?.isConversationSummary
11613
+ ? `message ${messageIndex} is the cross-turn conversation summary and has no expanded form. To read the underlying history, use the conversation history tools (${ConversationToolNames.join(', ')}) instead — do not request expansion of this message again.`
11614
+ : `message ${messageIndex} is not expandable (it carries no compacted original content) — do not request this expansion again.`;
10848
11615
  }
10849
11616
  // Restore original content
10850
11617
  message.content = message.metadata.originalContent;
@@ -10862,6 +11629,7 @@ The context is now within limits. Please retry your request with the recovered c
10862
11629
  if (params.verbose) {
10863
11630
  console.log(`[Turn ${currentTurn}] Expanded message at index ${messageIndex}`);
10864
11631
  }
11632
+ return null;
10865
11633
  }
10866
11634
  /**
10867
11635
  * Generic template resolver for loop iterations - extracts from LoopAgentType to make available to all agent types