@memberjunction/ai-agents 5.48.0 → 5.49.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/README.md +14 -0
  2. package/dist/AgentRunner.d.ts.map +1 -1
  3. package/dist/AgentRunner.js +5 -0
  4. package/dist/AgentRunner.js.map +1 -1
  5. package/dist/ConversationCompactionManager.d.ts +198 -0
  6. package/dist/ConversationCompactionManager.d.ts.map +1 -0
  7. package/dist/ConversationCompactionManager.js +384 -0
  8. package/dist/ConversationCompactionManager.js.map +1 -0
  9. package/dist/ConversationToolManager.d.ts +154 -0
  10. package/dist/ConversationToolManager.d.ts.map +1 -0
  11. package/dist/ConversationToolManager.js +336 -0
  12. package/dist/ConversationToolManager.js.map +1 -0
  13. package/dist/agent-types/loop-agent-prompt-params.d.ts +8 -0
  14. package/dist/agent-types/loop-agent-prompt-params.d.ts.map +1 -1
  15. package/dist/agent-types/loop-agent-prompt-params.js +1 -0
  16. package/dist/agent-types/loop-agent-prompt-params.js.map +1 -1
  17. package/dist/agent-types/loop-agent-response-type.d.ts +10 -1
  18. package/dist/agent-types/loop-agent-response-type.d.ts.map +1 -1
  19. package/dist/agent-types/loop-agent-response-type.js.map +1 -1
  20. package/dist/agent-types/loop-agent-type.d.ts +23 -34
  21. package/dist/agent-types/loop-agent-type.d.ts.map +1 -1
  22. package/dist/agent-types/loop-agent-type.js +75 -2
  23. package/dist/agent-types/loop-agent-type.js.map +1 -1
  24. package/dist/base-agent.d.ts +314 -4
  25. package/dist/base-agent.d.ts.map +1 -1
  26. package/dist/base-agent.js +869 -92
  27. package/dist/base-agent.js.map +1 -1
  28. package/dist/conversation-history-format.d.ts +19 -0
  29. package/dist/conversation-history-format.d.ts.map +1 -0
  30. package/dist/conversation-history-format.js +40 -0
  31. package/dist/conversation-history-format.js.map +1 -0
  32. package/dist/index.d.ts +4 -0
  33. package/dist/index.d.ts.map +1 -1
  34. package/dist/index.js +4 -0
  35. package/dist/index.js.map +1 -1
  36. package/dist/prior-turn-tool-result-cache.d.ts +71 -0
  37. package/dist/prior-turn-tool-result-cache.d.ts.map +1 -0
  38. package/dist/prior-turn-tool-result-cache.js +85 -0
  39. package/dist/prior-turn-tool-result-cache.js.map +1 -0
  40. package/dist/realtime/realtime-client-session-service.d.ts.map +1 -1
  41. package/dist/realtime/realtime-client-session-service.js +7 -3
  42. package/dist/realtime/realtime-client-session-service.js.map +1 -1
  43. package/dist/realtime/realtime-coagent-config.d.ts +41 -0
  44. package/dist/realtime/realtime-coagent-config.d.ts.map +1 -1
  45. package/dist/realtime/realtime-coagent-config.js +69 -0
  46. package/dist/realtime/realtime-coagent-config.js.map +1 -1
  47. package/dist/realtime/realtime-session-runner.d.ts +72 -1
  48. package/dist/realtime/realtime-session-runner.d.ts.map +1 -1
  49. package/dist/realtime/realtime-session-runner.js +215 -13
  50. package/dist/realtime/realtime-session-runner.js.map +1 -1
  51. package/dist/realtime/realtime-tool-broker.d.ts +7 -2
  52. package/dist/realtime/realtime-tool-broker.d.ts.map +1 -1
  53. package/dist/realtime/realtime-tool-broker.js +13 -10
  54. package/dist/realtime/realtime-tool-broker.js.map +1 -1
  55. package/dist/tool-result-format.d.ts +78 -0
  56. package/dist/tool-result-format.d.ts.map +1 -0
  57. package/dist/tool-result-format.js +46 -0
  58. package/dist/tool-result-format.js.map +1 -0
  59. package/package.json +18 -18
@@ -23,7 +23,7 @@ import { CrushJSON, DescribeCrush, PartitionStablePrefix } from '@memberjunction
23
23
  import { CrushCode } from '@memberjunction/context-crush/code';
24
24
  import { RealtimeSessionRunner } from './realtime/realtime-session-runner.js';
25
25
  import { ResolveNarrationInstructionsTemplate } from './realtime/realtime-narration.js';
26
- import { BuildRealtimeOverridesJson, BuildVoiceMannerSection, GetNarrationPaceMs, GetProviderVoiceSettings, ResolveEffectiveRealtimeConfig } from './realtime/realtime-coagent-config.js';
26
+ import { BuildRealtimeOverridesJson, BuildVoiceMannerSection, GetNarrationPaceMs, GetProviderVoiceSettings, GetSessionTuningSettings, DeepMergeConfigs, ResolveEffectiveRealtimeConfig } from './realtime/realtime-coagent-config.js';
27
27
  import { RealtimeClientSessionService } from './realtime/realtime-client-session-service.js';
28
28
  import { BuildRealtimeAgentFraming } from './realtime/realtime-tool-broker.js';
29
29
  import { RealtimeRecordingController } from './realtime/realtime-recording-capture.js';
@@ -32,9 +32,13 @@ import { AIEngine } from '@memberjunction/aiengine';
32
32
  import { ActionEngineServer } from '@memberjunction/actions';
33
33
  import { AIAgentPermissionHelper } from '@memberjunction/ai-engine-base';
34
34
  import { AgentMemoryContextBuilder } from './agent-memory-context-builder.js';
35
+ import { ConversationCompactionManager } from './ConversationCompactionManager.js';
36
+ import { ConversationToolManager, ConversationToolNames, MAX_CONVERSATION_TOOL_CALLS_PER_TURN } from './ConversationToolManager.js';
37
+ import { FormatToolResultSection, FormatToolErrorSection, RenderToolResultData, CarryForwardToolFamily } from './tool-result-format.js';
38
+ import { PriorTurnToolResultCache } from './prior-turn-tool-result-cache.js';
35
39
  import { PromptComponentResolver, InjectScopedPromptParts } from './prompt-component-resolver.js';
36
40
  import { ScopedPromptConfigResolver, ApplyScopedPromptConfig } from './scoped-prompt-config-resolver.js';
37
- import { AIPromptParams, ChildPromptParam, ConversationUtility, ParseFileOutputRef, parseAssignmentStrategy, ResolveClientTools, initAgentRunStep, finalizeAgentRunStep, AgentRunStepSaveQueue } from '@memberjunction/ai-core-plus';
41
+ import { AIPromptParams, ChildPromptParam, ConversationUtility, ParseFileOutputRef, parseAssignmentStrategy, ResolveClientTools, initAgentRunStep, finalizeAgentRunStep, AgentRunStepSaveQueue, ExtractPromptResultText } from '@memberjunction/ai-core-plus';
38
42
  import { AgentRunner } from './AgentRunner.js';
39
43
  import { PayloadManager } from './PayloadManager.js';
40
44
  import { ScratchpadManager } from './ScratchpadManager.js';
@@ -215,6 +219,11 @@ export class BaseAgent {
215
219
  * Allows agents to explore input artifacts on demand.
216
220
  */
217
221
  this._artifactToolManager = new ArtifactToolManager();
222
+ /**
223
+ * Manages conversation-history retrieval tools for the current agent run.
224
+ * Armed only when the run has a conversationId (the cross-turn context gate).
225
+ */
226
+ this._conversationToolManager = new ConversationToolManager();
218
227
  /**
219
228
  * Manages in-flight durable memory writes for the current agent run.
220
229
  * Only consulted when the agent has AllowMemoryWrite enabled.
@@ -332,11 +341,30 @@ export class BaseAgent {
332
341
  */
333
342
  // ── Realtime per-session capture state (scoped to one executeRealtimeSession run) ──────────
334
343
  /**
335
- * In-flight realtime turn rows keyed by transcript role (`'user'`/`'assistant'`), driving the
336
- * create-on-start / update-on-complete persistence lifecycle. Reset at the start of every
337
- * realtime session so a prior run can never leak an in-flight id into the next.
344
+ * The current realtime turn row per transcript role (`'user'`/`'assistant'`), driving the
345
+ * create-on-start / update-on-complete persistence lifecycle. `open` is true while the row is an
346
+ * unfinalized In-Progress interim (so subsequent interim deltas fold into it and a following final
347
+ * finalizes it in place); it flips false once finalized, but the entry is KEPT so a streamed
348
+ * `ReplacesPrevious` re-final can still update the same row. A new turn is detected when the next
349
+ * interim (or non-replacing final) arrives with the current entry already closed. Reset at the
350
+ * start of every realtime session so a prior run can never leak a row id into the next.
338
351
  */
339
352
  this.realtimeInFlightTurns = new Map();
353
+ /**
354
+ * Per-role serialization queue for transcript persistence.
355
+ *
356
+ * The runner dispatches provider transcript frames FIRE-AND-FORGET (`void this.handleTranscript(t)`),
357
+ * so frames for the same role can be in flight CONCURRENTLY. {@link persistRealtimeTranscript} does a
358
+ * check-then-act on {@link realtimeInFlightTurns} that spans `await`s (GetEntityObject / Load / Save):
359
+ * without serialization, two captions arriving a few ms apart both observe "no tracked row yet", both
360
+ * take the create branch, and the turn is persisted TWICE. Observed in production against a streamed
361
+ * Grok session (two byte-identical rows, the second created 17ms before the first's final update).
362
+ *
363
+ * Each role's calls are therefore chained through this map so the read-modify-write is atomic with
364
+ * respect to other frames of the SAME role. Roles are independent (separate `realtimeInFlightTurns`
365
+ * entries), so they are not serialized against each other. Reset per session alongside the turn map.
366
+ */
367
+ this.realtimePersistQueues = new Map();
340
368
  /** Active audio recording controller for the current realtime session, or `null` when recording is off. */
341
369
  this.realtimeRecording = null;
342
370
  /** Storage account id the active recording stores to (RecordingStorageProviderID ?? AttachmentStorageProviderID). */
@@ -1089,6 +1117,11 @@ export class BaseAgent {
1089
1117
  ...params,
1090
1118
  onProgress: this.wrapProgressCallback(params.onProgress)
1091
1119
  };
1120
+ // Capture for lifecycle hooks that don't receive params (post-turn compaction
1121
+ // inside finalizeAgentRun reads conversationId / verbose / provider from here).
1122
+ this._executeParams = wrappedParams;
1123
+ this._agentConfig = undefined;
1124
+ this._lastModelSelectionInfo = undefined;
1092
1125
  // Convert UI markup in conversation messages to plain text if requested (default: true)
1093
1126
  if (params.convertUIMarkupToPlainText !== false) {
1094
1127
  this.convertUIMarkupInMessages(wrappedParams.conversationMessages);
@@ -1097,6 +1130,10 @@ export class BaseAgent {
1097
1130
  this._scratchpadManager.Clear();
1098
1131
  this._artifactToolManager.Clear();
1099
1132
  this._memoryWriteManager.Clear();
1133
+ // Arm conversation-history retrieval tools — available only when the run has a
1134
+ // conversation to page against (the same gate as all cross-turn context features).
1135
+ this._conversationToolManager.Initialize(wrappedParams.conversationId || null, params.contextUser);
1136
+ this._conversationToolManager.SetSummaryHost(this.buildConversationSummaryHost(wrappedParams));
1100
1137
  // Initialize artifact tools with any input artifacts attached to the run.
1101
1138
  // Artifacts arrive as a typed first-class field on ExecuteAgentParams —
1102
1139
  // they are NOT routed through `data` because prompt-template rendering
@@ -1224,7 +1261,11 @@ export class BaseAgent {
1224
1261
  this.loadAgentConfiguration(params.agent),
1225
1262
  this.preloadAgentData(wrappedParams),
1226
1263
  this.InjectContextMemory(typeof inputText === 'string' ? inputText : '', params.agent, userId, companyId, params.contextUser, wrappedParams.conversationMessages, primaryScopeEntityId, primaryScopeRecordId, secondaryScopes, scopeConfig),
1227
- this.InjectPreExecutionRAG(typeof inputText === 'string' ? inputText : '', params.agent, params.contextUser, wrappedParams.conversationMessages, params.conversationMessages, primaryScopeEntityId, primaryScopeRecordId, secondaryScopes, params.payload)
1264
+ this.InjectPreExecutionRAG(typeof inputText === 'string' ? inputText : '', params.agent, params.contextUser, wrappedParams.conversationMessages, params.conversationMessages, primaryScopeEntityId, primaryScopeRecordId, secondaryScopes, params.payload),
1265
+ // Carry the previous turn's tool results forward (no-op without a
1266
+ // conversationId). Runs here so the results are in the messages before
1267
+ // the pre-turn compaction check and the first prompt.
1268
+ this.injectPriorTurnToolResults(wrappedParams)
1228
1269
  ]);
1229
1270
  // Inject scope-resolved prompt parts (role-faithful) for this agent's prompt, alongside
1230
1271
  // memory/RAG. Synchronous — parts are cached on AIEngine. Uses the same run scope.
@@ -1239,6 +1280,7 @@ export class BaseAgent {
1239
1280
  // --- PHASE 3: Agent type initialization (sequential) ---
1240
1281
  // Must wait for config from Phase 2 because it needs the resolved agent type and
1241
1282
  // prompt configuration to initialize the type-specific state machine.
1283
+ this._agentConfig = config;
1242
1284
  await this.initializeAgentType(wrappedParams, config);
1243
1285
  // =====================================================================================
1244
1286
  // SESSION-DRIVEN BRANCH (Realtime agent type)
@@ -1255,6 +1297,10 @@ export class BaseAgent {
1255
1297
  this.logStatus(`🎙️ Agent '${params.agent.Name}' is session-driven — routing to RealtimeSessionRunner`, true, params);
1256
1298
  return await this.executeRealtimeSession(wrappedParams, config);
1257
1299
  }
1300
+ // Cross-turn compaction PRE-TURN fallback: only when the assembled window is
1301
+ // ALREADY over the trigger budget before the first prompt (the normal path is
1302
+ // the post-turn fire-and-forget in finalizeAgentRun, which hides the latency).
1303
+ await this.checkPreTurnCompaction(wrappedParams, config);
1258
1304
  // Execute the agent's internal logic with wrapped parameters
1259
1305
  this.logStatus(`🚀 Executing agent '${params.agent.Name}' internal logic`, true, params);
1260
1306
  const executionResult = await this.executeAgentInternal(wrappedParams, config);
@@ -1362,6 +1408,7 @@ export class BaseAgent {
1362
1408
  // 3) Resolve recording (OFF by default; runtime > agent > off; consent + storage gated) and reset
1363
1409
  // the per-session turn-lifecycle state, then build the injected deps and run the session.
1364
1410
  this.realtimeInFlightTurns = new Map();
1411
+ this.realtimePersistQueues = new Map();
1365
1412
  const recording = await this.resolveRealtimeRecording(params);
1366
1413
  this.realtimeRecording = recording?.controller ?? null;
1367
1414
  this.realtimeRecordingAccountId = recording?.storageAccountId ?? null;
@@ -1660,9 +1707,14 @@ export class BaseAgent {
1660
1707
  DelegateToTarget: (request) => this.delegateRealtimeToTarget(params, config, request),
1661
1708
  ExecuteTool: (call) => this.executeRealtimeTool(params, call),
1662
1709
  PersistTranscript: (transcript) => this.persistRealtimeTranscript(params, transcript),
1710
+ FlushTranscripts: () => this.flushRealtimeTranscriptQueues(),
1663
1711
  Recording: this.realtimeRecording ?? undefined,
1664
1712
  FinalizeRecording: () => this.finalizeRealtimeRecording(params),
1665
1713
  CheckpointUsage: (usage) => this.checkpointRealtimeUsage(promptRun, usage),
1714
+ // The chained agent cancellation signal (caller token + agent timeout) — the runner
1715
+ // observes it so a realtime session honors the same wall-clock/cancel semantics as
1716
+ // every other agent run instead of living until the janitor sweeps it.
1717
+ AbortSignal: params.cancellationToken,
1666
1718
  // DB-driven spoken-progress wording (shared lookup with the client-direct path);
1667
1719
  // null → the runner's documented built-in first-person fallback.
1668
1720
  NarrationInstructionsTemplate: ResolveNarrationInstructionsTemplate(),
@@ -1699,15 +1751,20 @@ export class BaseAgent {
1699
1751
  const systemPrompt = [framing, basePrompt, voiceManner, memoryContext]
1700
1752
  .filter(part => part && part.trim().length > 0)
1701
1753
  .join('\n\n');
1702
- // Provider-matched voice settings (realtime.voice.providers.<provider>) flow into the
1703
- // driver's open Config bag — the same pact every other config entry rides.
1754
+ // Provider-matched voice settings (realtime.voice.providers.<provider>) AND session-tuning
1755
+ // knobs (realtime.session) flow into the driver's open Config bag — the same pact every
1756
+ // other config entry rides, mirroring the client-direct builder's cascade exactly.
1704
1757
  const providerVoice = GetProviderVoiceSettings(effectiveConfig, driverClass ?? null);
1758
+ const sessionTuning = GetSessionTuningSettings(effectiveConfig);
1759
+ const configBag = (sessionTuning || providerVoice)
1760
+ ? DeepMergeConfigs(sessionTuning, providerVoice)
1761
+ : undefined;
1705
1762
  return {
1706
1763
  Model: modelApiName,
1707
1764
  SystemPrompt: systemPrompt,
1708
1765
  InitialContext: memoryContext || undefined,
1709
- // JSONObjectLike -> JSONObject: safe — the settings object came from JSON.parse.
1710
- Config: providerVoice ? providerVoice : undefined
1766
+ // JSONObjectLike -> JSONObject: safe — the settings objects came from JSON.parse.
1767
+ Config: configBag
1711
1768
  };
1712
1769
  }
1713
1770
  /**
@@ -1922,7 +1979,45 @@ export class BaseAgent {
1922
1979
  * @param transcript The transcript turn (interim delta or final) emitted by the model.
1923
1980
  * @returns The created row id on first creation of a turn, else `null`.
1924
1981
  */
1925
- async persistRealtimeTranscript(params, transcript) {
1982
+ persistRealtimeTranscript(params, transcript) {
1983
+ // Serialize per role — see realtimePersistQueues. Transcript frames arrive fire-and-forget, so
1984
+ // without this chain two concurrent captions can both pass the "is there a tracked row?" check
1985
+ // before either has written one back, and the turn is persisted twice.
1986
+ const roleKey = transcript.Role;
1987
+ const run = () => this.persistRealtimeTranscriptSerialized(params, transcript);
1988
+ const prior = this.realtimePersistQueues.get(roleKey) ?? Promise.resolve();
1989
+ // `.then(run, run)` (not `.then(run)`) so a rejected predecessor never strands the rest of the
1990
+ // queue — each frame runs regardless of how the previous one settled.
1991
+ const result = prior.then(run, run);
1992
+ // The stored link swallows outcomes: the queue only needs ordering, and an unhandled rejection
1993
+ // parked in the map would surface as an unhandled promise rejection.
1994
+ this.realtimePersistQueues.set(roleKey, result.then(() => undefined, () => undefined));
1995
+ return result;
1996
+ }
1997
+ /**
1998
+ * Waits for every role's queued transcript writes to settle.
1999
+ *
2000
+ * Transcript frames are dispatched fire-and-forget, so writes for the last turns of a session can
2001
+ * still be in flight at teardown. The session runner calls this during `Stop()` — after the provider
2002
+ * session is closed, so no new frames can arrive — under its own hard timeout, which is why this
2003
+ * method itself is unbounded and simply awaits what is queued.
2004
+ *
2005
+ * Awaits the STORED queue links, which are outcome-swallowing by construction, so a failed write
2006
+ * can never reject here and abort the drain for other roles.
2007
+ */
2008
+ async flushRealtimeTranscriptQueues() {
2009
+ const pending = [...this.realtimePersistQueues.values()];
2010
+ if (pending.length === 0) {
2011
+ return;
2012
+ }
2013
+ await Promise.all(pending);
2014
+ }
2015
+ /**
2016
+ * The actual persistence work for one transcript frame. Runs under the per-role queue established by
2017
+ * {@link persistRealtimeTranscript}, so it may safely read-modify-write {@link realtimeInFlightTurns}
2018
+ * across its `await`s without another frame of the same role interleaving.
2019
+ */
2020
+ async persistRealtimeTranscriptSerialized(params, transcript) {
1926
2021
  if (!transcript.Text?.trim()) {
1927
2022
  return null;
1928
2023
  }
@@ -1935,9 +2030,12 @@ export class BaseAgent {
1935
2030
  const mjRole = transcript.Role === 'user' ? 'User' : 'AI';
1936
2031
  // ── INTERIM: create the In-Progress row once per turn (first delta) ───────────────────────
1937
2032
  if (!transcript.IsFinal) {
1938
- if (this.realtimeInFlightTurns.has(roleKey)) {
1939
- return null; // already created for this turn; ignore subsequent deltas
2033
+ if (this.realtimeInFlightTurns.get(roleKey)?.open) {
2034
+ return null; // an In-Progress row for THIS turn already exists; fold this delta into it
1940
2035
  }
2036
+ // A closed entry (a prior turn's finalized row still tracked for streamed re-finals) means
2037
+ // THIS delta begins a NEW turn — fall through and create a fresh In-Progress row, replacing
2038
+ // the tracked entry below.
1941
2039
  const detail = await md.GetEntityObject('MJ: Conversation Details', params.contextUser);
1942
2040
  detail.NewRecord();
1943
2041
  detail.ConversationID = conversationID;
@@ -1954,18 +2052,35 @@ export class BaseAgent {
1954
2052
  });
1955
2053
  return null;
1956
2054
  }
1957
- this.realtimeInFlightTurns.set(roleKey, detail.ID);
2055
+ this.realtimeInFlightTurns.set(roleKey, { id: detail.ID, open: true });
1958
2056
  return detail.ID;
1959
2057
  }
1960
- // ── FINAL: update the in-flight row (or create+finalize when no interim was seen) ─────────
1961
- const inFlightId = this.realtimeInFlightTurns.get(roleKey);
1962
- this.realtimeInFlightTurns.delete(roleKey);
1963
- let detail = await md.GetEntityObject('MJ: Conversation Details', params.contextUser);
1964
- let created = false;
1965
- if (inFlightId && await detail.Load(inFlightId)) {
1966
- // updating the existing streaming row → not a new turn
2058
+ // ── FINAL: update the in-flight row, or create+finalize a fresh turn ──────────────────────
2059
+ // Every real provider shape must yield exactly ONE row per turn:
2060
+ // 1. interim-based (OpenAI): delta(s) open the In-Progress row → final finalizes it;
2061
+ // 2. streamed re-finals (Grok user captions): the SAME turn emits repeated finals, each the
2062
+ // full growing text — the 2nd+ carry ReplacesPrevious=true (stamped by the driver) and
2063
+ // REPLACE the turn's row, not append;
2064
+ // 3. finals-only single (+ ElevenLabs corrections): one non-replacing final, optionally
2065
+ // followed by a ReplacesPrevious correction;
2066
+ // 4. (robustness) a provider that emits BOTH interim deltas AND repeated completeds.
2067
+ //
2068
+ // Reuse the tracked row iff this final REPLACES the turn (ReplacesPrevious) OR the tracked row
2069
+ // is still an OPEN interim (this final finalizes it). A non-replacing final whose tracked entry
2070
+ // is already CLOSED (a prior turn's finalized row) starts a NEW turn. The entry is then KEPT
2071
+ // (closed) rather than deleted, so a later streamed re-final can still update this same row and
2072
+ // the next interim/non-replacing-final correctly detects the turn boundary via `open`.
2073
+ const inFlight = this.realtimeInFlightTurns.get(roleKey);
2074
+ let detail = null;
2075
+ if (inFlight && (transcript.ReplacesPrevious || inFlight.open)) {
2076
+ const candidate = await md.GetEntityObject('MJ: Conversation Details', params.contextUser);
2077
+ if (await candidate.Load(inFlight.id)) {
2078
+ detail = candidate; // update the existing row in place → not a new turn
2079
+ }
1967
2080
  }
1968
- else {
2081
+ let created = false;
2082
+ if (!detail) {
2083
+ detail = await md.GetEntityObject('MJ: Conversation Details', params.contextUser);
1969
2084
  detail.NewRecord();
1970
2085
  detail.ConversationID = conversationID;
1971
2086
  detail.Role = mjRole;
@@ -1981,11 +2096,19 @@ export class BaseAgent {
1981
2096
  if (this.realtimeRecording) {
1982
2097
  detail.UtteranceEndMs = this.realtimeRecording.NowOffsetMs();
1983
2098
  }
1984
- if (!await detail.Save()) {
2099
+ const saved = await detail.Save();
2100
+ if (!saved) {
1985
2101
  this.logError(`Failed to finalize realtime transcript turn: ${detail.LatestResult?.CompleteMessage ?? 'unknown error'}`, {
1986
2102
  agent: params.agent, category: 'RealtimeSession'
1987
2103
  });
1988
2104
  }
2105
+ // Track this turn's now-finalized (closed) row so a subsequent ReplacesPrevious re-final updates
2106
+ // it in place, and so the next interim / non-replacing final detects the new-turn boundary via
2107
+ // `open === false`. Only bind a real id — a failed create leaves an empty id that would poison
2108
+ // the next lookup, so leave the prior entry untouched in that case.
2109
+ if (saved && detail.ID) {
2110
+ this.realtimeInFlightTurns.set(roleKey, { id: detail.ID, open: false });
2111
+ }
1989
2112
  return created ? detail.ID : null;
1990
2113
  }
1991
2114
  /**
@@ -2125,6 +2248,18 @@ export class BaseAgent {
2125
2248
  promptRun.TokensPrompt = usage.InputTokens;
2126
2249
  promptRun.TokensCompletion = usage.OutputTokens;
2127
2250
  promptRun.TokensUsed = usage.InputTokens + usage.OutputTokens;
2251
+ // Per-modality detail (audio vs text vs cached) — REQUIRED for correct multi-channel cost
2252
+ // attribution (audio-in bills ~8x text-in on GPT Realtime 2.1). The realtime prompt run's
2253
+ // Result column is otherwise unused (a live session has no single prompt output), so the
2254
+ // detail rides there as JSON for the cost pipeline / dashboards to consume.
2255
+ if (usage.InputTokenDetails || usage.OutputTokenDetails) {
2256
+ promptRun.Result = JSON.stringify({
2257
+ realtimeUsageDetails: {
2258
+ input: usage.InputTokenDetails ?? null,
2259
+ output: usage.OutputTokenDetails ?? null,
2260
+ },
2261
+ });
2262
+ }
2128
2263
  if (!await promptRun.Save()) {
2129
2264
  this.logError(`Failed to checkpoint realtime usage: ${promptRun.LatestResult?.CompleteMessage ?? 'unknown error'}`, {
2130
2265
  category: 'RealtimeSession'
@@ -2779,6 +2914,13 @@ export class BaseAgent {
2779
2914
  else if (this._artifactToolManager.HasArtifacts()) {
2780
2915
  this.logStatus(`[ArtifactTools] Artifacts present but tools disabled by agent config (includeArtifactToolsDocs=false)`, true, params);
2781
2916
  }
2917
+ // Inject conversation-history retrieval tool docs when the run has a
2918
+ // conversation to page against. Like artifact tools, results are pushed as
2919
+ // one-shot conversation messages, never re-rendered per turn.
2920
+ const conversationToolsEnabled = agentTypePromptParams?.includeConversationToolsDocs !== false;
2921
+ if (conversationToolsEnabled && this._conversationToolManager.IsAvailable) {
2922
+ promptParams.data['_CONVERSATION_TOOLS'] = this._conversationToolManager.GetToolDocumentation();
2923
+ }
2782
2924
  // Enable the memory-writes response field + docs only for agents that opted in
2783
2925
  // via AllowMemoryWrite. Disabled agents never see the docs, so a well-behaved
2784
2926
  // LLM never emits the field (the turn loop still guards against drift).
@@ -4420,17 +4562,387 @@ The context is now within limits. Please retry your request with the recovered c
4420
4562
  },
4421
4563
  });
4422
4564
  const stored = await this._artifactToolManager.ExecuteSingleToolCall(call);
4423
- await this.finalizeStepEntity(toolStep, stored.result.success, stored.result.success ? undefined : stored.result.errorMessage, {
4565
+ // Typed cross-turn contract — see CarryForwardToolStepOutput (read back by
4566
+ // BuildPriorTurnToolResultsMessage on the next run; StepName is display-only).
4567
+ const carryForwardOutput = {
4568
+ toolFamily: CarryForwardToolFamily.Artifact,
4424
4569
  artifactId: stored.artifactId,
4425
4570
  tool: stored.tool,
4426
4571
  input: stored.input,
4427
4572
  result: stored.result,
4428
4573
  durationMs: stored.durationMs,
4429
- });
4574
+ };
4575
+ await this.finalizeStepEntity(toolStep, stored.result.success, stored.result.success ? undefined : stored.result.errorMessage, carryForwardOutput);
4430
4576
  return stored;
4431
4577
  }));
4432
4578
  return results;
4433
4579
  }
4580
+ /**
4581
+ * Carries the PREVIOUS turn's tool results forward into this run's context.
4582
+ *
4583
+ * Inline tool results (artifact + conversation tools) are injected into the run's
4584
+ * in-memory messages only — the next turn rebuilds messages from the conversation
4585
+ * window, so a result paged in on turn N is gone on turn N+1 and the agent must
4586
+ * re-call the tool. The results already persist in each Tool step's OutputData;
4587
+ * this re-injects the immediately previous completed run's successful results as
4588
+ * one transient message. One-turn memory by construction: each run carries only
4589
+ * its direct predecessor's results, so context never compounds.
4590
+ *
4591
+ * Gated on conversationId + root depth — programmatic runs and sub-agents skip it.
4592
+ * @protected
4593
+ */
4594
+ async injectPriorTurnToolResults(params) {
4595
+ if (!params.conversationId || this._depth !== 0) {
4596
+ return;
4597
+ }
4598
+ try {
4599
+ const steps = await this.loadPriorTurnToolResultSteps(params);
4600
+ const body = BaseAgent.BuildPriorTurnToolResultsMessage(steps, this.maxStandaloneToolResultChars);
4601
+ if (!body) {
4602
+ return;
4603
+ }
4604
+ const message = {
4605
+ role: 'user',
4606
+ content: body,
4607
+ metadata: {
4608
+ turnAdded: 0,
4609
+ messageType: BaseAgent.toolResultMessageType,
4610
+ expirationTurns: 2,
4611
+ expirationMode: 'Compact',
4612
+ compactMode: 'First N Chars',
4613
+ compactLength: 500,
4614
+ compactPromptId: '',
4615
+ },
4616
+ };
4617
+ params.conversationMessages.push(message);
4618
+ this.logStatus(`[PriorTurnToolResults] Carried ${steps.length} tool result(s) forward from the previous run`, true, params);
4619
+ }
4620
+ catch (error) {
4621
+ // Carry-forward is an optimization — never let it break the run.
4622
+ this.logStatus(`[PriorTurnToolResults] Skipped (contained error): ${error instanceof Error ? error.message : error}`, true, params);
4623
+ }
4624
+ }
4625
+ /**
4626
+ * Loads this agent's previous settled root run's Tool steps for this conversation
4627
+ * (settled = {@link settledRunStatuses}: Completed or AwaitingFeedback).
4628
+ * Deliberately loads ALL completed Tool steps — eligibility for carry-forward is
4629
+ * decided structurally by {@link BuildPriorTurnToolResultsMessage} via the
4630
+ * `toolFamily` field the executors stamp into OutputData, never by StepName
4631
+ * (which is a display label and free to change).
4632
+ *
4633
+ * Consults {@link PriorTurnToolResultCache} first — the completing run populates it
4634
+ * in {@link finalizeAgentRun} from its in-memory steps, so on this node the common
4635
+ * case (including "prior run made no tool calls") costs zero DB queries; the
4636
+ * RunView pair below is the cache-miss fallback (first turn, restart, other node).
4637
+ * @private
4638
+ */
4639
+ async loadPriorTurnToolResultSteps(params) {
4640
+ const cached = PriorTurnToolResultCache.Instance.Get(params.conversationId, params.agent.ID);
4641
+ if (cached) {
4642
+ this.logStatus(`[PriorTurnToolResults] Prior-run tool results served from cache (${cached.length} step(s), no DB lookup)`, true, params);
4643
+ return cached;
4644
+ }
4645
+ const predicate = BaseAgent.carryForwardPredicate;
4646
+ const statusList = predicate.runStatuses.map(s => `'${s}'`).join(', ');
4647
+ const rv = RunView.FromMetadataProvider(this.ProviderToUse);
4648
+ // AgentID scopes provenance: in a multi-agent conversation, agent B must never
4649
+ // inherit agent A's results labeled "your previous turn".
4650
+ const priorRun = await rv.RunView({
4651
+ EntityName: 'MJ: AI Agent Runs',
4652
+ ExtraFilter: `ConversationID='${params.conversationId}' AND Status IN (${statusList}) AND ParentRunID IS NULL AND AgentID='${params.agent.ID}'`,
4653
+ OrderBy: '__mj_CreatedAt DESC',
4654
+ MaxRows: 1,
4655
+ Fields: ['ID'],
4656
+ ResultType: 'simple',
4657
+ }, params.contextUser);
4658
+ const priorRunId = priorRun.Success ? priorRun.Results?.[0]?.ID : undefined;
4659
+ if (!priorRunId) {
4660
+ return [];
4661
+ }
4662
+ const steps = await rv.RunView({
4663
+ EntityName: 'MJ: AI Agent Run Steps',
4664
+ ExtraFilter: `AgentRunID='${priorRunId}' AND StepType='${predicate.stepType}' AND Status='${predicate.stepStatus}'`,
4665
+ OrderBy: 'StartedAt ASC',
4666
+ Fields: ['OutputData'],
4667
+ ResultType: 'simple',
4668
+ }, params.contextUser);
4669
+ return steps.Success ? (steps.Results || []) : [];
4670
+ }
4671
+ /**
4672
+ * Publishes this run's completed Tool-step results to {@link PriorTurnToolResultCache}
4673
+ * so the conversation's next turn skips the prior-run DB lookups. Applies the same
4674
+ * row predicate as the DB path ({@link carryForwardPredicate}): root runs only, and
4675
+ * only when the run row settled as {@link settledRunStatuses} (Completed OR
4676
+ * AwaitingFeedback — the normal chat-turn ending) — a failed run leaves the previous
4677
+ * settled run's entry standing, just as the RunView filter would. Scoped to this
4678
+ * run's agent (cache key = conversation + agent) so parallel agents in one
4679
+ * conversation never cross-pollinate. An empty projection is cached too (the
4680
+ * negative-cache case that spares tool-free conversations the queries every turn).
4681
+ * Same-node edge semantics (failed step INSERTs, concurrent completions) are
4682
+ * documented on the cache class. Called from {@link finalizeAgentRun}.
4683
+ * @private
4684
+ */
4685
+ cachePriorTurnToolResults() {
4686
+ const predicate = BaseAgent.carryForwardPredicate;
4687
+ const conversationId = this._executeParams?.conversationId;
4688
+ if (!conversationId || this._depth !== 0 || !this._agentRun
4689
+ || !predicate.runStatuses.includes(this._agentRun.Status)) {
4690
+ return;
4691
+ }
4692
+ const records = (this._agentRun.Steps || [])
4693
+ .filter(s => s.StepType === predicate.stepType && s.Status === predicate.stepStatus)
4694
+ .map(s => ({ OutputData: s.OutputData || null }));
4695
+ PriorTurnToolResultCache.Instance.Set(conversationId, this._agentRun.AgentID, records);
4696
+ }
4697
+ /**
4698
+ * Tool families whose step results are eligible for prior-turn carry-forward —
4699
+ * derived from {@link CarryForwardToolFamily} (the single source the stamp sites use).
4700
+ * Read-tool families only: memory writes, pipelines, and client tools also record
4701
+ * `StepType='Tool'` steps but must never be replayed as reusable results.
4702
+ */
4703
+ static { this.CarryForwardToolFamilies = Object.values(CarryForwardToolFamily); }
4704
+ /**
4705
+ * Run statuses that count as a successfully settled root turn. 'AwaitingFeedback' is
4706
+ * included because a Chat final step is the NORMAL per-turn completion for
4707
+ * conversational agents — {@link finalizeAgentRun} maps `step === 'Chat'` to
4708
+ * `Status='AwaitingFeedback'` with `Success=true`, so gating on 'Completed' alone
4709
+ * silently disables post-turn compaction and carry-forward for the most common
4710
+ * agent shape (a chat agent in a long conversation).
4711
+ *
4712
+ * Deliberately `ReadonlyArray<Union>` rather than an `as const` tuple: a narrowed
4713
+ * tuple type would make `.includes(status)` fail to typecheck against the wider
4714
+ * entity union, while this form keeps the compile-time check that each literal is a
4715
+ * valid status (a CHECK-constraint change still surfaces here). Do not "tighten" it.
4716
+ *
4717
+ * Single source for the consumers that must agree: the carry-forward predicate
4718
+ * ({@link carryForwardPredicate} → DB filter + cache-population gate) and the
4719
+ * post-turn compaction gate ({@link startPostTurnCompaction}).
4720
+ */
4721
+ static { this.settledRunStatuses = ['Completed', 'AwaitingFeedback']; }
4722
+ /**
4723
+ * The carry-forward row predicate — the SINGLE source shared by the two places that
4724
+ * must select the same rows or the cache diverges from the DB path: the RunView
4725
+ * `ExtraFilter`s in {@link loadPriorTurnToolResultSteps} (DB fallback) and the
4726
+ * in-memory gate/projection in {@link cachePriorTurnToolResults} (cache population).
4727
+ * Values are typed from the entity unions so a CHECK-constraint change surfaces here
4728
+ * at compile time instead of silently desynchronizing the two loaders.
4729
+ *
4730
+ * The executing agent's ID also scopes both paths (SQL `AgentID=` clause + cache
4731
+ * key) but is per-run data, not a literal contract — it lives at the call sites,
4732
+ * not here.
4733
+ */
4734
+ static { this.carryForwardPredicate = {
4735
+ stepType: 'Tool',
4736
+ stepStatus: 'Completed',
4737
+ runStatuses: BaseAgent.settledRunStatuses,
4738
+ }; }
4739
+ /**
4740
+ * Display name of the seeded system prompt behind summarizeRange's recursive
4741
+ * sub-call (see metadata/prompts/.summarize-range-prompt.json). Resolved with a
4742
+ * trimmed, case-insensitive compare — never an exact-case inline literal.
4743
+ */
4744
+ static { this.SummarizeRangePromptName = 'Summarize Conversation Range'; }
4745
+ /**
4746
+ * The `messageType` marker stamped on injected tool-result messages and matched by
4747
+ * the compaction/pruning eligibility checks — single-sourced so writers and matchers
4748
+ * cannot drift. (Value participates in the AgentChatMessageMetadata union.)
4749
+ */
4750
+ static { this.toolResultMessageType = 'tool-result'; }
4751
+ /**
4752
+ * Header stems for injected tool-result messages. These exact headers are a contract:
4753
+ * the loop-agent system template (loop-agent-type-system-prompt.template.md, "header
4754
+ * `Conversation history tool result:`" / "`Artifact tool result:`") teaches the model
4755
+ * to recognize them — change the template in lockstep.
4756
+ */
4757
+ static conversationToolResultsHeader(count) {
4758
+ return count === 1 ? 'Conversation history tool result:' : `Conversation history tool results (${count} calls):`;
4759
+ }
4760
+ /** Artifact analog of {@link conversationToolResultsHeader} — same template contract. */
4761
+ static artifactToolResultsHeader(count) {
4762
+ return count === 1 ? 'Artifact tool result:' : `Artifact tool results (${count} calls):`;
4763
+ }
4764
+ /**
4765
+ * Renders prior-turn tool-result steps into the carried-forward message body.
4766
+ * Pure and static for testability: keeps only steps whose OutputData satisfies the
4767
+ * structured contract stamped by the tool executors — a carry-forward-eligible
4768
+ * `toolFamily` (see {@link CarryForwardToolFamilies}) AND a non-empty `tool` name.
4769
+ * Tolerant of missing/invalid OutputData JSON, keeps only successful results,
4770
+ * caps each result and the total under `maxChars` (adding an explicit truncation
4771
+ * note when results are dropped). Returns null when nothing usable remains.
4772
+ */
4773
+ static BuildPriorTurnToolResultsMessage(steps, maxChars) {
4774
+ const sections = [];
4775
+ let usedChars = 0;
4776
+ let dropped = 0;
4777
+ for (const step of steps) {
4778
+ if (!step.OutputData)
4779
+ continue;
4780
+ let parsed;
4781
+ try {
4782
+ parsed = JSON.parse(step.OutputData);
4783
+ }
4784
+ catch {
4785
+ continue;
4786
+ }
4787
+ if (!parsed.toolFamily || !BaseAgent.CarryForwardToolFamilies.includes(parsed.toolFamily))
4788
+ continue;
4789
+ if (typeof parsed.tool !== 'string' || parsed.tool.length === 0)
4790
+ continue;
4791
+ if (parsed.result?.success !== true)
4792
+ continue;
4793
+ const section = FormatToolResultSection({ tool: parsed.tool, input: parsed.input }, RenderToolResultData(parsed.result.data));
4794
+ if (usedChars + section.length > maxChars && sections.length > 0) {
4795
+ dropped++;
4796
+ continue;
4797
+ }
4798
+ const capped = section.length > maxChars
4799
+ ? `${section.slice(0, maxChars)}\n[truncated]`
4800
+ : section;
4801
+ usedChars += capped.length;
4802
+ sections.push(capped);
4803
+ }
4804
+ if (sections.length === 0) {
4805
+ return null;
4806
+ }
4807
+ const header = 'Tool results from your previous turn (still valid — reuse instead of re-calling):';
4808
+ const droppedNote = dropped > 0 ? `\n\n[${dropped} additional result(s) omitted for size — re-call those tools if needed]` : '';
4809
+ return `${header}\n${sections.join('\n\n')}${droppedNote}`;
4810
+ }
4811
+ /**
4812
+ * Builds the summarizeRange recursive-sub-call host: resolves the seeded
4813
+ * 'Summarize Conversation Range' prompt (priority-ordered cheap models — the RLM
4814
+ * "strong root model, cheap sub-call model" split) and runs it via the standard
4815
+ * prompt runner so the AIPromptRun records itself.
4816
+ * @protected
4817
+ */
4818
+ buildConversationSummaryHost(params) {
4819
+ return {
4820
+ RunSummaryPrompt: async (rangeText, lens) => {
4821
+ // Trimmed, case-insensitive name lookup (the AIPromptRunner 'Repair JSON'
4822
+ // style) so cosmetic re-casing of the seeded prompt can't break the tool.
4823
+ const targetName = BaseAgent.SummarizeRangePromptName.toLowerCase();
4824
+ const prompt = AIEngine.Instance.Prompts.find(p => p.Name.trim().toLowerCase() === targetName);
4825
+ if (!prompt) {
4826
+ throw new Error(`The '${BaseAgent.SummarizeRangePromptName}' system prompt is not present in this environment`);
4827
+ }
4828
+ const promptParams = new AIPromptParams();
4829
+ promptParams.prompt = prompt;
4830
+ // Keys are the summarize-range.template.md contract ({{ lens }}, {{ messages }})
4831
+ promptParams.data = { lens, messages: rangeText };
4832
+ promptParams.contextUser = params.contextUser;
4833
+ if (this._agentRun) {
4834
+ // Link the sub-call's AIPromptRun to this agent run (AgentRunID) so
4835
+ // per-run cost rollups include the recursive summarization spend.
4836
+ promptParams.agentRunId = this._agentRun.ID;
4837
+ }
4838
+ const result = await this._promptRunner.ExecutePrompt(promptParams);
4839
+ const text = ExtractPromptResultText(result);
4840
+ if (!result.success || text.length === 0) {
4841
+ throw new Error(result.errorMessage || 'summarizeRange sub-call returned no content');
4842
+ }
4843
+ return { text, promptRunId: result.promptRun?.ID };
4844
+ }
4845
+ };
4846
+ }
4847
+ /**
4848
+ * Executes conversation-history retrieval tool calls, wrapping each invocation in
4849
+ * its own AIAgentRunStep (StepType='Tool', "Conversation Tool: {tool}") — the same
4850
+ * per-call observability shape as artifact tools. Reads are served from the
4851
+ * ConversationEngine cache; per-call failures are contained in the result.
4852
+ *
4853
+ * At most {@link MAX_CONVERSATION_TOOL_CALLS_PER_TURN} calls execute per response;
4854
+ * the excess come back as skipped failure-shaped results (no run steps recorded)
4855
+ * telling the model to re-request them next turn.
4856
+ *
4857
+ * @protected
4858
+ */
4859
+ async executeConversationToolCallsAsSteps(calls, params) {
4860
+ // Per-turn fan-out cap: each call is a run step (summarizeRange a full LLM
4861
+ // sub-call) — excess calls are reported back as skipped failure-shaped results
4862
+ // through the normal rendering path so the model can re-request them next turn.
4863
+ // Deliberately no DB rows for skipped calls (zero I/O for work not done; the
4864
+ // Status union has no 'Skipped' and Failed steps would pollute failure metrics).
4865
+ const callsToExecute = calls.slice(0, MAX_CONVERSATION_TOOL_CALLS_PER_TURN);
4866
+ const skippedCalls = calls.slice(MAX_CONVERSATION_TOOL_CALLS_PER_TURN);
4867
+ if (skippedCalls.length > 0) {
4868
+ this.logStatus(`[ConversationTools] ${calls.length} calls requested — executing first ${callsToExecute.length}, skipping ${skippedCalls.length} (per-turn cap)`, true, params);
4869
+ }
4870
+ const executedResults = await Promise.all(callsToExecute.map(async (call) => {
4871
+ const toolStep = await this.createStepEntity({
4872
+ stepType: 'Tool',
4873
+ stepName: `Conversation Tool: ${call.tool}`,
4874
+ contextUser: params.contextUser,
4875
+ inputData: {
4876
+ tool: call.tool,
4877
+ input: call.input,
4878
+ conversationId: params.conversationId,
4879
+ },
4880
+ });
4881
+ const executed = await this._conversationToolManager.ExecuteSingleToolCall(call);
4882
+ // summarizeRange's recursive LLM sub-call records an AIPromptRun — link it
4883
+ // through this Tool step's TargetLogID (one step + one prompt run: full
4884
+ // lineage without a duplicate Prompt step for the same call).
4885
+ if (executed.promptRunId) {
4886
+ toolStep.TargetLogID = executed.promptRunId;
4887
+ }
4888
+ // Typed cross-turn contract — see CarryForwardToolStepOutput (read back by
4889
+ // BuildPriorTurnToolResultsMessage on the next run; StepName is display-only).
4890
+ const carryForwardOutput = {
4891
+ toolFamily: CarryForwardToolFamily.Conversation,
4892
+ tool: executed.tool,
4893
+ input: executed.input,
4894
+ result: executed.result,
4895
+ durationMs: executed.durationMs,
4896
+ ...(executed.promptRunId && { promptRunId: executed.promptRunId }),
4897
+ };
4898
+ await this.finalizeStepEntity(toolStep, executed.result.success, executed.result.success ? undefined : executed.result.errorMessage, carryForwardOutput);
4899
+ return executed;
4900
+ }));
4901
+ const skippedResults = skippedCalls.map(call => ({
4902
+ tool: call.tool,
4903
+ input: call.input,
4904
+ result: {
4905
+ success: false,
4906
+ errorMessage: `Skipped — per-turn cap of ${MAX_CONVERSATION_TOOL_CALLS_PER_TURN} conversation tool calls reached. Re-request this call on your next turn.`,
4907
+ },
4908
+ durationMs: 0,
4909
+ }));
4910
+ return [...executedResults, ...skippedResults];
4911
+ }
4912
+ /**
4913
+ * Pushes a single user-role message containing rendered conversation-tool results
4914
+ * into the conversation — the same inject-once-then-expire lifecycle as artifact
4915
+ * tool results.
4916
+ *
4917
+ * @protected
4918
+ */
4919
+ injectConversationToolResultsMessage(params, toolResults) {
4920
+ if (toolResults.length === 0)
4921
+ return;
4922
+ const header = BaseAgent.conversationToolResultsHeader(toolResults.length);
4923
+ const body = toolResults.map((r, i) => {
4924
+ const parts = { tool: r.tool, input: r.input, ordinal: i + 1 };
4925
+ if (r.result.success) {
4926
+ const data = this.capStandaloneToolResultText(RenderToolResultData(r.result.data));
4927
+ return FormatToolResultSection(parts, data);
4928
+ }
4929
+ return FormatToolErrorSection(parts, r.result.errorMessage);
4930
+ }).join('\n\n');
4931
+ const message = {
4932
+ role: 'user',
4933
+ content: `${header}\n${body}`,
4934
+ metadata: {
4935
+ turnAdded: this._promptTurnCount,
4936
+ messageType: BaseAgent.toolResultMessageType,
4937
+ expirationTurns: 3,
4938
+ expirationMode: 'Compact',
4939
+ compactMode: 'First N Chars',
4940
+ compactLength: 500,
4941
+ compactPromptId: '',
4942
+ },
4943
+ };
4944
+ params.conversationMessages.push(message);
4945
+ }
4434
4946
  /**
4435
4947
  * Pushes a single user-role message containing rendered artifact-tool
4436
4948
  * results into the conversation. This mirrors the action-result
@@ -4444,19 +4956,14 @@ The context is now within limits. Please retry your request with the recovered c
4444
4956
  injectArtifactToolResultsMessage(params, toolResults) {
4445
4957
  if (toolResults.length === 0)
4446
4958
  return;
4447
- const header = toolResults.length === 1
4448
- ? 'Artifact tool result:'
4449
- : `Artifact tool results (${toolResults.length} calls):`;
4959
+ const header = BaseAgent.artifactToolResultsHeader(toolResults.length);
4450
4960
  const body = toolResults.map((r, i) => {
4451
- const heading = `### ${i + 1}. ${r.artifactId}.${r.tool}(${JSON.stringify(r.input)})`;
4961
+ const parts = { tool: r.tool, input: r.input, ordinal: i + 1, signaturePrefix: r.artifactId };
4452
4962
  if (r.result.success) {
4453
- const raw = typeof r.result.data === 'string'
4454
- ? r.result.data
4455
- : JSON.stringify(r.result.data, null, 2);
4456
- const data = this.capStandaloneToolResultText(raw);
4457
- return `${heading}\n\`\`\`json\n${data}\n\`\`\``;
4963
+ const data = this.capStandaloneToolResultText(RenderToolResultData(r.result.data));
4964
+ return FormatToolResultSection(parts, data);
4458
4965
  }
4459
- return `${heading}\n**Error:** ${r.result.errorMessage}`;
4966
+ return FormatToolErrorSection(parts, r.result.errorMessage);
4460
4967
  }).join('\n\n');
4461
4968
  const message = {
4462
4969
  role: 'user',
@@ -5082,6 +5589,7 @@ The context is now within limits. Please retry your request with the recovered c
5082
5589
  { docsFlag: 'includeWhileDocs', responseTypeKey: 'while' },
5083
5590
  { docsFlag: 'includeScratchpadDocs', responseTypeKey: 'scratchpad' },
5084
5591
  { docsFlag: 'includeArtifactToolsDocs', responseTypeKey: 'artifactToolCalls' },
5592
+ { docsFlag: 'includeConversationToolsDocs', responseTypeKey: 'conversationToolCalls' },
5085
5593
  { docsFlag: 'includePipelineDocs', responseTypeKey: 'pipeline' },
5086
5594
  { docsFlag: 'includeMemoryWritesDocs', responseTypeKey: 'memoryWrites' }
5087
5595
  ];
@@ -6331,6 +6839,16 @@ The context is now within limits. Please retry your request with the recovered c
6331
6839
  if (skillsForStep && skillsForStep.length > 0) {
6332
6840
  stepEntity.Skills = JSON.stringify(skillsForStep);
6333
6841
  }
6842
+ // Completed-at-creation steps: stamp the terminal state NOW so the INSERT below is the
6843
+ // step's ONLY write (same shared helper + OutputData treatment finalizeStepEntity uses).
6844
+ if (params.completed) {
6845
+ finalizeAgentRunStep(stepEntity, {
6846
+ success: params.completed.success,
6847
+ errorMessage: params.completed.errorMessage,
6848
+ outputData: params.completed.outputData ? CopyScalarsAndArrays(params.completed.outputData, true) : undefined,
6849
+ completedAt: new Date()
6850
+ });
6851
+ }
6334
6852
  // Fire-and-forget the 'started' INSERT — the agent flow never blocks on a step save. The queue
6335
6853
  // tracks the INSERT so every later UPDATE (queueStepSave) chains AFTER it commits.
6336
6854
  // When the step has a parent, chain the INSERT AFTER the parent's INSERT to satisfy the
@@ -6607,7 +7125,19 @@ The context is now within limits. Please retry your request with the recovered c
6607
7125
  // Check if this is a message expansion request
6608
7126
  if (previousDecision.messageIndex !== undefined) {
6609
7127
  // Handle message expansion before retrying
6610
- this.executeExpandMessageStep(previousDecision, params, this._promptTurnCount);
7128
+ const expandFailure = this.executeExpandMessageStep(previousDecision, params, this._promptTurnCount);
7129
+ if (expandFailure) {
7130
+ // A failed expansion MUST NOT leave the loop state unchanged: the model
7131
+ // re-requests the identical expansion forever (observed live when a
7132
+ // spliced cross-turn summary message — which has no expanded form — was
7133
+ // requested for expansion; the silent no-op produced an unbounded Retry
7134
+ // loop that exhausted the process heap). Surface the failure into the
7135
+ // conversation so the next prompt steers the model away.
7136
+ params.conversationMessages.push({
7137
+ role: 'user',
7138
+ content: `Message expansion failed: ${expandFailure}`
7139
+ });
7140
+ }
6611
7141
  }
6612
7142
  return await this.executePromptStep(params, config, previousDecision, stepCount);
6613
7143
  case 'Sub-Agent':
@@ -6839,6 +7369,12 @@ The context is now within limits. Please retry your request with the recovered c
6839
7369
  stepEntity.PromptRun = promptResult.promptRun; // transient related object (not a persisted field)
6840
7370
  // don't save here, we save when we call finalizeStepEntity()
6841
7371
  }
7372
+ // Remember the most recent model selection — cross-turn compaction resolves its
7373
+ // effective budget against "the model about to run", and the last prompt's
7374
+ // selection is the best available proxy for the next turn's model.
7375
+ if (promptResult.modelSelectionInfo) {
7376
+ this._lastModelSelectionInfo = promptResult.modelSelectionInfo;
7377
+ }
6842
7378
  // Check if prompt execution failed
6843
7379
  if (!promptResult.success) {
6844
7380
  // CRITICAL FIX: Preserve payload before finalizing step
@@ -6953,6 +7489,19 @@ The context is now within limits. Please retry your request with the recovered c
6953
7489
  else if (this._artifactToolManager.HasArtifacts()) {
6954
7490
  this.logStatus(`[ArtifactTools] LLM did not use artifact tools this turn (artifacts available but not accessed)`, true, params);
6955
7491
  }
7492
+ // Execute conversation-history retrieval tool calls if provided (zero turn cost —
7493
+ // processed inline, results delivered as a conversation message next turn)
7494
+ const conversationToolCalls = initialNextStep.conversationToolCalls;
7495
+ if (conversationToolCalls?.length) {
7496
+ if (this._conversationToolManager.IsAvailable) {
7497
+ this.logStatus(`[ConversationTools] LLM requested ${conversationToolCalls.length} tool call(s): ${conversationToolCalls.map(c => c.tool).join(', ')}`, true, params);
7498
+ const conversationToolResults = await this.executeConversationToolCallsAsSteps(conversationToolCalls, params);
7499
+ this.injectConversationToolResultsMessage(params, conversationToolResults);
7500
+ }
7501
+ else {
7502
+ this.logStatus(`[ConversationTools] LLM requested conversation tools but the run has no conversationId — ignored`, true, params);
7503
+ }
7504
+ }
6956
7505
  // Execute in-flight memory writes if provided (zero turn cost — processed inline)
6957
7506
  const memoryWrites = initialNextStep.memoryWrites;
6958
7507
  if (memoryWrites?.length) {
@@ -10168,13 +10717,7 @@ The context is now within limits. Please retry your request with the recovered c
10168
10717
  this._agentRun.Success = false;
10169
10718
  this._agentRun.ErrorMessage = errorMessage;
10170
10719
  // Calculate total tokens even for failed runs
10171
- const tokenStats = this.calculateTokenStats();
10172
- this._agentRun.TotalTokensUsed = tokenStats.totalTokens;
10173
- this._agentRun.TotalPromptTokensUsed = tokenStats.promptTokens;
10174
- this._agentRun.TotalCompletionTokensUsed = tokenStats.completionTokens;
10175
- this._agentRun.TotalCacheReadTokensUsed = tokenStats.cacheReadTokens;
10176
- this._agentRun.TotalCacheWriteTokensUsed = tokenStats.cacheWriteTokens;
10177
- this._agentRun.TotalCost = tokenStats.totalCost;
10720
+ this.applyTokenStatsToRun(this._agentRun, this.calculateTokenStats());
10178
10721
  await this._agentRun.Save();
10179
10722
  }
10180
10723
  return {
@@ -10196,13 +10739,7 @@ The context is now within limits. Please retry your request with the recovered c
10196
10739
  this._agentRun.Success = false;
10197
10740
  this._agentRun.ErrorMessage = message;
10198
10741
  // Calculate total tokens even for cancelled runs
10199
- const tokenStats = this.calculateTokenStats();
10200
- this._agentRun.TotalTokensUsed = tokenStats.totalTokens;
10201
- this._agentRun.TotalPromptTokensUsed = tokenStats.promptTokens;
10202
- this._agentRun.TotalCompletionTokensUsed = tokenStats.completionTokens;
10203
- this._agentRun.TotalCacheReadTokensUsed = tokenStats.cacheReadTokens;
10204
- this._agentRun.TotalCacheWriteTokensUsed = tokenStats.cacheWriteTokens;
10205
- this._agentRun.TotalCost = tokenStats.totalCost;
10742
+ this.applyTokenStatsToRun(this._agentRun, this.calculateTokenStats());
10206
10743
  await this._agentRun.Save();
10207
10744
  }
10208
10745
  return {
@@ -10272,17 +10809,21 @@ The context is now within limits. Please retry your request with the recovered c
10272
10809
  this._agentRun.FinalPayloadObject = resolvedPayload;
10273
10810
  this._agentRun.FinalPayload = finalPayloadJson;
10274
10811
  // Calculate total tokens from all prompts and sub-agents
10275
- const tokenStats = this.calculateTokenStats();
10276
- this._agentRun.TotalTokensUsed = tokenStats.totalTokens;
10277
- this._agentRun.TotalPromptTokensUsed = tokenStats.promptTokens;
10278
- this._agentRun.TotalCompletionTokensUsed = tokenStats.completionTokens;
10279
- this._agentRun.TotalCacheReadTokensUsed = tokenStats.cacheReadTokens;
10280
- this._agentRun.TotalCacheWriteTokensUsed = tokenStats.cacheWriteTokens;
10281
- this._agentRun.TotalCost = tokenStats.totalCost;
10812
+ this.applyTokenStatsToRun(this._agentRun, this.calculateTokenStats());
10282
10813
  const ok = await this._agentRun.Save();
10283
10814
  if (!ok) {
10284
10815
  LogError(`Failed to finalize agent run ${this._agentRun.ID}`);
10285
10816
  }
10817
+ else {
10818
+ // Hand the NEXT turn's carry-forward check this run's tool results straight
10819
+ // from memory, so it can skip its DB lookups (see PriorTurnToolResultCache).
10820
+ this.cachePriorTurnToolResults();
10821
+ }
10822
+ // Cross-turn compaction (post-turn, the primary path): fire-and-forget AFTER the
10823
+ // run row is final so the summary-LLM latency never delays the caller's
10824
+ // completion event. Errors are contained — a failed pass leaves the conversation
10825
+ // untouched and simply re-triggers on a later turn.
10826
+ this.startPostTurnCompaction();
10286
10827
  }
10287
10828
  // Also promote any media from the final step's promoteMediaOutputs
10288
10829
  if (finalStep.promoteMediaOutputs && finalStep.promoteMediaOutputs.length > 0) {
@@ -10322,7 +10863,7 @@ The context is now within limits. Please retry your request with the recovered c
10322
10863
  // Iterate through the agent run's steps to sum up tokens
10323
10864
  if (this._agentRun?.Steps) {
10324
10865
  for (const step of this._agentRun.Steps) {
10325
- if (step.StepType === 'Prompt' && step.PromptRun) {
10866
+ if ((step.StepType === 'Prompt' || step.StepType === 'Compaction') && step.PromptRun) {
10326
10867
  // Add tokens from prompt runs (rollup fields include any nested child prompt runs)
10327
10868
  totalTokens += step.PromptRun.TokensUsedRollup || 0;
10328
10869
  promptTokens += step.PromptRun.TokensPromptRollup || 0;
@@ -10344,6 +10885,21 @@ The context is now within limits. Please retry your request with the recovered c
10344
10885
  }
10345
10886
  return { totalTokens, promptTokens, completionTokens, cacheReadTokens, cacheWriteTokens, totalCost };
10346
10887
  }
10888
+ /**
10889
+ * Applies a {@link calculateTokenStats} result to a run entity's six denormalized
10890
+ * token/cost columns — the single source for the assignment shape shared by the
10891
+ * failure/cancel/finalize paths AND the post-turn compaction top-up
10892
+ * ({@link recordCompactionRunStep}).
10893
+ * @private
10894
+ */
10895
+ applyTokenStatsToRun(run, tokenStats) {
10896
+ run.TotalTokensUsed = tokenStats.totalTokens;
10897
+ run.TotalPromptTokensUsed = tokenStats.promptTokens;
10898
+ run.TotalCompletionTokensUsed = tokenStats.completionTokens;
10899
+ run.TotalCacheReadTokensUsed = tokenStats.cacheReadTokens;
10900
+ run.TotalCacheWriteTokensUsed = tokenStats.cacheWriteTokens;
10901
+ run.TotalCost = tokenStats.totalCost;
10902
+ }
10347
10903
  /**
10348
10904
  * Gets the count of how many times a specific action has been executed in this agent run.
10349
10905
  *
@@ -10549,6 +11105,225 @@ The context is now within limits. Please retry your request with the recovered c
10549
11105
  `${messagesToRemove.length} removed`);
10550
11106
  }
10551
11107
  }
11108
+ // =====================================================================================
11109
+ // CROSS-TURN (TIER A) CONVERSATION COMPACTION HOOKS
11110
+ // Durable summary layer per plans/agent-conversation-compaction.md. All hooks are
11111
+ // gated on params.conversationId + root depth — programmatic runs, sub-agents, and
11112
+ // tests without a conversation are untouched. Trigger math / boundary selection /
11113
+ // the boundary-row write live in ConversationCompactionManager; BaseAgent owns
11114
+ // budget resolution (it knows the model) and run-step recording.
11115
+ // =====================================================================================
11116
+ /**
11117
+ * Resolves the effective context budget for cross-turn compaction, validated against
11118
+ * the most recent prompt's model when available. Logs the clamp warning once when a
11119
+ * configured budget exceeded the model's MaxInputTokens.
11120
+ * @protected
11121
+ */
11122
+ resolveCompactionBudget(params, config) {
11123
+ const modelMax = this._lastModelSelectionInfo
11124
+ ? this.tryGetModelMaxInputTokens(this._lastModelSelectionInfo)
11125
+ : null;
11126
+ const budget = ConversationCompactionManager.ResolveEffectiveBudget(params.agent, config?.agentType || null, modelMax);
11127
+ if (budget.ClampedToModel) {
11128
+ // Verbose-only: this re-evaluates every turn while the budget stays mis-set, and
11129
+ // the clamp is already captured structurally in CompactionOutcome.Warnings → the
11130
+ // Compaction step's OutputData (§8: keep debug detail, don't spam info logs).
11131
+ this.logStatus(`⚠️ [CrossTurnCompaction] Configured ContextWindowMaxTokens exceeds the model's MaxInputTokens — clamped to ${budget.MaxTokens}`, true, params);
11132
+ }
11133
+ return budget;
11134
+ }
11135
+ /**
11136
+ * Pre-turn fallback: compacts synchronously when the assembled window is already over
11137
+ * the trigger budget BEFORE the first prompt of this run, then splices the fresh
11138
+ * summary into the live message array. Only runs with an EXPLICIT configured budget
11139
+ * (agent or type ContextWindowMaxTokens) — before the first prompt the model is
11140
+ * unknown, and compacting against the conservative default would over-trigger on
11141
+ * large-context models. The post-turn hook (real model known) covers those.
11142
+ * @protected
11143
+ */
11144
+ async checkPreTurnCompaction(params, config) {
11145
+ if (!params.conversationId || this._depth !== 0) {
11146
+ return;
11147
+ }
11148
+ const budget = this.resolveCompactionBudget(params, config);
11149
+ if (budget.BoundedBy !== 'Agent' && budget.BoundedBy !== 'AgentType') {
11150
+ return;
11151
+ }
11152
+ const estimatedTokens = this.estimateConversationTokens(params.conversationMessages);
11153
+ if (estimatedTokens < budget.TriggerTokens) {
11154
+ return;
11155
+ }
11156
+ this.logStatus(`🗜️ [CrossTurnCompaction] Pre-turn window ~${estimatedTokens} tokens ≥ trigger ${budget.TriggerTokens} — compacting before first prompt`, true, params);
11157
+ const outcome = await this.runCrossTurnCompaction('pre-turn', params, config, budget);
11158
+ if (outcome?.Fired && outcome.BoundarySequence !== undefined && outcome.SummaryText) {
11159
+ this.applyCompactionToLiveMessages(params.conversationMessages, outcome.BoundarySequence, outcome.SummaryText);
11160
+ }
11161
+ }
11162
+ /**
11163
+ * Post-turn hook (the primary path), called from {@link finalizeAgentRun} after the
11164
+ * run row is saved. Fire-and-forget by design: the caller's completion event never
11165
+ * waits on the summary LLM call. Fires for settled root runs with a conversation —
11166
+ * {@link settledRunStatuses}: 'Completed' AND 'AwaitingFeedback', because a Chat
11167
+ * final step (→ AwaitingFeedback) is the NORMAL ending of a conversational turn;
11168
+ * gating on 'Completed' alone silently disabled post-turn compaction for exactly
11169
+ * the long-chat scenario this feature targets.
11170
+ * @protected
11171
+ */
11172
+ startPostTurnCompaction() {
11173
+ const params = this._executeParams;
11174
+ if (!params?.conversationId || this._depth !== 0 || !this._agentRun
11175
+ || !BaseAgent.settledRunStatuses.includes(this._agentRun.Status)) {
11176
+ return;
11177
+ }
11178
+ const config = this._agentConfig;
11179
+ const budget = this.resolveCompactionBudget(params, config);
11180
+ void this.runCrossTurnCompaction('post-turn', params, config, budget).catch(error => {
11181
+ LogError(`Post-turn cross-turn compaction error (contained): ${error instanceof Error ? error.message : error}`);
11182
+ });
11183
+ }
11184
+ /**
11185
+ * Runs one compaction pass and records it as a `StepType='Compaction'` run step —
11186
+ * TargetID = the summary prompt, TargetLogID = the summary AIPromptRun (the same ID
11187
+ * written to `ConversationDetail.SummaryPromptRunID`, closing the lineage chain).
11188
+ * Quiet no-ops (window under trigger) record no step; fired passes and failures do.
11189
+ * @protected
11190
+ */
11191
+ async runCrossTurnCompaction(phase, params, config, budget) {
11192
+ if (!params.conversationId || !this._agentRun) {
11193
+ return undefined;
11194
+ }
11195
+ const outcome = await ConversationCompactionManager.CompactIfNeeded({
11196
+ ConversationId: params.conversationId,
11197
+ Agent: params.agent,
11198
+ AgentType: config?.agentType || null,
11199
+ Budget: budget,
11200
+ ContextUser: params.contextUser,
11201
+ Provider: this.ProviderToUse,
11202
+ EstimateTokens: (messages) => this.estimateConversationTokens(messages),
11203
+ Verbose: params.verbose,
11204
+ // The in-flight agent-response placeholder row: a post-turn pass runs while
11205
+ // the resolver may still be writing its Message — keep it out of the window
11206
+ // so the boundary can never land on it.
11207
+ ExcludeDetailIds: params.conversationDetailId ? [params.conversationDetailId] : undefined,
11208
+ AgentRunId: this._agentRun.ID
11209
+ });
11210
+ if (outcome.Fired || outcome.ErrorMessage) {
11211
+ await this.recordCompactionRunStep(phase, params, budget, outcome);
11212
+ }
11213
+ return outcome;
11214
+ }
11215
+ /**
11216
+ * Persists the Compaction run step for a fired or failed pass — as a SINGLE INSERT:
11217
+ * the pass is already over when this is called, so the step is created pre-finalized
11218
+ * via `createStepEntity`'s `completed` option instead of paying a second UPDATE
11219
+ * round trip. The summary AIPromptRun rides on the step's transient `PromptRun` so
11220
+ * {@link calculateTokenStats}'s Compaction branch counts it: pre-turn fires are
11221
+ * picked up by finalizeAgentRun's normal rollup for free; post-turn fires happen
11222
+ * AFTER that rollup ran, so this method tops the run's token columns up itself.
11223
+ * @private
11224
+ */
11225
+ async recordCompactionRunStep(phase, params, budget, outcome) {
11226
+ try {
11227
+ const stepEntity = await this.createStepEntity({
11228
+ stepType: 'Compaction',
11229
+ stepName: `Cross-Turn Conversation Compaction (${phase})`,
11230
+ contextUser: params.contextUser,
11231
+ targetId: outcome.PromptId,
11232
+ targetLogId: outcome.PromptRunId,
11233
+ inputData: {
11234
+ phase,
11235
+ conversationId: params.conversationId,
11236
+ budget
11237
+ },
11238
+ completed: {
11239
+ success: !outcome.ErrorMessage,
11240
+ errorMessage: outcome.ErrorMessage,
11241
+ outputData: {
11242
+ fired: outcome.Fired,
11243
+ boundarySequence: outcome.BoundarySequence,
11244
+ tokensBefore: outcome.TokensBefore,
11245
+ tokensAfter: outcome.TokensAfter,
11246
+ summaryLength: outcome.SummaryText?.length,
11247
+ promptRunId: outcome.PromptRunId,
11248
+ warnings: outcome.Warnings
11249
+ }
11250
+ }
11251
+ });
11252
+ if (outcome.PromptRun) {
11253
+ stepEntity.PromptRun = outcome.PromptRun;
11254
+ }
11255
+ if (phase === 'post-turn') {
11256
+ // finalizeAgentRun's flush has already run — drain this step's INSERT now.
11257
+ await this._stepSaveQueue.Flush();
11258
+ await this.topUpRunTokenTotalsAfterPostTurnCompaction(outcome, params.contextUser);
11259
+ }
11260
+ }
11261
+ catch (error) {
11262
+ LogError(`Failed to record Compaction run step (compaction itself ${outcome.Fired ? 'succeeded' : 'failed'}): ${error instanceof Error ? error.message : error}`);
11263
+ }
11264
+ }
11265
+ /**
11266
+ * After a fired POST-turn compaction, folds the summary prompt's tokens/cost into
11267
+ * the run row — finalizeAgentRun's rollup ran before the pass, so without this the
11268
+ * recursive summary spend would be missing from the run's denormalized totals.
11269
+ * Uses a FRESH-loaded run entity for the write: the persisted run may be
11270
+ * 'AwaitingFeedback' and a quick user reply could have resumed it — re-Saving the
11271
+ * stale in-memory `_agentRun` would clobber the resumed row's Status. Residual: the
11272
+ * token columns are last-writer-wins in the tiny Load→Save window (self-healing at
11273
+ * the resumed run's own finalize). Failures are contained (LogError only).
11274
+ * @private
11275
+ */
11276
+ async topUpRunTokenTotalsAfterPostTurnCompaction(outcome, contextUser) {
11277
+ if (!outcome.Fired || !outcome.PromptRun || !this._agentRun) {
11278
+ return;
11279
+ }
11280
+ const tokenStats = this.calculateTokenStats();
11281
+ const runUpdate = await this._activeProvider.GetEntityObject('MJ: AI Agent Runs', contextUser);
11282
+ if (!(await runUpdate.Load(this._agentRun.ID))) {
11283
+ LogError(`Post-turn compaction token top-up: failed to load run ${this._agentRun.ID}`);
11284
+ return;
11285
+ }
11286
+ this.applyTokenStatsToRun(runUpdate, tokenStats);
11287
+ if (!(await runUpdate.Save())) {
11288
+ LogError(`Post-turn compaction token top-up: save failed for run ${this._agentRun.ID}: ${runUpdate.LatestResult?.CompleteMessage || 'unknown error'}`);
11289
+ }
11290
+ }
11291
+ /**
11292
+ * Splices a freshly generated summary into the live message array in place: every
11293
+ * message covered by the new boundary (sequence below it, or a prior summary
11294
+ * message) collapses into one summary message; enrichment-bearing tail messages and
11295
+ * injected messages without sequence metadata are preserved untouched.
11296
+ * @private
11297
+ */
11298
+ applyCompactionToLiveMessages(messages, boundarySequence, summaryText) {
11299
+ const retained = [];
11300
+ let summaryInserted = false;
11301
+ for (const message of messages) {
11302
+ const metadata = message.metadata;
11303
+ const covered = metadata?.isConversationSummary === true
11304
+ || (metadata?.sequence !== undefined && metadata.sequence < boundarySequence);
11305
+ if (covered) {
11306
+ if (!summaryInserted) {
11307
+ const summaryMessage = {
11308
+ role: 'user',
11309
+ content: summaryText,
11310
+ metadata: {
11311
+ isConversationSummary: true,
11312
+ summaryBoundarySequence: boundarySequence,
11313
+ sequence: boundarySequence
11314
+ }
11315
+ };
11316
+ retained.push(summaryMessage);
11317
+ summaryInserted = true;
11318
+ }
11319
+ }
11320
+ else {
11321
+ retained.push(message);
11322
+ }
11323
+ }
11324
+ messages.length = 0;
11325
+ messages.push(...retained);
11326
+ }
10552
11327
  /**
10553
11328
  * Creates an AIAgentRunStep for message compaction operations.
10554
11329
  * Records the compaction attempt with context about the message being compacted.
@@ -10566,7 +11341,7 @@ The context is now within limits. Please retry your request with the recovered c
10566
11341
  const step = await (params.provider || this._activeProvider).GetEntityObject('MJ: AI Agent Run Steps', params.contextUser);
10567
11342
  step.NewRecord();
10568
11343
  step.AgentRunID = this._agentRun.ID;
10569
- step.StepType = 'Prompt';
11344
+ step.StepType = 'Compaction';
10570
11345
  step.Status = 'Running';
10571
11346
  step.InputData = JSON.stringify({
10572
11347
  stepName: 'Message Compaction',
@@ -10708,7 +11483,7 @@ The context is now within limits. Please retry your request with the recovered c
10708
11483
  const messageType = msg.metadata?.messageType;
10709
11484
  return messageType === 'action-result'
10710
11485
  || messageType === 'client-tool-result'
10711
- || messageType === 'tool-result';
11486
+ || messageType === BaseAgent.toolResultMessageType;
10712
11487
  }
10713
11488
  /**
10714
11489
  * Returns true if the message is a turn-generated result (action, tool, client tool,
@@ -10721,7 +11496,7 @@ The context is now within limits. Please retry your request with the recovered c
10721
11496
  const messageType = msg.metadata?.messageType;
10722
11497
  return messageType === 'action-result'
10723
11498
  || messageType === 'client-tool-result'
10724
- || messageType === 'tool-result'
11499
+ || messageType === BaseAgent.toolResultMessageType
10725
11500
  || messageType === 'sub-agent-result'
10726
11501
  || messageType === 'loop-result';
10727
11502
  }
@@ -10763,47 +11538,42 @@ The context is now within limits. Please retry your request with the recovered c
10763
11538
  getModelContextLimit(modelSelectionInfo) {
10764
11539
  // Default conservative limit if we can't determine the actual limit
10765
11540
  const DEFAULT_LIMIT = 8000;
10766
- if (!modelSelectionInfo) {
10767
- this.logStatus(`No model selection info available, using default limit: ${DEFAULT_LIMIT}`, true);
10768
- return DEFAULT_LIMIT;
11541
+ const known = modelSelectionInfo ? this.tryGetModelMaxInputTokens(modelSelectionInfo) : null;
11542
+ if (known === null) {
11543
+ this.logStatus(`Could not determine model context limit, using default limit: ${DEFAULT_LIMIT}`, true);
10769
11544
  }
11545
+ return known || DEFAULT_LIMIT;
11546
+ }
11547
+ /**
11548
+ * Extracts the vendor-specific MaxInputTokens from model selection info, returning
11549
+ * null when it genuinely cannot be determined. Callers that need a hard number use
11550
+ * {@link getModelContextLimit} (which falls back to a conservative default); callers
11551
+ * for whom a guessed default would be WRONG — e.g. cross-turn compaction budget
11552
+ * clamping, where a bogus 8000 would clamp a configured 200k budget — use this and
11553
+ * handle null explicitly.
11554
+ * @protected
11555
+ */
11556
+ tryGetModelMaxInputTokens(modelSelectionInfo) {
10770
11557
  try {
10771
- // Get the selected model and vendor from the model selection info
10772
11558
  const modelSelected = modelSelectionInfo.modelSelected;
10773
11559
  const vendorSelected = modelSelectionInfo.vendorSelected;
10774
- if (!modelSelected) {
10775
- this.logStatus(`No model selected in model selection info, using default limit: ${DEFAULT_LIMIT}`, true);
10776
- return DEFAULT_LIMIT;
10777
- }
10778
- // If no vendor selected, can't determine model-specific limit
10779
- if (!vendorSelected) {
10780
- this.logStatus(`No vendor selected, using default limit: ${DEFAULT_LIMIT}`, true);
10781
- return DEFAULT_LIMIT;
11560
+ if (!modelSelected || !vendorSelected) {
11561
+ return null;
10782
11562
  }
10783
- // Find the ModelVendor entry that matches the selected vendor
10784
11563
  const modelVendors = modelSelected.ModelVendors;
10785
11564
  if (!modelVendors || modelVendors.length === 0) {
10786
- this.logStatus(`No ModelVendors array found on model, using default limit: ${DEFAULT_LIMIT}`, true);
10787
- return DEFAULT_LIMIT;
11565
+ return null;
10788
11566
  }
10789
- // Find the vendor-specific entry
10790
11567
  const vendorEntry = modelVendors.find((mv) => UUIDsEqual(mv.VendorID, vendorSelected.ID));
10791
- if (!vendorEntry) {
10792
- this.logStatus(`No matching vendor entry found in ModelVendors, using default limit: ${DEFAULT_LIMIT}`, true);
10793
- return DEFAULT_LIMIT;
10794
- }
10795
- // Get MaxInputTokens from the vendor-specific entry
10796
- const maxInputTokens = vendorEntry.MaxInputTokens;
10797
- if (!maxInputTokens || maxInputTokens <= 0) {
10798
- this.logStatus(`MaxInputTokens not set or invalid on vendor entry, using default limit: ${DEFAULT_LIMIT}`, true);
10799
- return DEFAULT_LIMIT;
11568
+ if (!vendorEntry || !vendorEntry.MaxInputTokens || vendorEntry.MaxInputTokens <= 0) {
11569
+ return null;
10800
11570
  }
10801
- this.logStatus(`Using vendor-specific MaxInputTokens: ${maxInputTokens} (Model: ${modelSelected.Name}, Vendor: ${vendorSelected.Name})`, true);
10802
- return maxInputTokens;
11571
+ this.logStatus(`Using vendor-specific MaxInputTokens: ${vendorEntry.MaxInputTokens} (Model: ${modelSelected.Name}, Vendor: ${vendorSelected.Name})`, true);
11572
+ return vendorEntry.MaxInputTokens;
10803
11573
  }
10804
11574
  catch (error) {
10805
- this.logStatus(`Error extracting model context limit: ${error}, using default limit: ${DEFAULT_LIMIT}`, true);
10806
- return DEFAULT_LIMIT;
11575
+ this.logStatus(`Error extracting model context limit: ${error}`, true);
11576
+ return null;
10807
11577
  }
10808
11578
  }
10809
11579
  /**
@@ -10832,6 +11602,10 @@ The context is now within limits. Please retry your request with the recovered c
10832
11602
  * @param request - The expand message request
10833
11603
  * @param params - Agent execution parameters
10834
11604
  * @param currentTurn - Current turn number
11605
+ * @returns null when the expansion succeeded; otherwise a model-facing reason the
11606
+ * expansion is impossible. Callers must surface a non-null reason into the next
11607
+ * prompt's context — a silent no-op leaves the loop state identical and the model
11608
+ * re-requests the same expansion indefinitely.
10835
11609
  * @protected
10836
11610
  */
10837
11611
  executeExpandMessageStep(request, params, currentTurn) {
@@ -10839,12 +11613,14 @@ The context is now within limits. Please retry your request with the recovered c
10839
11613
  const reason = request.expandReason;
10840
11614
  if (messageIndex === undefined || messageIndex < 0 || messageIndex >= params.conversationMessages.length) {
10841
11615
  console.warn(`Cannot expand message: index ${messageIndex} out of bounds`);
10842
- return;
11616
+ return `message index ${messageIndex} is out of bounds — do not request this expansion again.`;
10843
11617
  }
10844
11618
  const message = params.conversationMessages[messageIndex];
10845
11619
  if (!message.metadata?.canExpand || !message.metadata?.originalContent) {
10846
11620
  console.warn(`Cannot expand message at index ${messageIndex}: not expandable or no original content`);
10847
- return;
11621
+ return message.metadata?.isConversationSummary
11622
+ ? `message ${messageIndex} is the cross-turn conversation summary and has no expanded form. To read the underlying history, use the conversation history tools (${ConversationToolNames.join(', ')}) instead — do not request expansion of this message again.`
11623
+ : `message ${messageIndex} is not expandable (it carries no compacted original content) — do not request this expansion again.`;
10848
11624
  }
10849
11625
  // Restore original content
10850
11626
  message.content = message.metadata.originalContent;
@@ -10862,6 +11638,7 @@ The context is now within limits. Please retry your request with the recovered c
10862
11638
  if (params.verbose) {
10863
11639
  console.log(`[Turn ${currentTurn}] Expanded message at index ${messageIndex}`);
10864
11640
  }
11641
+ return null;
10865
11642
  }
10866
11643
  /**
10867
11644
  * Generic template resolver for loop iterations - extracts from LoopAgentType to make available to all agent types