@memberjunction/ai-agents 5.38.0 → 5.40.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/dist/AgentRunner.d.ts +23 -0
  2. package/dist/AgentRunner.d.ts.map +1 -1
  3. package/dist/AgentRunner.js +81 -26
  4. package/dist/AgentRunner.js.map +1 -1
  5. package/dist/ArtifactToolManager.d.ts +13 -0
  6. package/dist/ArtifactToolManager.d.ts.map +1 -1
  7. package/dist/ArtifactToolManager.js +60 -1
  8. package/dist/ArtifactToolManager.js.map +1 -1
  9. package/dist/agent-run-watchdog.d.ts +109 -0
  10. package/dist/agent-run-watchdog.d.ts.map +1 -0
  11. package/dist/agent-run-watchdog.js +278 -0
  12. package/dist/agent-run-watchdog.js.map +1 -0
  13. package/dist/agent-types/loop-agent-prompt-params.d.ts +13 -0
  14. package/dist/agent-types/loop-agent-prompt-params.d.ts.map +1 -1
  15. package/dist/agent-types/loop-agent-prompt-params.js +3 -1
  16. package/dist/agent-types/loop-agent-prompt-params.js.map +1 -1
  17. package/dist/agent-types/loop-agent-response-type.d.ts +12 -3
  18. package/dist/agent-types/loop-agent-response-type.d.ts.map +1 -1
  19. package/dist/agent-types/loop-agent-response-type.js.map +1 -1
  20. package/dist/agent-types/loop-agent-type.d.ts.map +1 -1
  21. package/dist/agent-types/loop-agent-type.js +41 -2
  22. package/dist/agent-types/loop-agent-type.js.map +1 -1
  23. package/dist/artifact-tools/DataSnapshotToolLibrary.js +2 -2
  24. package/dist/artifact-tools/JSONToolLibrary.d.ts.map +1 -1
  25. package/dist/artifact-tools/JSONToolLibrary.js +7 -1
  26. package/dist/artifact-tools/JSONToolLibrary.js.map +1 -1
  27. package/dist/base-agent.d.ts +69 -1
  28. package/dist/base-agent.d.ts.map +1 -1
  29. package/dist/base-agent.js +294 -24
  30. package/dist/base-agent.js.map +1 -1
  31. package/dist/index.d.ts +2 -0
  32. package/dist/index.d.ts.map +1 -1
  33. package/dist/index.js +2 -0
  34. package/dist/index.js.map +1 -1
  35. package/dist/pipeline/coerce.d.ts +43 -0
  36. package/dist/pipeline/coerce.d.ts.map +1 -0
  37. package/dist/pipeline/coerce.js +92 -0
  38. package/dist/pipeline/coerce.js.map +1 -0
  39. package/dist/pipeline/index.d.ts +20 -0
  40. package/dist/pipeline/index.d.ts.map +1 -0
  41. package/dist/pipeline/index.js +20 -0
  42. package/dist/pipeline/index.js.map +1 -0
  43. package/dist/pipeline/jsonpath-eval.d.ts +29 -0
  44. package/dist/pipeline/jsonpath-eval.d.ts.map +1 -0
  45. package/dist/pipeline/jsonpath-eval.js +144 -0
  46. package/dist/pipeline/jsonpath-eval.js.map +1 -0
  47. package/dist/pipeline/operators.d.ts +15 -0
  48. package/dist/pipeline/operators.d.ts.map +1 -0
  49. package/dist/pipeline/operators.js +332 -0
  50. package/dist/pipeline/operators.js.map +1 -0
  51. package/dist/pipeline/path.d.ts +16 -0
  52. package/dist/pipeline/path.d.ts.map +1 -0
  53. package/dist/pipeline/path.js +34 -0
  54. package/dist/pipeline/path.js.map +1 -0
  55. package/dist/pipeline/pipeline-docs.d.ts +6 -0
  56. package/dist/pipeline/pipeline-docs.d.ts.map +1 -0
  57. package/dist/pipeline/pipeline-docs.js +137 -0
  58. package/dist/pipeline/pipeline-docs.js.map +1 -0
  59. package/dist/pipeline/pipeline-executor.d.ts +72 -0
  60. package/dist/pipeline/pipeline-executor.d.ts.map +1 -0
  61. package/dist/pipeline/pipeline-executor.js +326 -0
  62. package/dist/pipeline/pipeline-executor.js.map +1 -0
  63. package/dist/pipeline/pipeline-registry.d.ts +24 -0
  64. package/dist/pipeline/pipeline-registry.d.ts.map +1 -0
  65. package/dist/pipeline/pipeline-registry.js +28 -0
  66. package/dist/pipeline/pipeline-registry.js.map +1 -0
  67. package/dist/pipeline/pipeline.types.d.ts +89 -0
  68. package/dist/pipeline/pipeline.types.d.ts.map +1 -0
  69. package/dist/pipeline/pipeline.types.js +13 -0
  70. package/dist/pipeline/pipeline.types.js.map +1 -0
  71. package/dist/pipeline/predicate.d.ts +23 -0
  72. package/dist/pipeline/predicate.d.ts.map +1 -0
  73. package/dist/pipeline/predicate.js +306 -0
  74. package/dist/pipeline/predicate.js.map +1 -0
  75. package/dist/pipeline/providers/action-provider.d.ts +21 -0
  76. package/dist/pipeline/providers/action-provider.d.ts.map +1 -0
  77. package/dist/pipeline/providers/action-provider.js +24 -0
  78. package/dist/pipeline/providers/action-provider.js.map +1 -0
  79. package/dist/pipeline/providers/artifact-tool-provider.d.ts +20 -0
  80. package/dist/pipeline/providers/artifact-tool-provider.d.ts.map +1 -0
  81. package/dist/pipeline/providers/artifact-tool-provider.js +27 -0
  82. package/dist/pipeline/providers/artifact-tool-provider.js.map +1 -0
  83. package/dist/pipeline/providers/serialize.d.ts +23 -0
  84. package/dist/pipeline/providers/serialize.d.ts.map +1 -0
  85. package/dist/pipeline/providers/serialize.js +99 -0
  86. package/dist/pipeline/providers/serialize.js.map +1 -0
  87. package/dist/pipeline/template.d.ts +18 -0
  88. package/dist/pipeline/template.d.ts.map +1 -0
  89. package/dist/pipeline/template.js +47 -0
  90. package/dist/pipeline/template.js.map +1 -0
  91. package/package.json +19 -17
@@ -11,7 +11,8 @@
11
11
  * @since 2.49.0
12
12
  */
13
13
  import { FileStorageEngineBase } from '@memberjunction/core-entities';
14
- import { Metadata, RunView, LogStatus, LogStatusEx, LogError, LogErrorEx, IsVerboseLoggingEnabled } from '@memberjunction/core';
14
+ import { Metadata, RunView, LogStatus, LogStatusEx, LogError, LogErrorEx, IsVerboseLoggingEnabled, DatabaseProviderBase } from '@memberjunction/core';
15
+ import { AgentRunWatchdog } from './agent-run-watchdog.js';
15
16
  import { AIPromptRunner } from '@memberjunction/ai-prompts';
16
17
  import { BaseAgentType } from './agent-types/base-agent-type.js';
17
18
  import { CopyScalarsAndArrays, JSONValidator, SafeExpressionEvaluator, UUIDsEqual } from '@memberjunction/global';
@@ -26,6 +27,7 @@ import { AgentRunner } from './AgentRunner.js';
26
27
  import { PayloadManager } from './PayloadManager.js';
27
28
  import { ScratchpadManager } from './ScratchpadManager.js';
28
29
  import { ArtifactToolManager } from './ArtifactToolManager.js';
30
+ import { PipelineExecutor, PipelineToolRegistry, ActionInvocable, ArtifactToolInvocable, BuildPipelineToolDocs, formatFinalOutput, summarizePipelineStages, } from './pipeline/index.js';
29
31
  import { AgentDataPreloader } from './AgentDataPreloader.js';
30
32
  import { ClientToolRequestManager } from './ClientToolRequestManager.js';
31
33
  import { ConversationMessageResolver } from './utils/ConversationMessageResolver.js';
@@ -94,6 +96,11 @@ export class BaseAgent {
94
96
  /**
95
97
  * Queue map to chain database saves sequentially per step entity.
96
98
  * Prevents UPDATE queries running before INSERT queries on quick steps.
99
+ *
100
+ * Keyed by the step ENTITY INSTANCE, not its `ID`: a new step's `ID` is empty at create time and
101
+ * only gets populated during its INSERT `Save()`, so keying by `ID` would file the create and the
102
+ * finalize under different buckets and defeat the chain — letting the UPDATE race ahead of the
103
+ * INSERT on millisecond-fast steps (e.g. pipelines), which left them stuck at `Running`.
97
104
  */
98
105
  this._stepSavePromises = new Map();
99
106
  /**
@@ -285,6 +292,22 @@ export class BaseAgent {
285
292
  * @private
286
293
  */
287
294
  static { this.MAX_CONSECUTIVE_FAILED_STEPS = 10; }
295
+ /**
296
+ * Maximum consecutive *unproductive* retry steps before forcing termination.
297
+ *
298
+ * An unproductive retry is a 'Retry' next-step that carries an errorMessage — i.e. one
299
+ * produced by {@link BaseAgentType.createRetryStep} because the model's output could not
300
+ * be parsed or failed structural validation (e.g. the LLM returned conversational prose
301
+ * instead of the required JSON envelope). These do NOT count as 'Failed' steps, so they
302
+ * bypass {@link MAX_CONSECUTIVE_FAILED_STEPS} entirely and — without this guard — loop
303
+ * until the far-higher absolute iteration cap (effectively forever, burning time and tokens).
304
+ *
305
+ * Legitimate yield/await retries (pipeline / client-tools / sub-agent re-entry) are created
306
+ * via createNextStep('Retry', …) WITHOUT an errorMessage, so they do not increment this
307
+ * counter. Any productive (non-unproductive-retry) step resets it.
308
+ * @private
309
+ */
310
+ static { this.MAX_CONSECUTIVE_UNPRODUCTIVE_RETRIES = 10; }
288
311
  /**
289
312
  * Returns the active metadata provider for this agent run. Subclasses MUST
290
313
  * use this getter (rather than `new Metadata()` or `Metadata.Provider`) so
@@ -1209,6 +1232,7 @@ export class BaseAgent {
1209
1232
  let currentNextStep = null;
1210
1233
  let stepCount = 0;
1211
1234
  let consecutiveFailedSteps = 0;
1235
+ let consecutiveUnproductiveRetries = 0;
1212
1236
  while (continueExecution) {
1213
1237
  // Check for cancellation before each step
1214
1238
  if (params.cancellationToken?.aborted) {
@@ -1248,6 +1272,37 @@ export class BaseAgent {
1248
1272
  else if (nextStep.step !== 'Failed') {
1249
1273
  consecutiveFailedSteps = 0;
1250
1274
  }
1275
+ // Track consecutive *unproductive* retries to prevent infinite loops that the
1276
+ // consecutive-failed-steps net above cannot catch. A model that repeatedly returns
1277
+ // output we can't parse/validate (e.g. conversational prose instead of the required
1278
+ // JSON envelope) yields a stream of 'Retry' steps — never 'Failed' — so the failed-step
1279
+ // counter resets every turn and never trips. Such retries are produced via
1280
+ // createRetryStep(), which always sets an errorMessage; legitimate yield/await retries
1281
+ // (pipeline / client-tools / sub-agent re-entry) carry no errorMessage and are exempt.
1282
+ const isUnproductiveRetry = nextStep.step === 'Retry' && !nextStep.terminate && !!nextStep.errorMessage;
1283
+ if (isUnproductiveRetry) {
1284
+ consecutiveUnproductiveRetries++;
1285
+ if (consecutiveUnproductiveRetries >= BaseAgent.MAX_CONSECUTIVE_UNPRODUCTIVE_RETRIES) {
1286
+ this.logError(`⛔ Agent '${params.agent.Name}' reached maximum consecutive unproductive retries ` +
1287
+ `(${BaseAgent.MAX_CONSECUTIVE_UNPRODUCTIVE_RETRIES}). The model is repeatedly returning output ` +
1288
+ `that cannot be parsed or validated. Forcing termination to prevent infinite loop.`, {
1289
+ agent: params.agent,
1290
+ category: 'ExecutionSafetyNet',
1291
+ metadata: {
1292
+ consecutiveUnproductiveRetries,
1293
+ lastError: nextStep.errorMessage
1294
+ }
1295
+ });
1296
+ nextStep.step = 'Failed';
1297
+ nextStep.terminate = true;
1298
+ nextStep.errorMessage = `Agent terminated after ${consecutiveUnproductiveRetries} consecutive unproductive retries ` +
1299
+ `(model repeatedly returned output that could not be parsed or validated). ` +
1300
+ `Last error: ${nextStep.errorMessage || 'Unknown'}`;
1301
+ }
1302
+ }
1303
+ else {
1304
+ consecutiveUnproductiveRetries = 0;
1305
+ }
1251
1306
  // Check if we should continue or terminate
1252
1307
  if (nextStep.terminate) {
1253
1308
  continueExecution = false;
@@ -1813,6 +1868,21 @@ export class BaseAgent {
1813
1868
  else if (this._artifactToolManager.HasArtifacts()) {
1814
1869
  this.logStatus(`[ArtifactTools] Artifacts present but tools disabled by agent config (includeArtifactToolsDocs=false)`, true, params);
1815
1870
  }
1871
+ // Inject pipeline tool docs when pipelines are enabled and at least one source exists.
1872
+ // A pipeline's first step must be a source (Action or artifact tool); with none
1873
+ // available pipelines are impossible, so BuildPipelineToolDocs returns '' and the
1874
+ // template's `{{ _PIPELINE_TOOLS }}` block stays empty.
1875
+ const pipelineDocsEnabled = agentTypePromptParams?.includePipelineDocs !== false;
1876
+ if (pipelineDocsEnabled) {
1877
+ const sourceNames = [
1878
+ ...this.getEffectiveActionsForValidation(params.agent.ID).map((a) => a.Name),
1879
+ ...this._artifactToolManager.GetAvailableToolNames(),
1880
+ ];
1881
+ const pipelineDocs = BuildPipelineToolDocs(sourceNames);
1882
+ if (pipelineDocs) {
1883
+ promptParams.data['_PIPELINE_TOOLS'] = pipelineDocs;
1884
+ }
1885
+ }
1816
1886
  // Pass file artifacts as candidate native file inputs.
1817
1887
  // The AIPromptRunner will check these against the resolved driver's
1818
1888
  // FileCapabilities and attach qualifying files as native content blocks.
@@ -3305,9 +3375,10 @@ The context is now within limits. Please retry your request with the recovered c
3305
3375
  const body = toolResults.map((r, i) => {
3306
3376
  const heading = `### ${i + 1}. ${r.artifactId}.${r.tool}(${JSON.stringify(r.input)})`;
3307
3377
  if (r.result.success) {
3308
- const data = typeof r.result.data === 'string'
3378
+ const raw = typeof r.result.data === 'string'
3309
3379
  ? r.result.data
3310
3380
  : JSON.stringify(r.result.data, null, 2);
3381
+ const data = this.capStandaloneToolResultText(raw);
3311
3382
  return `${heading}\n\`\`\`json\n${data}\n\`\`\``;
3312
3383
  }
3313
3384
  return `${heading}\n**Error:** ${r.result.errorMessage}`;
@@ -3331,6 +3402,154 @@ The context is now within limits. Please retry your request with the recovered c
3331
3402
  };
3332
3403
  params.conversationMessages.push(message);
3333
3404
  }
3405
+ /**
3406
+ * Character budget (~4 chars/token) for a SINGLE standalone artifact-tool result injected into
3407
+ * the conversation. A `get_full` on a large artifact can otherwise dump the whole thing into
3408
+ * context and overflow the model's window — the exact failure pipelines exist to avoid. Override
3409
+ * in a subclass to tune. Pipelines are unaffected: their intermediate results never flow through
3410
+ * here, and the executor already caps a pipeline's final output.
3411
+ *
3412
+ * @protected
3413
+ */
3414
+ get maxStandaloneToolResultChars() {
3415
+ return 100_000; // ~25k tokens
3416
+ }
3417
+ /**
3418
+ * Bound a standalone tool result to {@link maxStandaloneToolResultChars}: return a head slice
3419
+ * plus a redirect that teaches the agent to page (`get_rows`) or reduce (`pipeline`) instead of
3420
+ * reading a whole large artifact. Mirrors how read/search tools cap output at the tool boundary.
3421
+ *
3422
+ * @protected
3423
+ */
3424
+ capStandaloneToolResultText(text) {
3425
+ const budget = this.maxStandaloneToolResultChars;
3426
+ if (text.length <= budget) {
3427
+ return text;
3428
+ }
3429
+ const omitted = text.length - budget;
3430
+ return (text.slice(0, budget) +
3431
+ `\n\n…[truncated ${omitted.toLocaleString()} chars. This artifact is too large to read whole ` +
3432
+ `(~${Math.round(text.length / 4000)}k tokens) — reading it in full overflows the context window. ` +
3433
+ `Instead: page it with get_rows(start, count), or run a pipeline that filters/aggregates it ` +
3434
+ `server-side (where / select / groupBy → only the small final result returns to you).]`);
3435
+ }
3436
+ /**
3437
+ * Builds a per-run {@link PipelineToolRegistry} that unifies the three pipeline-able
3438
+ * substrates behind one namespace: built-in transforms, the agent's effective Actions, and
3439
+ * the run's artifact tools. Transforms register first so their reserved names win; a source
3440
+ * whose name collides with a transform is skipped for pipeline use (still callable normally)
3441
+ * and logged, rather than aborting the whole pipeline.
3442
+ *
3443
+ * @protected
3444
+ */
3445
+ buildPipelineRegistry(params) {
3446
+ const registry = new PipelineToolRegistry();
3447
+ const register = (invocable) => {
3448
+ try {
3449
+ registry.Register(invocable);
3450
+ }
3451
+ catch (e) {
3452
+ this.logStatus(`[Pipeline] Skipped tool "${invocable.toolName}": ${e.message}`, true, params);
3453
+ }
3454
+ };
3455
+ // Operators (where/select/map/…) are pure code-defined verbs, not registry tools — only
3456
+ // capabilities (Actions + artifact tools) live here as pipeline sources/stages.
3457
+ // Actions — each wrapped to run via the existing single-action execution path.
3458
+ this.getEffectiveActionsForValidation(params.agent.ID).forEach((actionEntity) => register(new ActionInvocable(actionEntity.Name, (p) => this.ExecuteSingleAction(params, { name: actionEntity.Name, params: p }, actionEntity, params.contextUser))));
3459
+ // Artifact tools — one invocable per distinct tool name; `artifactId` is supplied as a
3460
+ // call-time param so the same `{ tool, params }` step shape works across all substrates.
3461
+ this._artifactToolManager.GetAvailableToolNames().forEach((toolName) => register(new ArtifactToolInvocable(toolName, async (tool, p) => {
3462
+ const stored = await this._artifactToolManager.ExecuteSingleToolCall({
3463
+ artifactId: String(p.artifactId ?? ''),
3464
+ tool,
3465
+ input: p,
3466
+ });
3467
+ return stored.result;
3468
+ })));
3469
+ return registry;
3470
+ }
3471
+ /**
3472
+ * Runs a tool pipeline as a single `Tool` step in the run tree (sibling of the prompt step that
3473
+ * requested it, matching artifact-tool steps). ALL pipeline observability lives in this step's
3474
+ * `OutputData` — the per-stage breakdown, totals, bytes saved, and the tool chain — so there are
3475
+ * no dedicated pipeline entities and no extra SQL I/O; the run tree alone carries everything a
3476
+ * debug UI needs.
3477
+ *
3478
+ * @protected
3479
+ */
3480
+ async executePipelineAsStep(pipeline, params) {
3481
+ const stepEntity = await this.createStepEntity({
3482
+ stepType: 'Tool',
3483
+ stepName: `Pipeline: ${pipeline.steps.length} step(s)`,
3484
+ contextUser: params.contextUser,
3485
+ inputData: { steps: pipeline.steps },
3486
+ });
3487
+ const registry = this.buildPipelineRegistry(params);
3488
+ // The executor converts stage-level errors into a failed RESULT (it doesn't throw for those),
3489
+ // but an unexpected throw — e.g. a tool returning a non-serializable value (BigInt/circular)
3490
+ // that trips JSON.stringify in the executor's byte-accounting — must NEVER leave this step
3491
+ // stuck on 'Running'. Catch it and materialize a failed result so finalize always runs and the
3492
+ // failure surfaces as a 'Failed' step (answering "do pipeline errors show as errors?": yes).
3493
+ let result;
3494
+ try {
3495
+ result = await new PipelineExecutor(registry).Execute(pipeline.steps);
3496
+ }
3497
+ catch (e) {
3498
+ result = {
3499
+ success: false,
3500
+ finalOutput: null,
3501
+ steps: [],
3502
+ error: `Pipeline crashed: ${e?.message ?? String(e)}`,
3503
+ contextBytesSaved: 0,
3504
+ };
3505
+ }
3506
+ // A pipeline is ONE run-step — not a parent + a child step per stage. It runs server-side in a
3507
+ // single fast pass, so the full per-stage breakdown + totals live in this step's OutputData for
3508
+ // a debug UI to visualize — no separate entities, no extra DB writes.
3509
+ await this.finalizeStepEntity(stepEntity, result.success, result.success ? undefined : result.error, {
3510
+ success: result.success,
3511
+ toolChain: summarizePipelineStages(result.steps),
3512
+ steps: result.steps,
3513
+ contextBytesSaved: result.contextBytesSaved,
3514
+ totalBytesStreamed: result.steps.reduce((sum, s) => sum + s.outputSize, 0),
3515
+ totalDurationMs: result.steps.reduce((sum, s) => sum + s.durationMs, 0),
3516
+ failedStepIndex: result.failedStepIndex,
3517
+ });
3518
+ return result;
3519
+ }
3520
+ /**
3521
+ * Pushes the pipeline's final output (or its failure message) into the conversation for the
3522
+ * LLM's next turn, mirroring the artifact-tool "inject once, then expire" pattern. Only the
3523
+ * final output is surfaced — intermediate step outputs never enter the context window.
3524
+ *
3525
+ * @protected
3526
+ */
3527
+ injectPipelineResultMessage(params, result) {
3528
+ const diagnostic = result.success && result.diagnostic
3529
+ ? `\n⚠ Empty result — ${result.diagnostic}`
3530
+ : '';
3531
+ // Identify which pipeline this result belongs to (stage chain, e.g. `get_rows → where →
3532
+ // select`). Without it, multiple pipeline results across turns are indistinguishable once
3533
+ // compacted — mirrors how artifact-tool results name their tool/artifact.
3534
+ const label = summarizePipelineStages(result.steps);
3535
+ const content = result.success
3536
+ ? `Pipeline result [${label}] (final stage value — intermediate stages stayed out of context, ~${result.contextBytesSaved} bytes saved):\n\`\`\`\n${formatFinalOutput(result.finalOutput)}\n\`\`\`${diagnostic}`
3537
+ : `Pipeline failed [${label}].\n${result.error}`;
3538
+ const message = {
3539
+ role: 'user',
3540
+ content,
3541
+ metadata: {
3542
+ turnAdded: this._promptTurnCount,
3543
+ messageType: 'tool-result',
3544
+ expirationTurns: 3,
3545
+ expirationMode: 'Compact',
3546
+ compactMode: 'First N Chars',
3547
+ compactLength: 500,
3548
+ compactPromptId: '',
3549
+ },
3550
+ };
3551
+ params.conversationMessages.push(message);
3552
+ }
3334
3553
  /**
3335
3554
  * Creates a chat message containing sub-agent execution results.
3336
3555
  *
@@ -3523,7 +3742,8 @@ The context is now within limits. Please retry your request with the recovered c
3523
3742
  { docsFlag: 'includeForEachDocs', responseTypeKey: 'forEach' },
3524
3743
  { docsFlag: 'includeWhileDocs', responseTypeKey: 'while' },
3525
3744
  { docsFlag: 'includeScratchpadDocs', responseTypeKey: 'scratchpad' },
3526
- { docsFlag: 'includeArtifactToolsDocs', responseTypeKey: 'artifactToolCalls' }
3745
+ { docsFlag: 'includeArtifactToolsDocs', responseTypeKey: 'artifactToolCalls' },
3746
+ { docsFlag: 'includePipelineDocs', responseTypeKey: 'pipeline' }
3527
3747
  ];
3528
3748
  for (const { docsFlag, responseTypeKey } of alignmentMappings) {
3529
3749
  // Check if the user explicitly set this response type property
@@ -4468,6 +4688,13 @@ The context is now within limits. Please retry your request with the recovered c
4468
4688
  const errorMessage = JSON.stringify(CopyScalarsAndArrays(this._agentRun.LatestResult));
4469
4689
  throw new Error(`Failed to create agent run record: Details: ${errorMessage}`);
4470
4690
  }
4691
+ // Hand the now-persisted run (it has a stable ID) to the watchdog so a process restart,
4692
+ // crash, or failed terminal-state write can't leave it stuck 'Running' forever. Only the
4693
+ // server-side DB provider can heartbeat via SQL; client/non-DB providers simply opt out.
4694
+ const runProvider = params.provider || this._activeProvider;
4695
+ if (runProvider instanceof DatabaseProviderBase && params.contextUser) {
4696
+ AgentRunWatchdog.Instance.Track(this._agentRun.ID, runProvider, params.contextUser);
4697
+ }
4471
4698
  // Invoke callback if provided
4472
4699
  if (modifiedParams.onAgentRunCreated) {
4473
4700
  try {
@@ -4672,15 +4899,16 @@ The context is now within limits. Please retry your request with the recovered c
4672
4899
  * @protected
4673
4900
  */
4674
4901
  queueStepSave(stepEntity) {
4675
- const id = stepEntity.ID;
4676
- const previousSave = this._stepSavePromises.get(id) ?? Promise.resolve();
4902
+ // Chain on the entity INSTANCE (stable), NOT stepEntity.ID — the ID is empty until the INSERT
4903
+ // Save() assigns it, so an ID-keyed chain breaks for fast create→finalize sequences.
4904
+ const previousSave = this._stepSavePromises.get(stepEntity) ?? Promise.resolve();
4677
4905
  const currentSave = previousSave.then(() => stepEntity.Save()).then((ok) => {
4678
4906
  if (!ok) {
4679
- LogError(`Failed to save agent run step record ${id}: ${stepEntity.LatestResult?.CompleteMessage ?? 'unknown error'}`);
4907
+ LogError(`Failed to save agent run step record ${stepEntity.ID || '(unsaved)'}: ${stepEntity.LatestResult?.CompleteMessage ?? 'unknown error'}`);
4680
4908
  }
4681
4909
  return ok;
4682
4910
  });
4683
- this._stepSavePromises.set(id, currentSave);
4911
+ this._stepSavePromises.set(stepEntity, currentSave);
4684
4912
  this._pendingSaves.push(currentSave);
4685
4913
  }
4686
4914
  /**
@@ -5183,6 +5411,14 @@ The context is now within limits. Please retry your request with the recovered c
5183
5411
  else if (this._artifactToolManager.HasArtifacts()) {
5184
5412
  this.logStatus(`[ArtifactTools] LLM did not use artifact tools this turn (artifacts available but not accessed)`, true, params);
5185
5413
  }
5414
+ // Execute a tool pipeline if provided (zero turn cost — processed inline). Each step's
5415
+ // output is threaded into the next server-side; only the final step's output returns to
5416
+ // the LLM, so intermediate payloads never enter the context window.
5417
+ if (initialNextStep.pipeline?.steps?.length) {
5418
+ this.logStatus(`[Pipeline] LLM requested a ${initialNextStep.pipeline.steps.length}-stage pipeline: ${initialNextStep.pipeline.steps.map(s => s.tool ?? Object.keys(s)[0]).join(' | ')}`, true, params);
5419
+ const pipelineResult = await this.executePipelineAsStep(initialNextStep.pipeline, params);
5420
+ this.injectPipelineResultMessage(params, pipelineResult);
5421
+ }
5186
5422
  // now that we have processed the payload, we can process the next step which does validation and changes the next step if
5187
5423
  // validation fails
5188
5424
  const updatedNextStep = await this.processNextStep(initialNextStep, params, config.agentType, promptResult, finalPayload, stepEntity);
@@ -5440,9 +5676,13 @@ The context is now within limits. Please retry your request with the recovered c
5440
5676
  }
5441
5677
  });
5442
5678
  // Add assistant message indicating we're executing a sub-agent
5679
+ // Recorded as a `user`-role environment annotation (not an `assistant` turn) for the same
5680
+ // reason as the action record above: the model's real output is the JSON envelope, and
5681
+ // storing framework prose as an `assistant` turn trains strong in-context models to imitate
5682
+ // the prose and drift off the required JSON format. See the note at the action-record push.
5443
5683
  params.conversationMessages.push({
5444
- role: 'assistant',
5445
- content: `I'm delegating this task to the "${subAgentRequest.name}" agent.\n\nReason: ${subAgentRequest.message}`
5684
+ role: 'user',
5685
+ content: `[You delegated this task to the "${subAgentRequest.name}" agent. Reason: ${subAgentRequest.message}]`
5446
5686
  });
5447
5687
  // Prepare input data for the step
5448
5688
  const inputData = {
@@ -5842,9 +6082,11 @@ The context is now within limits. Please retry your request with the recovered c
5842
6082
  hierarchicalStep: this.buildHierarchicalStep(stepCount + 1, this._parentStepCounts)
5843
6083
  }
5844
6084
  });
6085
+ // `user`-role environment annotation (not an `assistant` turn) — see the note on the
6086
+ // single-delegation push above for why framework prose must not be stored as assistant turns.
5845
6087
  params.conversationMessages.push({
5846
- role: 'assistant',
5847
- content: `I'm delegating this task to the parallel sub-agent "${request.name}".\n\nReason: ${request.message}`
6088
+ role: 'user',
6089
+ content: `[You delegated this task to the parallel sub-agent "${request.name}". Reason: ${request.message}]`
5848
6090
  });
5849
6091
  return { request: request, subAgentEntity, relationship };
5850
6092
  }
@@ -6138,9 +6380,13 @@ The context is now within limits. Please retry your request with the recovered c
6138
6380
  }
6139
6381
  });
6140
6382
  // Add assistant message indicating we're executing a related sub-agent
6383
+ // Recorded as a `user`-role environment annotation (not an `assistant` turn) for the same
6384
+ // reason as the action record above: the model's real output is the JSON envelope, and
6385
+ // storing framework prose as an `assistant` turn trains strong in-context models to imitate
6386
+ // the prose and drift off the required JSON format. See the note at the action-record push.
6141
6387
  params.conversationMessages.push({
6142
- role: 'assistant',
6143
- content: `I'm delegating this task to the "${subAgentRequest.name}" agent.\n\nReason: ${subAgentRequest.message}`
6388
+ role: 'user',
6389
+ content: `[You delegated this task to the "${subAgentRequest.name}" agent. Reason: ${subAgentRequest.message}]`
6144
6390
  });
6145
6391
  // Prepare input data for the step
6146
6392
  const inputData = {
@@ -6579,12 +6825,23 @@ The context is now within limits. Please retry your request with the recovered c
6579
6825
  },
6580
6826
  displayMode: 'live' // Only show in live mode
6581
6827
  });
6582
- // Build detailed action execution message with parameters using markdown formatting
6583
- // This creates a permanent, lightweight record of what was requested
6828
+ // Build a detailed record of the action(s) invoked, with parameters, in markdown.
6829
+ // This is a permanent, lightweight memory of what was requested.
6830
+ //
6831
+ // IMPORTANT — this record is injected as a `user`-role environment annotation, NOT an
6832
+ // `assistant` turn. The model's actual output is the JSON envelope, but we don't store
6833
+ // that raw JSON; we store this human-readable summary instead. If it were recorded as an
6834
+ // `assistant` turn, then after a few action-heavy turns the model's entire visible
6835
+ // assistant history would be prose like "I'm executing the X action with parameters: …",
6836
+ // and strong in-context learners (e.g. Gemini Flash) imitate that demonstrated pattern
6837
+ // over the system-prompt instruction — drifting into prose and breaking JSON parsing,
6838
+ // which (pre-guardrail) looped forever. Phrasing it in second person under the `user`
6839
+ // role keeps the memory while removing the false assistant-prose exemplar. The
6840
+ // human-facing narration is emitted separately via onProgress above.
6584
6841
  let actionMessage;
6585
6842
  if (actions.length === 1) {
6586
6843
  const aa = actions[0];
6587
- actionMessage = `I'm executing the **${aa.name}** action`;
6844
+ actionMessage = `[You invoked the **${aa.name}** action`;
6588
6845
  // Add parameters if they exist
6589
6846
  if (aa.params && Object.keys(aa.params).length > 0) {
6590
6847
  const paramsList = Object.entries(aa.params)
@@ -6593,14 +6850,14 @@ The context is now within limits. Please retry your request with the recovered c
6593
6850
  return `• **${key}**: ${displayValue}`;
6594
6851
  })
6595
6852
  .join('\n');
6596
- actionMessage += ` with parameters:\n${paramsList}`;
6853
+ actionMessage += ` with parameters:\n${paramsList}\n]`;
6597
6854
  }
6598
6855
  else {
6599
- actionMessage += '.';
6856
+ actionMessage += '.]';
6600
6857
  }
6601
6858
  }
6602
6859
  else {
6603
- actionMessage = `I'm executing **${actions.length} actions** in parallel:\n\n` + actions.map((aa, index) => {
6860
+ actionMessage = `[You invoked **${actions.length} actions** in parallel:\n\n` + actions.map((aa, index) => {
6604
6861
  let actionText = `${index + 1}. **${aa.name}**`;
6605
6862
  // Add parameters if they exist
6606
6863
  if (aa.params && Object.keys(aa.params).length > 0) {
@@ -6613,12 +6870,13 @@ The context is now within limits. Please retry your request with the recovered c
6613
6870
  actionText += `\n${paramsList}`;
6614
6871
  }
6615
6872
  return actionText;
6616
- }).join('\n\n');
6873
+ }).join('\n\n') + '\n]';
6617
6874
  }
6618
6875
  if (addConversationMessage) {
6619
- // Add assistant message (no metadata - this is a permanent record)
6876
+ // Record as a `user`-role environment annotation (no metadata - permanent record).
6877
+ // See the note above on why this is NOT an `assistant` turn.
6620
6878
  params.conversationMessages.push({
6621
- role: 'assistant',
6879
+ role: 'user',
6622
6880
  content: actionMessage
6623
6881
  });
6624
6882
  }
@@ -7901,6 +8159,8 @@ The context is now within limits. Please retry your request with the recovered c
7901
8159
  this._agentRun.TotalTokensUsed = tokenStats.totalTokens;
7902
8160
  this._agentRun.TotalPromptTokensUsed = tokenStats.promptTokens;
7903
8161
  this._agentRun.TotalCompletionTokensUsed = tokenStats.completionTokens;
8162
+ this._agentRun.TotalCacheReadTokensUsed = tokenStats.cacheReadTokens;
8163
+ this._agentRun.TotalCacheWriteTokensUsed = tokenStats.cacheWriteTokens;
7904
8164
  this._agentRun.TotalCost = tokenStats.totalCost;
7905
8165
  await this._agentRun.Save();
7906
8166
  }
@@ -7927,6 +8187,8 @@ The context is now within limits. Please retry your request with the recovered c
7927
8187
  this._agentRun.TotalTokensUsed = tokenStats.totalTokens;
7928
8188
  this._agentRun.TotalPromptTokensUsed = tokenStats.promptTokens;
7929
8189
  this._agentRun.TotalCompletionTokensUsed = tokenStats.completionTokens;
8190
+ this._agentRun.TotalCacheReadTokensUsed = tokenStats.cacheReadTokens;
8191
+ this._agentRun.TotalCacheWriteTokensUsed = tokenStats.cacheWriteTokens;
7930
8192
  this._agentRun.TotalCost = tokenStats.totalCost;
7931
8193
  await this._agentRun.Save();
7932
8194
  }
@@ -8008,6 +8270,8 @@ The context is now within limits. Please retry your request with the recovered c
8008
8270
  this._agentRun.TotalTokensUsed = tokenStats.totalTokens;
8009
8271
  this._agentRun.TotalPromptTokensUsed = tokenStats.promptTokens;
8010
8272
  this._agentRun.TotalCompletionTokensUsed = tokenStats.completionTokens;
8273
+ this._agentRun.TotalCacheReadTokensUsed = tokenStats.cacheReadTokens;
8274
+ this._agentRun.TotalCacheWriteTokensUsed = tokenStats.cacheWriteTokens;
8011
8275
  this._agentRun.TotalCost = tokenStats.totalCost;
8012
8276
  const ok = await this._agentRun.Save();
8013
8277
  if (!ok) {
@@ -8046,15 +8310,19 @@ The context is now within limits. Please retry your request with the recovered c
8046
8310
  let totalTokens = 0;
8047
8311
  let promptTokens = 0;
8048
8312
  let completionTokens = 0;
8313
+ let cacheReadTokens = 0;
8314
+ let cacheWriteTokens = 0;
8049
8315
  let totalCost = 0;
8050
8316
  // Iterate through the agent run's steps to sum up tokens
8051
8317
  if (this._agentRun?.Steps) {
8052
8318
  for (const step of this._agentRun.Steps) {
8053
8319
  if (step.StepType === 'Prompt' && step.PromptRun) {
8054
- // Add tokens from prompt runs
8320
+ // Add tokens from prompt runs (rollup fields include any nested child prompt runs)
8055
8321
  totalTokens += step.PromptRun.TokensUsedRollup || 0;
8056
8322
  promptTokens += step.PromptRun.TokensPromptRollup || 0;
8057
8323
  completionTokens += step.PromptRun.TokensCompletionRollup || 0;
8324
+ cacheReadTokens += step.PromptRun.TokensCacheReadRollup || 0;
8325
+ cacheWriteTokens += step.PromptRun.TokensCacheWriteRollup || 0;
8058
8326
  totalCost += step.PromptRun.TotalCost || 0;
8059
8327
  }
8060
8328
  else if (step.StepType === 'Sub-Agent' && step.SubAgentRun) {
@@ -8062,11 +8330,13 @@ The context is now within limits. Please retry your request with the recovered c
8062
8330
  totalTokens += step.SubAgentRun.TotalTokensUsed || 0;
8063
8331
  promptTokens += step.SubAgentRun.TotalPromptTokensUsed || 0;
8064
8332
  completionTokens += step.SubAgentRun.TotalCompletionTokensUsed || 0;
8333
+ cacheReadTokens += step.SubAgentRun.TotalCacheReadTokensUsed || 0;
8334
+ cacheWriteTokens += step.SubAgentRun.TotalCacheWriteTokensUsed || 0;
8065
8335
  totalCost += step.SubAgentRun.TotalCost || 0;
8066
8336
  }
8067
8337
  }
8068
8338
  }
8069
- return { totalTokens, promptTokens, completionTokens, totalCost };
8339
+ return { totalTokens, promptTokens, completionTokens, cacheReadTokens, cacheWriteTokens, totalCost };
8070
8340
  }
8071
8341
  /**
8072
8342
  * Gets the count of how many times a specific action has been executed in this agent run.