@memberjunction/ai-prompts 5.41.0 → 5.43.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,12 +1,11 @@
1
1
  import { BaseLLM, ChatParams, ChatMessageRole, GetAIAPIKey, ErrorAnalyzer, ResolveFileInputStrategy } from '@memberjunction/ai';
2
2
  import { AIModelRunner } from './AIModelRunner.js';
3
3
  import { AIModelSelectionInfo } from '@memberjunction/ai-core-plus';
4
- import { LogErrorEx, LogStatus, LogStatusEx, IsVerboseLoggingEnabled, Metadata } from '@memberjunction/core';
4
+ import { BaseEntitySaveQueue, LogErrorEx, LogStatus, LogStatusEx, IsVerboseLoggingEnabled, Metadata } from '@memberjunction/core';
5
5
  import { CleanJSON, MJGlobal, JSONValidator, ValidationResult, ValidationErrorInfo, ValidationErrorType, UUIDsEqual, NormalizeUUID } from '@memberjunction/global';
6
6
  import { CredentialEngine } from '@memberjunction/credentials';
7
7
  import { TemplateEngineServer } from '@memberjunction/templates';
8
8
  import { ExecutionPlanner } from './ExecutionPlanner.js';
9
- import { ParallelExecutionCoordinator } from './ParallelExecutionCoordinator.js';
10
9
  import { AIEngine } from '@memberjunction/aiengine';
11
10
  import { AIEngineBase } from '@memberjunction/ai-engine-base';
12
11
  import { SystemPlaceholderManager } from '@memberjunction/ai-core-plus';
@@ -55,22 +54,48 @@ export class AIPromptRunner {
55
54
  constructor() {
56
55
  this._provider = null;
57
56
  /**
58
- * Instance-keyed chain of in-flight AIPromptRun saves. Mirrors the BaseAgent step-save pattern:
59
- * prompt-run persistence is fire-and-forget so the execution path never blocks on a DB
60
- * round-trip, but saves for the SAME entity are sequenced — the initial 'Running' INSERT always
61
- * completes before the finalize UPDATE, so a slow INSERT can never clobber the finalized row.
62
- * Keyed by the entity INSTANCE (stable), not its ID. See {@link queuePromptRunSave}.
57
+ * Fire-and-forget AIPromptRun persistence. Prompt-run logging never blocks the execution path on a
58
+ * DB round-trip; the shared {@link BaseEntitySaveQueue} sequences saves for the SAME entity (the
59
+ * initial 'Running' INSERT always completes before the finalize UPDATE, and the finalize mutation
60
+ * runs INSIDE the post-INSERT task so a slow INSERT can never clobber the finalized row). Failures
61
+ * stay in this runner's structured log stream via the queue's `onError` hook.
63
62
  */
64
- this._promptRunSaveChains = new Map();
65
- /** All queued prompt-run save promises, for optional flushing via {@link WaitForPendingPromptRunSaves}. */
66
- this._pendingPromptRunSaves = [];
63
+ this._promptRunQueue = new BaseEntitySaveQueue({
64
+ onError: (message) => this.logError(message, { category: 'PromptRunSave' }),
65
+ });
67
66
  this._metadata = this._provider ?? new Metadata();
68
67
  this._templateEngine = TemplateEngineServer.Instance;
69
68
  this._executionPlanner = new ExecutionPlanner();
70
- this._parallelCoordinator = new ParallelExecutionCoordinator();
71
69
  this._jsonValidator = new JSONValidator();
72
70
  this._modelRunner = new AIModelRunner();
73
71
  }
72
+ /** ClassFactory key the parallel coordinator self-registers under (see {@link ParallelCoordinator}). */
73
+ static { this.PARALLEL_COORDINATOR_KEY = 'ParallelExecutionCoordinator'; }
74
+ /**
75
+ * Lazily resolves the parallel execution coordinator.
76
+ *
77
+ * The coordinator is a SUBCLASS of AIPromptRunner — it inherits {@link executeModel} and the rest
78
+ * of the battle-tested execution path so there is a single source of truth for credential / driver
79
+ * / ChatParams / streaming resolution. That subclass relationship means the base cannot statically
80
+ * `new` it without a hard circular import, so we resolve it through the ClassFactory instead (the
81
+ * coordinator self-registers via `@RegisterClass(AIPromptRunner, PARALLEL_COORDINATOR_KEY)`).
82
+ * Created once per runner and reused; the runner's Provider override is propagated. Throws a clear
83
+ * error if the coordinator class was never loaded/registered — otherwise ClassFactory would silently
84
+ * fall back to a plain AIPromptRunner that lacks the parallel methods.
85
+ */
86
+ get ParallelCoordinator() {
87
+ if (!this._parallelCoordinator) {
88
+ const instance = MJGlobal.Instance.ClassFactory.CreateInstance(AIPromptRunner, AIPromptRunner.PARALLEL_COORDINATOR_KEY);
89
+ if (!instance || typeof instance.executeTasksInParallel !== 'function') {
90
+ throw new Error(`ParallelExecutionCoordinator is not registered with the ClassFactory. Ensure ` +
91
+ `'@memberjunction/ai-prompts' is fully loaded (it is exported from the package index and ` +
92
+ `picked up by the class-registration manifest).`);
93
+ }
94
+ instance.Provider = this.Provider;
95
+ this._parallelCoordinator = instance;
96
+ }
97
+ return this._parallelCoordinator;
98
+ }
74
99
  /**
75
100
  * Access the underlying AIModelRunner for embedding and other non-LLM model calls.
76
101
  * Use this when you need tracked embedding execution with AIPromptRun record creation.
@@ -457,25 +482,20 @@ export class AIPromptRunner {
457
482
  let renderedPromptText = '';
458
483
  // For hierarchical prompts, we need to create the parent prompt run first to get its ID
459
484
  let parentPromptRun;
460
- let selectedModel;
461
485
  let childTemplateRenderingResult;
462
- let modelSelectionInfo;
486
+ let selection;
463
487
  // Handle different prompt execution modes
464
488
  if (params.childPrompts && params.childPrompts.length > 0) {
465
489
  // Hierarchical template composition mode - render child templates first, then compose
466
- //this.logStatus(`🌳 Composing prompt with ${params.childPrompts.length} child templates in hierarchical mode`, true, params);
467
490
  // Determine which prompt to use for model selection
468
491
  let modelSelectionPrompt = prompt;
469
492
  if (params.modelSelectionPrompt) {
470
493
  modelSelectionPrompt = params.modelSelectionPrompt;
471
- //this.logStatus(`🎯 Using prompt "${modelSelectionPrompt.Name}" for model selection instead of parent prompt`, true, params);
472
494
  }
473
- // Select model using the appropriate prompt
474
- const modelResult = await this.selectModel(modelSelectionPrompt, params.override?.modelId, params.contextUser, params.configurationId, params.override?.vendorId, params);
475
- selectedModel = modelResult.model;
476
- modelSelectionInfo = modelResult.selectionInfo;
477
- if (!selectedModel) {
478
- throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, modelSelectionInfo));
495
+ // Select model using the appropriate prompt — capture the FULL result
496
+ selection = await this.selectModel(modelSelectionPrompt, params.override?.modelId, params.contextUser, params.configurationId, params.override?.vendorId, params);
497
+ if (!selection.model) {
498
+ throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, selection.selectionInfo));
479
499
  }
480
500
  // Check if we have a system prompt override
481
501
  if (params.systemPromptOverride) {
@@ -490,7 +510,7 @@ export class AIPromptRunner {
490
510
  renderedPromptText = await this.renderPromptWithChildTemplates(prompt, params, childTemplateRenderingResult.renderedTemplates);
491
511
  }
492
512
  // Create parent prompt run for the final composed prompt execution
493
- parentPromptRun = await this.createPromptRun(prompt, selectedModel, params, renderedPromptText, startTime, params.override?.vendorId, modelSelectionInfo);
513
+ parentPromptRun = await this.createPromptRun(prompt, selection.model, params, renderedPromptText, startTime, params.override?.vendorId, selection.selectionInfo);
494
514
  }
495
515
  else if (prompt.TemplateID && (!params.conversationMessages || params.templateMessageRole !== 'none')) {
496
516
  // Check if we have a system prompt override
@@ -520,30 +540,28 @@ export class AIPromptRunner {
520
540
  if (params.cancellationToken?.aborted) {
521
541
  throw new Error('Prompt execution was cancelled during template rendering');
522
542
  }
523
- // If no model was selected yet (no template case), select one now
524
- if (!selectedModel) {
543
+ // If no model was selected yet (non-hierarchical case), select one now — capture the FULL result
544
+ if (!selection?.model) {
525
545
  let modelSelectionPrompt = prompt;
526
546
  if (params.modelSelectionPrompt) {
527
547
  modelSelectionPrompt = params.modelSelectionPrompt;
528
548
  this.logStatus(`🎯 Using prompt "${modelSelectionPrompt.Name}" for model selection instead of main prompt`, true, params);
529
549
  }
530
- const modelResult = await this.selectModel(modelSelectionPrompt, params.override?.modelId, params.contextUser, params.configurationId, params.override?.vendorId, params);
531
- selectedModel = modelResult.model;
532
- modelSelectionInfo = modelResult.selectionInfo;
533
- if (!selectedModel) {
534
- throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, modelSelectionInfo));
550
+ selection = await this.selectModel(modelSelectionPrompt, params.override?.modelId, params.contextUser, params.configurationId, params.override?.vendorId, params);
551
+ if (!selection.model) {
552
+ throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, selection.selectionInfo));
535
553
  }
536
554
  }
537
555
  // Check if we need parallel execution based on ParallelizationMode
538
556
  const shouldUseParallelExecution = prompt.ParallelizationMode && prompt.ParallelizationMode !== 'None';
539
557
  let result;
540
558
  if (shouldUseParallelExecution) {
541
- // Use parallel execution path
542
- result = await this.executePromptInParallel(prompt, renderedPromptText, params, startTime, parentPromptRun, selectedModel, modelSelectionInfo);
559
+ // Use parallel execution path — pass full selection through
560
+ result = await this.executePromptInParallel(prompt, renderedPromptText, params, startTime, parentPromptRun, selection);
543
561
  }
544
562
  else {
545
- // Use traditional single execution path
546
- result = await this.executeSinglePrompt(prompt, renderedPromptText, params, startTime, parentPromptRun, selectedModel, modelSelectionInfo);
563
+ // Use traditional single execution path — pass full selection through
564
+ result = await this.executeSinglePrompt(prompt, renderedPromptText, params, startTime, parentPromptRun, selection);
547
565
  }
548
566
  // Note: With template composition, we only execute once so no rollup calculations needed
549
567
  // The final composed prompt is executed as a single operation
@@ -613,33 +631,22 @@ export class AIPromptRunner {
613
631
  * @param startTime - Execution start time
614
632
  * @returns Promise<AIPromptRunResult<T>> - The execution result
615
633
  */
616
- async executeSinglePrompt(prompt, renderedPromptText, params, startTime, existingPromptRun, existingModel, existingModelSelectionInfo) {
634
+ async executeSinglePrompt(prompt, renderedPromptText, params, startTime, existingPromptRun, existingSelection) {
617
635
  // Check for cancellation before model selection
618
636
  if (params.cancellationToken?.aborted) {
619
637
  throw new Error('Prompt execution was cancelled before model selection');
620
638
  }
621
- // Use existing model if provided (hierarchical case) or select one
622
- let selectedModel = existingModel;
623
- let modelSelectionInfo = existingModelSelectionInfo;
624
- let vendorDriverClass;
625
- let vendorApiName;
626
- let vendorSupportsEffortLevel;
627
- let modelEffortLevel;
628
- let allCandidates = [];
629
- if (modelSelectionInfo) {
630
- // we received model selection info, need to lookup vendor driver class and api name from there
631
- const vendorID = modelSelectionInfo.vendorSelected?.ID;
632
- const modelID = modelSelectionInfo.modelSelected.ID;
633
- const modelVendor = AIEngine.Instance.ModelVendorsByModelID.get(NormalizeUUID(modelID))
634
- ?.find(mv => UUIDsEqual(mv.VendorID, vendorID));
635
- if (modelVendor) {
636
- vendorDriverClass = modelVendor.DriverClass;
637
- vendorApiName = modelVendor.APIName;
638
- vendorSupportsEffortLevel = modelVendor.SupportsEffortLevel;
639
- }
640
- // Extract valid candidates from selection info for retry logic
641
- allCandidates = this.buildCandidatesFromSelectionInfo(modelSelectionInfo);
642
- }
639
+ // Use existing selection if provided (hierarchical case) or select now
640
+ let selectedModel = existingSelection?.model ?? undefined;
641
+ let modelSelectionInfo = existingSelection?.selectionInfo;
642
+ let vendorDriverClass = existingSelection?.vendorDriverClass;
643
+ let vendorApiName = existingSelection?.vendorApiName;
644
+ let vendorSupportsEffortLevel = existingSelection?.vendorSupportsEffortLevel;
645
+ let modelEffortLevel = existingSelection?.modelEffortLevel;
646
+ let allCandidates = existingSelection?.allCandidates ?? [];
647
+ // Credential probes already done during selection — reused by failover so it doesn't
648
+ // recompute hasCredentialsAvailable for the prefix it walks before the selected candidate.
649
+ let credentialAvailability = existingSelection?.credentialAvailability;
643
650
  if (!selectedModel) {
644
651
  // Determine which prompt to use for model selection
645
652
  let modelSelectionPrompt = prompt;
@@ -655,6 +662,7 @@ export class AIPromptRunner {
655
662
  modelEffortLevel = modelResult.modelEffortLevel;
656
663
  modelSelectionInfo = modelResult.selectionInfo;
657
664
  allCandidates = modelResult.allCandidates || [];
665
+ credentialAvailability = modelResult.credentialAvailability;
658
666
  if (!selectedModel) {
659
667
  throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, modelSelectionInfo));
660
668
  }
@@ -670,7 +678,8 @@ export class AIPromptRunner {
670
678
  throw new Error('Prompt execution was cancelled before model execution');
671
679
  }
672
680
  // Execute with retry logic for validation failures
673
- const { modelResult, parsedResult, validationAttempts, cumulativeTokens } = await this.executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel // Pass model-specific effort level
681
+ const { modelResult, parsedResult, validationAttempts, cumulativeTokens } = await this.executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, // Pass model-specific effort level
682
+ credentialAvailability // Reuse credential probes from selection
674
683
  );
675
684
  // Calculate execution metrics
676
685
  const endTime = new Date();
@@ -730,7 +739,7 @@ export class AIPromptRunner {
730
739
  * @param startTime - Execution start time
731
740
  * @returns Promise<AIPromptRunResult<T>> - The aggregated execution result
732
741
  */
733
- async executePromptInParallel(prompt, renderedPromptText, params, startTime, existingPromptRun, existingModel, existingModelSelectionInfo) {
742
+ async executePromptInParallel(prompt, renderedPromptText, params, startTime, existingPromptRun, existingSelection) {
734
743
  // Check for cancellation before starting parallel execution
735
744
  if (params.cancellationToken?.aborted) {
736
745
  throw new Error('Parallel execution was cancelled before starting');
@@ -738,21 +747,21 @@ export class AIPromptRunner {
738
747
  // Load AI Engine to get models and prompt models
739
748
  await AIEngine.Instance.Config(false, params.contextUser);
740
749
  let executionTasks;
741
- // If a model is already selected (from hierarchical template composition),
750
+ // If a model is already selected (from hierarchical template composition),
742
751
  // create a single task with that model instead of using the planner
743
- if (existingModel) {
752
+ if (existingSelection?.model) {
744
753
  // Create a single execution task with the pre-selected model
745
754
  executionTasks = [{
746
755
  taskId: 'pre-selected',
747
- model: existingModel,
748
- vendorDriverClass: undefined, // Would need to look up vendor entity for this
749
- vendorApiName: existingModel.Vendor, // Vendor is already the name string
756
+ model: existingSelection.model,
757
+ vendorDriverClass: existingSelection.vendorDriverClass,
758
+ vendorApiName: existingSelection.vendorApiName,
750
759
  messages: params.conversationMessages || [],
751
760
  promptText: renderedPromptText,
752
761
  templateMessageRole: params.templateMessageRole || 'system',
753
762
  contextUser: params.contextUser
754
763
  }];
755
- this.logStatus(` Using pre-selected model "${existingModel.Name}" for parallel execution`, true, params);
764
+ this.logStatus(` Using pre-selected model "${existingSelection.model.Name}" for parallel execution`, true, params);
756
765
  }
757
766
  else {
758
767
  // Normal parallel execution path - let the planner decide
@@ -777,7 +786,7 @@ export class AIPromptRunner {
777
786
  throw new Error('Parallel execution was cancelled before task execution');
778
787
  }
779
788
  // Execute tasks in parallel
780
- const parallelResult = await this._parallelCoordinator.executeTasksInParallel(params, executionTasks, undefined, undefined, params.cancellationToken, undefined, params.agentRunId);
789
+ const parallelResult = await this.ParallelCoordinator.executeTasksInParallel(params, executionTasks, undefined, undefined, params.cancellationToken, undefined, params.agentRunId);
781
790
  if (!parallelResult.success) {
782
791
  throw new Error(`Parallel execution failed: ${parallelResult.errors.join(', ')}`);
783
792
  }
@@ -793,7 +802,7 @@ export class AIPromptRunner {
793
802
  method: 'PromptSelector',
794
803
  selectorPromptId: prompt.ResultSelectorPromptID,
795
804
  };
796
- const aiSelectedResult = await this._parallelCoordinator.selectBestResult(successfulResults, selectionConfig, undefined, params.cancellationToken);
805
+ const aiSelectedResult = await this.ParallelCoordinator.selectBestResult(successfulResults, selectionConfig, undefined, params.cancellationToken);
797
806
  if (aiSelectedResult) {
798
807
  selectedResult = aiSelectedResult;
799
808
  }
@@ -822,7 +831,7 @@ export class AIPromptRunner {
822
831
  }
823
832
  // Use existing prompt run if provided (hierarchical case) or create new one
824
833
  // Use the model selection info if provided (from hierarchical execution)
825
- const consolidatedPromptRun = existingPromptRun || await this.createPromptRun(prompt, selectedResult.task.model, params, renderedPromptText, startTime, params.override?.vendorId, existingModelSelectionInfo);
834
+ const consolidatedPromptRun = existingPromptRun || await this.createPromptRun(prompt, selectedResult.task.model, params, renderedPromptText, startTime, params.override?.vendorId, existingSelection?.selectionInfo);
826
835
  // Update with parallel execution metadata
827
836
  const endTime = new Date();
828
837
  consolidatedPromptRun.CompletedAt = endTime;
@@ -875,8 +884,10 @@ export class AIPromptRunner {
875
884
  // Set Status and WasSelectedResult for parallel execution
876
885
  consolidatedPromptRun.Status = parallelResult.successCount > 0 ? 'Completed' : 'Failed';
877
886
  consolidatedPromptRun.WasSelectedResult = true; // This is the consolidated result chosen by judge
878
- // Persist the consolidated run fire-and-forget; chains after its INSERT via the save queue.
879
- this.queuePromptRunSave(consolidatedPromptRun);
887
+ // Persist the consolidated run fire-and-forget; the finalize UPDATE chains after its INSERT via
888
+ // the save queue. These fields are set after all the awaited parallel work, so the INSERT has long
889
+ // landed — a plain Update (no post-INSERT callback) is race-safe here.
890
+ this._promptRunQueue.Update(consolidatedPromptRun);
880
891
  // Create additional results from all other successful results (excluding the best one)
881
892
  const additionalResults = [];
882
893
  // Sort successful results by ranking (if available) or keep original order
@@ -944,11 +955,11 @@ export class AIPromptRunner {
944
955
  modelInfo: {
945
956
  modelId: selectedResult.task.model.ID,
946
957
  modelName: selectedResult.task.model.Name,
947
- vendorId: existingModelSelectionInfo?.vendorSelected?.ID, // VendorID not directly available on AIModel
958
+ vendorId: existingSelection?.selectionInfo?.vendorSelected?.ID, // VendorID not directly available on AIModel
948
959
  vendorName: selectedResult.task.model.Vendor,
949
960
  },
950
961
  judgeMetadata: selectedResult.judgeMetadata,
951
- modelSelectionInfo: existingModelSelectionInfo, // Include model selection info if provided
962
+ modelSelectionInfo: existingSelection?.selectionInfo, // Include model selection info if provided
952
963
  };
953
964
  }
954
965
  /**
@@ -1275,7 +1286,7 @@ export class AIPromptRunner {
1275
1286
  // });
1276
1287
  // }
1277
1288
  // Select the first candidate with available credentials and track all attempts
1278
- const { selected, consideredModels } = await this.selectModelWithAPIKeyTracked(candidates, prompt.ID, params);
1289
+ const { selected, consideredModels, credentialAvailability } = await this.selectModelWithAPIKeyTracked(candidates, prompt.ID, params);
1279
1290
  // Merge considered models into our tracking
1280
1291
  modelsConsidered.push(...consideredModels);
1281
1292
  if (!selected) {
@@ -1287,6 +1298,7 @@ export class AIPromptRunner {
1287
1298
  vendorSupportsEffortLevel: undefined,
1288
1299
  modelEffortLevel: undefined,
1289
1300
  allCandidates: candidates,
1301
+ credentialAvailability,
1290
1302
  selectionInfo: this.createSelectionInfo({
1291
1303
  aiConfiguration: configuration,
1292
1304
  modelsConsidered,
@@ -1331,6 +1343,7 @@ export class AIPromptRunner {
1331
1343
  vendorSupportsEffortLevel: selected.supportsEffortLevel,
1332
1344
  modelEffortLevel: selected.effortLevel, // Pass through model-specific effort level
1333
1345
  allCandidates: candidates,
1346
+ credentialAvailability,
1334
1347
  selectionInfo: this.createSelectionInfo({
1335
1348
  aiConfiguration: configuration,
1336
1349
  modelsConsidered,
@@ -1879,33 +1892,6 @@ export class AIPromptRunner {
1879
1892
  Object.assign(info, data);
1880
1893
  return info;
1881
1894
  }
1882
- /**
1883
- * Converts model selection info into ModelVendorCandidate array for retry logic.
1884
- * Extracts only the valid candidates (those with available API keys) from the selection info.
1885
- *
1886
- * @param selectionInfo - Model selection information containing considered models
1887
- * @returns Array of valid model-vendor candidates sorted by priority
1888
- */
1889
- buildCandidatesFromSelectionInfo(selectionInfo) {
1890
- const validModels = selectionInfo.extractValidCandidates();
1891
- return validModels.map(considered => {
1892
- // Find matching model vendor for driver and API info
1893
- const modelVendor = considered.vendor
1894
- ? considered.model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, considered.vendor.ID))
1895
- : undefined;
1896
- return {
1897
- model: considered.model,
1898
- vendorId: considered.vendor?.ID,
1899
- vendorName: considered.vendor?.Name,
1900
- driverClass: modelVendor?.DriverClass || considered.model.DriverClass,
1901
- apiName: modelVendor?.APIName || considered.model.APIName,
1902
- supportsEffortLevel: modelVendor?.SupportsEffortLevel ?? considered.model.SupportsEffortLevel ?? false,
1903
- isPreferredVendor: false, // Can't determine from selection info alone
1904
- priority: considered.priority,
1905
- source: (selectionInfo.selectionStrategy === 'ByPower' ? 'power-rank' : 'model-type')
1906
- };
1907
- }).sort((a, b) => b.priority - a.priority); // Sort by priority descending
1908
- }
1909
1895
  /**
1910
1896
  * Enhanced version of selectModelWithAPIKey that tracks all considered models
1911
1897
  * for model selection reporting. Uses the hierarchical credential resolution
@@ -1997,7 +1983,7 @@ export class AIPromptRunner {
1997
1983
  maxErrorLength: params?.maxErrorLength
1998
1984
  });
1999
1985
  }
2000
- return { selected: selectedCandidate, consideredModels };
1986
+ return { selected: selectedCandidate, consideredModels, credentialAvailability: credentialCache };
2001
1987
  }
2002
1988
  /**
2003
1989
  * Builds a descriptive error message when no model could be selected for a prompt.
@@ -2049,51 +2035,12 @@ export class AIPromptRunner {
2049
2035
  };
2050
2036
  }
2051
2037
  /**
2052
- * Queues a fire-and-forget `Save()` for a prompt-run entity. Saves for the same instance are
2053
- * chained (via {@link _promptRunSaveChains}) so the initial INSERT always completes before any
2054
- * finalize UPDATE — guaranteeing a slow INSERT can't overwrite the finalized row. The whole
2055
- * chain runs independently of the execution flow (callers do NOT await it), so the model call
2056
- * is never delayed by a DB write.
2057
- *
2058
- * Save failures are logged (non-fatal): the AIPromptRun record is observability, not part of the
2059
- * prompt's success contract, so a rare persistence failure must not fail the prompt. The chained
2060
- * promise is tracked in {@link _pendingPromptRunSaves} so {@link WaitForPendingPromptRunSaves}
2061
- * can flush them when determinism is required (e.g. tests, or a caller that needs the rows
2062
- * durably written). Returns that promise.
2063
- */
2064
- queuePromptRunSave(promptRun) {
2065
- const previous = this._promptRunSaveChains.get(promptRun) ?? Promise.resolve(true);
2066
- const current = previous
2067
- .then(async () => {
2068
- const ok = await promptRun.Save();
2069
- if (!ok) {
2070
- this.logError(`Failed to save AIPromptRun ${promptRun.ID || '(unsaved)'}: ${promptRun.LatestResult?.CompleteMessage || 'Unknown error'}`, {
2071
- category: 'PromptRunSave',
2072
- metadata: { promptRunId: promptRun.ID }
2073
- });
2074
- }
2075
- return ok;
2076
- })
2077
- .catch((err) => {
2078
- // Infrastructure-level throw (network, etc.) — log and swallow so the fire-and-forget
2079
- // promise never surfaces as an unhandled rejection.
2080
- this.logError(err instanceof Error ? err : new Error(String(err)), {
2081
- category: 'PromptRunSave',
2082
- metadata: { promptRunId: promptRun.ID }
2083
- });
2084
- return false;
2085
- });
2086
- this._promptRunSaveChains.set(promptRun, current);
2087
- this._pendingPromptRunSaves.push(current);
2088
- return current;
2089
- }
2090
- /**
2091
- * Awaits all in-flight prompt-run saves queued by this runner instance. The normal execution
2092
- * path does NOT call this — prompt-run persistence is intentionally fire-and-forget. Exposed for
2093
- * tests and for callers that need the AIPromptRun rows durably written before proceeding.
2038
+ * Awaits all in-flight prompt-run saves queued by this runner instance. The normal execution path
2039
+ * does NOT call this — prompt-run persistence is intentionally fire-and-forget. Exposed for tests
2040
+ * and for callers that need the AIPromptRun rows durably written before proceeding.
2094
2041
  */
2095
2042
  async WaitForPendingPromptRunSaves() {
2096
- await Promise.allSettled(this._pendingPromptRunSaves);
2043
+ await this._promptRunQueue.Flush();
2097
2044
  }
2098
2045
  async createPromptRun(prompt, model, params, systemPromptText, startTime, vendorId, modelSelectionInfo) {
2099
2046
  const provider = params.provider ?? Metadata.Provider;
@@ -2110,7 +2057,7 @@ export class AIPromptRunner {
2110
2057
  promptRun.Status = 'Running';
2111
2058
  promptRun.Cancelled = false;
2112
2059
  promptRun.CacheHit = false;
2113
- promptRun.StreamingEnabled = false;
2060
+ promptRun.StreamingEnabled = !!params.onStreaming;
2114
2061
  promptRun.WasSelectedResult = false;
2115
2062
  // Set model selection tracking fields
2116
2063
  if (modelSelectionInfo) {
@@ -2261,7 +2208,7 @@ export class AIPromptRunner {
2261
2208
  // NewRecord() above, so callers (and the onPromptRunCreated callback) have it immediately —
2262
2209
  // we don't block the model call on the INSERT. The finalize UPDATE chains after this INSERT
2263
2210
  // via the instance-keyed save queue, so ordering is guaranteed.
2264
- this.queuePromptRunSave(promptRun);
2211
+ this._promptRunQueue.Insert(promptRun);
2265
2212
  // Invoke callback if provided. The ID is available without awaiting the save (client-generated
2266
2213
  // by NewRecord()), so agent-run/step linking that depends on it works immediately.
2267
2214
  if (params.onPromptRunCreated) {
@@ -2353,7 +2300,7 @@ export class AIPromptRunner {
2353
2300
  * - updatePromptRunWithFailoverFailure: Records failed failover metadata
2354
2301
  * - createFailoverErrorResult: Creates standardized error response
2355
2302
  */
2356
- async executeModelWithFailover(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, allCandidates, promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel) {
2303
+ async executeModelWithFailover(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, allCandidates, promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability) {
2357
2304
  // Get failover configuration (used for errorScope filtering)
2358
2305
  const failoverConfig = this.getFailoverConfiguration(prompt);
2359
2306
  // If no candidates provided or failover disabled, execute normally with first model
@@ -2363,10 +2310,45 @@ export class AIPromptRunner {
2363
2310
  // Track failover attempts
2364
2311
  const failoverAttempts = [];
2365
2312
  let lastError = null;
2313
+ // Cache credential availability per driver:model:vendor for the duration of this failover
2314
+ // scan so we don't repeat env-var / binding lookups while walking the candidate list.
2315
+ //
2316
+ // PERF: seed it with the probes model SELECTION already performed (same key format). Selection
2317
+ // walks the priority list until it finds the first credentialed candidate, so this map holds
2318
+ // the prefix it rejected (known false) PLUS the selected candidate (known true) — which is
2319
+ // exactly the segment failover re-walks on the happy path. Reusing those results means the
2320
+ // common case (and any caller looping failover) does ZERO redundant hasCredentialsAvailable
2321
+ // calls. The not-evaluated tail is intentionally absent, so failover still lazily probes it
2322
+ // only if a real failure forces it to walk down there.
2323
+ const failoverCredentialCache = credentialAvailability
2324
+ ? new Map(credentialAvailability)
2325
+ : new Map();
2326
+ const candidateHasCredentials = (c) => {
2327
+ const key = `${c.driverClass}:${c.model.ID}:${c.vendorId || 'default'}`;
2328
+ let has = failoverCredentialCache.get(key);
2329
+ if (has === undefined) {
2330
+ has = this.hasCredentialsAvailable(c.driverClass, prompt.ID, c.model.ID, c.vendorId, params);
2331
+ failoverCredentialCache.set(key, has);
2332
+ }
2333
+ return has;
2334
+ };
2335
+ let skippedForCredentials = 0;
2366
2336
  // Iterate through all candidates in priority order with instant failover
2367
2337
  for (let i = 0; i < allCandidates.length; i++) {
2368
2338
  const candidate = allCandidates[i];
2369
2339
  const attemptStartTime = Date.now();
2340
+ // Skip candidates with no credentials configured. `allCandidates` is intentionally the
2341
+ // FULL priority-ordered list (see the DECISION note in selectModelWithAPIKeyTracked),
2342
+ // so it can include vendors that have no API key in this environment. Firing a live
2343
+ // request at one of those produces a misleading "401 invalid API key" — and because an
2344
+ // Authentication error is treated as fatal, it would halt failover before any
2345
+ // credentialed candidate is ever reached. Skipping here makes failover land on the
2346
+ // first candidate that can actually authenticate (mirroring model selection's own
2347
+ // highest-priority-with-credentials rule).
2348
+ if (!candidateHasCredentials(candidate)) {
2349
+ skippedForCredentials++;
2350
+ continue;
2351
+ }
2370
2352
  try {
2371
2353
  // Log the attempt if not the first one
2372
2354
  if (i > 0) {
@@ -2435,6 +2417,11 @@ export class AIPromptRunner {
2435
2417
  if (promptRun && failoverAttempts.length > 0) {
2436
2418
  this.updatePromptRunWithFailoverFailure(promptRun, failoverAttempts);
2437
2419
  }
2420
+ // If every candidate was skipped for missing credentials we never attempted a call and
2421
+ // have no underlying error to report — surface an actionable message instead of null.
2422
+ if (!lastError && failoverAttempts.length === 0 && skippedForCredentials > 0) {
2423
+ lastError = new Error(`No API credentials configured for any of the ${skippedForCredentials} candidate model-vendor combination(s) for prompt "${prompt.Name}".`);
2424
+ }
2438
2425
  return this.createFailoverErrorResult(lastError, failoverAttempts);
2439
2426
  }
2440
2427
  /**
@@ -2570,7 +2557,11 @@ export class AIPromptRunner {
2570
2557
  };
2571
2558
  }
2572
2559
  /**
2573
- * Executes the AI model with the rendered prompt
2560
+ * Executes the AI model with the rendered prompt.
2561
+ *
2562
+ * `protected` so the parallel coordinator subclass reuses this exact code path — credential
2563
+ * resolution, driver selection, ChatParams construction, prefill, media handling, and streaming
2564
+ * all live here ONCE. Do not duplicate this logic elsewhere.
2574
2565
  */
2575
2566
  async executeModel(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel) {
2576
2567
  // define these variables here to ensure they're available in the catch block
@@ -2712,6 +2703,17 @@ export class AIPromptRunner {
2712
2703
  this.stripUnsupportedMediaBlocks(llm, chatParams, model, verbose, params);
2713
2704
  // Apply assistant prefill (native or fallback) based on prompt config and provider support
2714
2705
  this.applyAssistantPrefill(chatParams, prompt, model, vendorId, llm);
2706
+ // Streaming: wire the prompt-level onStreaming callback into the LLM call. This is the SINGLE
2707
+ // place streaming is configured for prompt execution, so the single-model path and the parallel
2708
+ // path (which bridges its per-task callbacks into params.onStreaming) stream through identical
2709
+ // code — no second streaming implementation that can drift.
2710
+ if (params.onStreaming) {
2711
+ const onStreaming = params.onStreaming;
2712
+ chatParams.streaming = true;
2713
+ chatParams.streamingCallbacks = {
2714
+ OnContent: (chunk, isComplete) => onStreaming({ content: chunk, isComplete, modelName: model.Name }),
2715
+ };
2716
+ }
2715
2717
  // Execute the model with cancellation support
2716
2718
  if (cancellationToken) {
2717
2719
  // If cancellation token is provided, wrap the execution to handle cancellation
@@ -3091,7 +3093,7 @@ export class AIPromptRunner {
3091
3093
  /**
3092
3094
  * Executes the model with retry logic for validation failures
3093
3095
  */
3094
- async executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel) {
3096
+ async executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability) {
3095
3097
  const validationAttempts = [];
3096
3098
  const maxRetries = Math.max(0, prompt.MaxRetries || 0);
3097
3099
  let lastError = null;
@@ -3111,7 +3113,8 @@ export class AIPromptRunner {
3111
3113
  }
3112
3114
  // Execute the AI model with failover support
3113
3115
  const modelResult = await this.executeModelWithFailover(selectedModel, renderedPromptText, prompt, params, promptRun.VendorID, params.conversationMessages, params.templateMessageRole || 'system', params.cancellationToken, allCandidates, // Pass the candidates from initial selection
3114
- promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel);
3116
+ promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability // Reuse credential probes from selection
3117
+ );
3115
3118
  // Check for fatal errors - don't attempt validation/retry on these
3116
3119
  // Fatal errors (like ContextLengthExceeded when all models exhausted) cannot be resolved by retrying
3117
3120
  if (!modelResult.success && modelResult.errorInfo?.severity === 'Fatal') {
@@ -3982,6 +3985,18 @@ export class AIPromptRunner {
3982
3985
  * Updates the AIPromptRun entity with execution results
3983
3986
  */
3984
3987
  async updatePromptRun(promptRun, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens) {
3988
+ // Fire-and-forget finalize UPDATE. The field mutations run INSIDE the post-INSERT task (after the
3989
+ // 'Running' INSERT + its finalizeSave reload land), so the reload can never revert them and the
3990
+ // chained UPDATE persists the finalized state — the "stuck at Running" race is structurally
3991
+ // impossible. The execution flow does NOT await the save.
3992
+ this._promptRunQueue.Update(promptRun, () => this.applyFinalizedPromptRunFields(promptRun, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens));
3993
+ }
3994
+ /**
3995
+ * Populates a prompt-run's finalized fields (result, tokens, cost, timing, rollups) from the model
3996
+ * result. Runs INSIDE the post-INSERT save task — see {@link updatePromptRun}. Errors here are
3997
+ * logged (non-fatal): the AIPromptRun is observability, not part of the prompt's success contract.
3998
+ */
3999
+ applyFinalizedPromptRunFields(promptRun, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens) {
3985
4000
  try {
3986
4001
  promptRun.CompletedAt = endTime;
3987
4002
  promptRun.ExecutionTimeMS = executionTimeMS;
@@ -4168,10 +4183,6 @@ export class AIPromptRunner {
4168
4183
  if (promptRun.Cost !== undefined) {
4169
4184
  promptRun.TotalCost = promptRun.Cost;
4170
4185
  }
4171
- // Finalize fire-and-forget. Chains after the initial INSERT (same entity instance) so the
4172
- // 'Running' INSERT can never overwrite this finalized 'Completed'/'Failed' state. The
4173
- // execution flow does NOT await this — see queuePromptRunSave / WaitForPendingPromptRunSaves.
4174
- this.queuePromptRunSave(promptRun);
4175
4186
  }
4176
4187
  catch (error) {
4177
4188
  this.logError(error, {