@memberjunction/ai-prompts 5.40.2 → 5.42.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,7 +6,6 @@ import { CleanJSON, MJGlobal, JSONValidator, ValidationResult, ValidationErrorIn
6
6
  import { CredentialEngine } from '@memberjunction/credentials';
7
7
  import { TemplateEngineServer } from '@memberjunction/templates';
8
8
  import { ExecutionPlanner } from './ExecutionPlanner.js';
9
- import { ParallelExecutionCoordinator } from './ParallelExecutionCoordinator.js';
10
9
  import { AIEngine } from '@memberjunction/aiengine';
11
10
  import { AIEngineBase } from '@memberjunction/ai-engine-base';
12
11
  import { SystemPlaceholderManager } from '@memberjunction/ai-core-plus';
@@ -26,6 +25,21 @@ function mimeFromBlockType(type) {
26
25
  }
27
26
  }
28
27
  export class AIPromptRunner {
28
+ /**
29
+ * Process-wide cache of parsed `OutputExample` JSON, keyed by the raw example string.
30
+ * A prompt's OutputExample is a static string reused across every run and every validation
31
+ * retry, so re-parsing it each time is pure waste. Keyed by content (not prompt ID) so two
32
+ * prompts sharing an identical example share one parsed entry and an edited example never
33
+ * serves a stale parse. Stores `{ parsed }` on success or `{ error }` on failure so we cache
34
+ * the failure too rather than re-throwing-and-reparsing bad JSON every attempt.
35
+ */
36
+ static { this._outputExampleCache = new Map(); }
37
+ /**
38
+ * Marker used in `AIModelSelectionInfo.modelsConsidered[].unavailableReason` for candidates
39
+ * that were intentionally NOT credential-checked because a higher-priority candidate had
40
+ * already been selected. See the DECISION note in {@link selectModelWithAPIKeyTracked}.
41
+ */
42
+ static { this.NOT_EVALUATED_REASON = 'Not evaluated (a higher-priority candidate was already selected; set AIPromptParams.forceFullModelEvaluation to probe all)'; }
29
43
  /**
30
44
  * Optional metadata provider override. Callers should set
31
45
  * `instance.Provider = providerToUse` before invoking run methods
@@ -39,13 +53,49 @@ export class AIPromptRunner {
39
53
  }
40
54
  constructor() {
41
55
  this._provider = null;
56
+ /**
57
+ * Instance-keyed chain of in-flight AIPromptRun saves. Mirrors the BaseAgent step-save pattern:
58
+ * prompt-run persistence is fire-and-forget so the execution path never blocks on a DB
59
+ * round-trip, but saves for the SAME entity are sequenced — the initial 'Running' INSERT always
60
+ * completes before the finalize UPDATE, so a slow INSERT can never clobber the finalized row.
61
+ * Keyed by the entity INSTANCE (stable), not its ID. See {@link queuePromptRunSave}.
62
+ */
63
+ this._promptRunSaveChains = new Map();
64
+ /** All queued prompt-run save promises, for optional flushing via {@link WaitForPendingPromptRunSaves}. */
65
+ this._pendingPromptRunSaves = [];
42
66
  this._metadata = this._provider ?? new Metadata();
43
67
  this._templateEngine = TemplateEngineServer.Instance;
44
68
  this._executionPlanner = new ExecutionPlanner();
45
- this._parallelCoordinator = new ParallelExecutionCoordinator();
46
69
  this._jsonValidator = new JSONValidator();
47
70
  this._modelRunner = new AIModelRunner();
48
71
  }
72
+ /** ClassFactory key the parallel coordinator self-registers under (see {@link ParallelCoordinator}). */
73
+ static { this.PARALLEL_COORDINATOR_KEY = 'ParallelExecutionCoordinator'; }
74
+ /**
75
+ * Lazily resolves the parallel execution coordinator.
76
+ *
77
+ * The coordinator is a SUBCLASS of AIPromptRunner — it inherits {@link executeModel} and the rest
78
+ * of the battle-tested execution path so there is a single source of truth for credential / driver
79
+ * / ChatParams / streaming resolution. That subclass relationship means the base cannot statically
80
+ * `new` it without a hard circular import, so we resolve it through the ClassFactory instead (the
81
+ * coordinator self-registers via `@RegisterClass(AIPromptRunner, PARALLEL_COORDINATOR_KEY)`).
82
+ * Created once per runner and reused; the runner's Provider override is propagated. Throws a clear
83
+ * error if the coordinator class was never loaded/registered — otherwise ClassFactory would silently
84
+ * fall back to a plain AIPromptRunner that lacks the parallel methods.
85
+ */
86
+ get ParallelCoordinator() {
87
+ if (!this._parallelCoordinator) {
88
+ const instance = MJGlobal.Instance.ClassFactory.CreateInstance(AIPromptRunner, AIPromptRunner.PARALLEL_COORDINATOR_KEY);
89
+ if (!instance || typeof instance.executeTasksInParallel !== 'function') {
90
+ throw new Error(`ParallelExecutionCoordinator is not registered with the ClassFactory. Ensure ` +
91
+ `'@memberjunction/ai-prompts' is fully loaded (it is exported from the package index and ` +
92
+ `picked up by the class-registration manifest).`);
93
+ }
94
+ instance.Provider = this.Provider;
95
+ this._parallelCoordinator = instance;
96
+ }
97
+ return this._parallelCoordinator;
98
+ }
49
99
  /**
50
100
  * Access the underlying AIModelRunner for embedding and other non-LLM model calls.
51
101
  * Use this when you need tracked embedding execution with AIPromptRun record creation.
@@ -116,19 +166,15 @@ export class AIPromptRunner {
116
166
  });
117
167
  }
118
168
  /**
119
- * Checks if a model vendor is configured as an inference provider
169
+ * Checks if a model vendor is configured as an inference provider.
170
+ * Delegates to the memoized {@link AIEngine.IsInferenceProvider} helper so the
171
+ * "Inference Provider" vendor-type lookup happens once per engine load rather than on
172
+ * every candidate in every selection pass.
120
173
  * @param modelVendor The model vendor to check
121
174
  * @returns true if the vendor is an inference provider
122
175
  */
123
176
  isInferenceProvider(modelVendor) {
124
- // Find the inference provider type from cached vendor type definitions
125
- const inferenceProviderType = AIEngine.Instance.VendorTypeDefinitions.find(vt => vt.Name === 'Inference Provider');
126
- if (!inferenceProviderType) {
127
- // Fallback to checking if it's not a model developer (should rarely happen)
128
- const modelDeveloperType = AIEngine.Instance.VendorTypeDefinitions.find(vt => vt.Name === 'Model Developer');
129
- return !UUIDsEqual(modelVendor.TypeID, modelDeveloperType?.ID);
130
- }
131
- return UUIDsEqual(modelVendor.TypeID, inferenceProviderType.ID);
177
+ return AIEngine.Instance.IsInferenceProvider(modelVendor);
132
178
  }
133
179
  /**
134
180
  * Resolves credentials for AI model execution using a hierarchical resolution system.
@@ -171,7 +217,8 @@ export class AIPromptRunner {
171
217
  }
172
218
  // Priority 3: ModelVendor bindings - with failover
173
219
  if (modelId && vendorId) {
174
- const modelVendor = AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.ModelID, modelId) && UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
220
+ const modelVendor = AIEngine.Instance.ModelVendorsByModelID.get(NormalizeUUID(modelId))
221
+ ?.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
175
222
  if (modelVendor) {
176
223
  const bindings = AIEngine.Instance.GetCredentialBindingsForTarget('ModelVendor', modelVendor.ID);
177
224
  const result = await this.tryCredentialBindingsWithFailover(bindings, 'AICredentialBinding(ModelVendor)', params, verbose);
@@ -189,7 +236,7 @@ export class AIPromptRunner {
189
236
  // Priority 5: Type-based default credential
190
237
  // If the vendor declares a CredentialTypeID, try to find a default credential of that type
191
238
  if (vendorId) {
192
- const vendor = AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, vendorId));
239
+ const vendor = AIEngine.Instance.VendorsByID.get(NormalizeUUID(vendorId));
193
240
  if (vendor?.CredentialTypeID) {
194
241
  const defaultCredential = this.findDefaultCredentialByType(vendor.CredentialTypeID);
195
242
  if (defaultCredential) {
@@ -348,7 +395,8 @@ export class AIPromptRunner {
348
395
  }
349
396
  // Priority 3: ModelVendor bindings
350
397
  if (modelId && vendorId) {
351
- const modelVendor = AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.ModelID, modelId) && UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
398
+ const modelVendor = AIEngine.Instance.ModelVendorsByModelID.get(NormalizeUUID(modelId))
399
+ ?.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
352
400
  if (modelVendor && AIEngine.Instance.HasCredentialBindings('ModelVendor', modelVendor.ID)) {
353
401
  return true;
354
402
  }
@@ -361,7 +409,7 @@ export class AIPromptRunner {
361
409
  }
362
410
  // Priority 5: Type-based default credential
363
411
  if (vendorId) {
364
- const vendor = AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, vendorId));
412
+ const vendor = AIEngine.Instance.VendorsByID.get(NormalizeUUID(vendorId));
365
413
  if (vendor?.CredentialTypeID) {
366
414
  const defaultCredential = this.findDefaultCredentialByType(vendor.CredentialTypeID);
367
415
  if (defaultCredential) {
@@ -434,25 +482,20 @@ export class AIPromptRunner {
434
482
  let renderedPromptText = '';
435
483
  // For hierarchical prompts, we need to create the parent prompt run first to get its ID
436
484
  let parentPromptRun;
437
- let selectedModel;
438
485
  let childTemplateRenderingResult;
439
- let modelSelectionInfo;
486
+ let selection;
440
487
  // Handle different prompt execution modes
441
488
  if (params.childPrompts && params.childPrompts.length > 0) {
442
489
  // Hierarchical template composition mode - render child templates first, then compose
443
- //this.logStatus(`🌳 Composing prompt with ${params.childPrompts.length} child templates in hierarchical mode`, true, params);
444
490
  // Determine which prompt to use for model selection
445
491
  let modelSelectionPrompt = prompt;
446
492
  if (params.modelSelectionPrompt) {
447
493
  modelSelectionPrompt = params.modelSelectionPrompt;
448
- //this.logStatus(`🎯 Using prompt "${modelSelectionPrompt.Name}" for model selection instead of parent prompt`, true, params);
449
494
  }
450
- // Select model using the appropriate prompt
451
- const modelResult = await this.selectModel(modelSelectionPrompt, params.override?.modelId, params.contextUser, params.configurationId, params.override?.vendorId, params);
452
- selectedModel = modelResult.model;
453
- modelSelectionInfo = modelResult.selectionInfo;
454
- if (!selectedModel) {
455
- throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, modelSelectionInfo));
495
+ // Select model using the appropriate prompt — capture the FULL result
496
+ selection = await this.selectModel(modelSelectionPrompt, params.override?.modelId, params.contextUser, params.configurationId, params.override?.vendorId, params);
497
+ if (!selection.model) {
498
+ throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, selection.selectionInfo));
456
499
  }
457
500
  // Check if we have a system prompt override
458
501
  if (params.systemPromptOverride) {
@@ -467,7 +510,7 @@ export class AIPromptRunner {
467
510
  renderedPromptText = await this.renderPromptWithChildTemplates(prompt, params, childTemplateRenderingResult.renderedTemplates);
468
511
  }
469
512
  // Create parent prompt run for the final composed prompt execution
470
- parentPromptRun = await this.createPromptRun(prompt, selectedModel, params, renderedPromptText, startTime, params.override?.vendorId, modelSelectionInfo);
513
+ parentPromptRun = await this.createPromptRun(prompt, selection.model, params, renderedPromptText, startTime, params.override?.vendorId, selection.selectionInfo);
471
514
  }
472
515
  else if (prompt.TemplateID && (!params.conversationMessages || params.templateMessageRole !== 'none')) {
473
516
  // Check if we have a system prompt override
@@ -497,30 +540,28 @@ export class AIPromptRunner {
497
540
  if (params.cancellationToken?.aborted) {
498
541
  throw new Error('Prompt execution was cancelled during template rendering');
499
542
  }
500
- // If no model was selected yet (no template case), select one now
501
- if (!selectedModel) {
543
+ // If no model was selected yet (non-hierarchical case), select one now — capture the FULL result
544
+ if (!selection?.model) {
502
545
  let modelSelectionPrompt = prompt;
503
546
  if (params.modelSelectionPrompt) {
504
547
  modelSelectionPrompt = params.modelSelectionPrompt;
505
548
  this.logStatus(`🎯 Using prompt "${modelSelectionPrompt.Name}" for model selection instead of main prompt`, true, params);
506
549
  }
507
- const modelResult = await this.selectModel(modelSelectionPrompt, params.override?.modelId, params.contextUser, params.configurationId, params.override?.vendorId, params);
508
- selectedModel = modelResult.model;
509
- modelSelectionInfo = modelResult.selectionInfo;
510
- if (!selectedModel) {
511
- throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, modelSelectionInfo));
550
+ selection = await this.selectModel(modelSelectionPrompt, params.override?.modelId, params.contextUser, params.configurationId, params.override?.vendorId, params);
551
+ if (!selection.model) {
552
+ throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, selection.selectionInfo));
512
553
  }
513
554
  }
514
555
  // Check if we need parallel execution based on ParallelizationMode
515
556
  const shouldUseParallelExecution = prompt.ParallelizationMode && prompt.ParallelizationMode !== 'None';
516
557
  let result;
517
558
  if (shouldUseParallelExecution) {
518
- // Use parallel execution path
519
- result = await this.executePromptInParallel(prompt, renderedPromptText, params, startTime, parentPromptRun, selectedModel, modelSelectionInfo);
559
+ // Use parallel execution path — pass full selection through
560
+ result = await this.executePromptInParallel(prompt, renderedPromptText, params, startTime, parentPromptRun, selection);
520
561
  }
521
562
  else {
522
- // Use traditional single execution path
523
- result = await this.executeSinglePrompt(prompt, renderedPromptText, params, startTime, parentPromptRun, selectedModel, modelSelectionInfo);
563
+ // Use traditional single execution path — pass full selection through
564
+ result = await this.executeSinglePrompt(prompt, renderedPromptText, params, startTime, parentPromptRun, selection);
524
565
  }
525
566
  // Note: With template composition, we only execute once so no rollup calculations needed
526
567
  // The final composed prompt is executed as a single operation
@@ -590,33 +631,22 @@ export class AIPromptRunner {
590
631
  * @param startTime - Execution start time
591
632
  * @returns Promise<AIPromptRunResult<T>> - The execution result
592
633
  */
593
- async executeSinglePrompt(prompt, renderedPromptText, params, startTime, existingPromptRun, existingModel, existingModelSelectionInfo) {
634
+ async executeSinglePrompt(prompt, renderedPromptText, params, startTime, existingPromptRun, existingSelection) {
594
635
  // Check for cancellation before model selection
595
636
  if (params.cancellationToken?.aborted) {
596
637
  throw new Error('Prompt execution was cancelled before model selection');
597
638
  }
598
- // Use existing model if provided (hierarchical case) or select one
599
- let selectedModel = existingModel;
600
- let modelSelectionInfo = existingModelSelectionInfo;
601
- let vendorDriverClass;
602
- let vendorApiName;
603
- let vendorSupportsEffortLevel;
604
- let modelEffortLevel;
605
- let allCandidates = [];
606
- if (modelSelectionInfo) {
607
- // we received model selection info, need to lookup vendor driver class and api name from there
608
- const vendorID = modelSelectionInfo.vendorSelected?.ID;
609
- const modelID = modelSelectionInfo.modelSelected.ID;
610
- const modelVendor = AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, vendorID) &&
611
- UUIDsEqual(mv.ModelID, modelID));
612
- if (modelVendor) {
613
- vendorDriverClass = modelVendor.DriverClass;
614
- vendorApiName = modelVendor.APIName;
615
- vendorSupportsEffortLevel = modelVendor.SupportsEffortLevel;
616
- }
617
- // Extract valid candidates from selection info for retry logic
618
- allCandidates = this.buildCandidatesFromSelectionInfo(modelSelectionInfo);
619
- }
639
+ // Use existing selection if provided (hierarchical case) or select now
640
+ let selectedModel = existingSelection?.model ?? undefined;
641
+ let modelSelectionInfo = existingSelection?.selectionInfo;
642
+ let vendorDriverClass = existingSelection?.vendorDriverClass;
643
+ let vendorApiName = existingSelection?.vendorApiName;
644
+ let vendorSupportsEffortLevel = existingSelection?.vendorSupportsEffortLevel;
645
+ let modelEffortLevel = existingSelection?.modelEffortLevel;
646
+ let allCandidates = existingSelection?.allCandidates ?? [];
647
+ // Credential probes already done during selection — reused by failover so it doesn't
648
+ // recompute hasCredentialsAvailable for the prefix it walks before the selected candidate.
649
+ let credentialAvailability = existingSelection?.credentialAvailability;
620
650
  if (!selectedModel) {
621
651
  // Determine which prompt to use for model selection
622
652
  let modelSelectionPrompt = prompt;
@@ -632,6 +662,7 @@ export class AIPromptRunner {
632
662
  modelEffortLevel = modelResult.modelEffortLevel;
633
663
  modelSelectionInfo = modelResult.selectionInfo;
634
664
  allCandidates = modelResult.allCandidates || [];
665
+ credentialAvailability = modelResult.credentialAvailability;
635
666
  if (!selectedModel) {
636
667
  throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, modelSelectionInfo));
637
668
  }
@@ -647,7 +678,8 @@ export class AIPromptRunner {
647
678
  throw new Error('Prompt execution was cancelled before model execution');
648
679
  }
649
680
  // Execute with retry logic for validation failures
650
- const { modelResult, parsedResult, validationAttempts, cumulativeTokens } = await this.executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel // Pass model-specific effort level
681
+ const { modelResult, parsedResult, validationAttempts, cumulativeTokens } = await this.executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, // Pass model-specific effort level
682
+ credentialAvailability // Reuse credential probes from selection
651
683
  );
652
684
  // Calculate execution metrics
653
685
  const endTime = new Date();
@@ -707,7 +739,7 @@ export class AIPromptRunner {
707
739
  * @param startTime - Execution start time
708
740
  * @returns Promise<AIPromptRunResult<T>> - The aggregated execution result
709
741
  */
710
- async executePromptInParallel(prompt, renderedPromptText, params, startTime, existingPromptRun, existingModel, existingModelSelectionInfo) {
742
+ async executePromptInParallel(prompt, renderedPromptText, params, startTime, existingPromptRun, existingSelection) {
711
743
  // Check for cancellation before starting parallel execution
712
744
  if (params.cancellationToken?.aborted) {
713
745
  throw new Error('Parallel execution was cancelled before starting');
@@ -715,21 +747,21 @@ export class AIPromptRunner {
715
747
  // Load AI Engine to get models and prompt models
716
748
  await AIEngine.Instance.Config(false, params.contextUser);
717
749
  let executionTasks;
718
- // If a model is already selected (from hierarchical template composition),
750
+ // If a model is already selected (from hierarchical template composition),
719
751
  // create a single task with that model instead of using the planner
720
- if (existingModel) {
752
+ if (existingSelection?.model) {
721
753
  // Create a single execution task with the pre-selected model
722
754
  executionTasks = [{
723
755
  taskId: 'pre-selected',
724
- model: existingModel,
725
- vendorDriverClass: undefined, // Would need to look up vendor entity for this
726
- vendorApiName: existingModel.Vendor, // Vendor is already the name string
756
+ model: existingSelection.model,
757
+ vendorDriverClass: existingSelection.vendorDriverClass,
758
+ vendorApiName: existingSelection.vendorApiName,
727
759
  messages: params.conversationMessages || [],
728
760
  promptText: renderedPromptText,
729
761
  templateMessageRole: params.templateMessageRole || 'system',
730
762
  contextUser: params.contextUser
731
763
  }];
732
- this.logStatus(` Using pre-selected model "${existingModel.Name}" for parallel execution`, true, params);
764
+ this.logStatus(` Using pre-selected model "${existingSelection.model.Name}" for parallel execution`, true, params);
733
765
  }
734
766
  else {
735
767
  // Normal parallel execution path - let the planner decide
@@ -754,7 +786,7 @@ export class AIPromptRunner {
754
786
  throw new Error('Parallel execution was cancelled before task execution');
755
787
  }
756
788
  // Execute tasks in parallel
757
- const parallelResult = await this._parallelCoordinator.executeTasksInParallel(params, executionTasks, undefined, undefined, params.cancellationToken, undefined, params.agentRunId);
789
+ const parallelResult = await this.ParallelCoordinator.executeTasksInParallel(params, executionTasks, undefined, undefined, params.cancellationToken, undefined, params.agentRunId);
758
790
  if (!parallelResult.success) {
759
791
  throw new Error(`Parallel execution failed: ${parallelResult.errors.join(', ')}`);
760
792
  }
@@ -770,7 +802,7 @@ export class AIPromptRunner {
770
802
  method: 'PromptSelector',
771
803
  selectorPromptId: prompt.ResultSelectorPromptID,
772
804
  };
773
- const aiSelectedResult = await this._parallelCoordinator.selectBestResult(successfulResults, selectionConfig, undefined, params.cancellationToken);
805
+ const aiSelectedResult = await this.ParallelCoordinator.selectBestResult(successfulResults, selectionConfig, undefined, params.cancellationToken);
774
806
  if (aiSelectedResult) {
775
807
  selectedResult = aiSelectedResult;
776
808
  }
@@ -799,7 +831,7 @@ export class AIPromptRunner {
799
831
  }
800
832
  // Use existing prompt run if provided (hierarchical case) or create new one
801
833
  // Use the model selection info if provided (from hierarchical execution)
802
- const consolidatedPromptRun = existingPromptRun || await this.createPromptRun(prompt, selectedResult.task.model, params, renderedPromptText, startTime, params.override?.vendorId, existingModelSelectionInfo);
834
+ const consolidatedPromptRun = existingPromptRun || await this.createPromptRun(prompt, selectedResult.task.model, params, renderedPromptText, startTime, params.override?.vendorId, existingSelection?.selectionInfo);
803
835
  // Update with parallel execution metadata
804
836
  const endTime = new Date();
805
837
  consolidatedPromptRun.CompletedAt = endTime;
@@ -852,18 +884,8 @@ export class AIPromptRunner {
852
884
  // Set Status and WasSelectedResult for parallel execution
853
885
  consolidatedPromptRun.Status = parallelResult.successCount > 0 ? 'Completed' : 'Failed';
854
886
  consolidatedPromptRun.WasSelectedResult = true; // This is the consolidated result chosen by judge
855
- const saveResult = await consolidatedPromptRun.Save();
856
- if (!saveResult) {
857
- this.logError(`Failed to save consolidated AIPromptRun: ${consolidatedPromptRun.LatestResult?.CompleteMessage || 'Unknown error'}`, {
858
- category: 'ConsolidatedPromptRunSave',
859
- metadata: {
860
- promptRunId: consolidatedPromptRun.ID,
861
- executionTasks: executionTasks.length,
862
- successfulResults: successfulResults.length
863
- },
864
- maxErrorLength: params.maxErrorLength
865
- });
866
- }
887
+ // Persist the consolidated run fire-and-forget; chains after its INSERT via the save queue.
888
+ this.queuePromptRunSave(consolidatedPromptRun);
867
889
  // Create additional results from all other successful results (excluding the best one)
868
890
  const additionalResults = [];
869
891
  // Sort successful results by ranking (if available) or keep original order
@@ -931,11 +953,11 @@ export class AIPromptRunner {
931
953
  modelInfo: {
932
954
  modelId: selectedResult.task.model.ID,
933
955
  modelName: selectedResult.task.model.Name,
934
- vendorId: existingModelSelectionInfo?.vendorSelected?.ID, // VendorID not directly available on AIModel
956
+ vendorId: existingSelection?.selectionInfo?.vendorSelected?.ID, // VendorID not directly available on AIModel
935
957
  vendorName: selectedResult.task.model.Vendor,
936
958
  },
937
959
  judgeMetadata: selectedResult.judgeMetadata,
938
- modelSelectionInfo: existingModelSelectionInfo, // Include model selection info if provided
960
+ modelSelectionInfo: existingSelection?.selectionInfo, // Include model selection info if provided
939
961
  };
940
962
  }
941
963
  /**
@@ -1225,7 +1247,7 @@ export class AIPromptRunner {
1225
1247
  }
1226
1248
  // Get configuration info if provided
1227
1249
  if (configurationId) {
1228
- configuration = AIEngine.Instance.Configurations.find(c => UUIDsEqual(c.ID, configurationId));
1250
+ configuration = AIEngine.Instance.ConfigurationsByID.get(NormalizeUUID(configurationId));
1229
1251
  configurationName = configuration?.Name;
1230
1252
  }
1231
1253
  // Build unified list of model-vendor candidates
@@ -1262,7 +1284,7 @@ export class AIPromptRunner {
1262
1284
  // });
1263
1285
  // }
1264
1286
  // Select the first candidate with available credentials and track all attempts
1265
- const { selected, consideredModels } = await this.selectModelWithAPIKeyTracked(candidates, prompt.ID, params);
1287
+ const { selected, consideredModels, credentialAvailability } = await this.selectModelWithAPIKeyTracked(candidates, prompt.ID, params);
1266
1288
  // Merge considered models into our tracking
1267
1289
  modelsConsidered.push(...consideredModels);
1268
1290
  if (!selected) {
@@ -1274,6 +1296,7 @@ export class AIPromptRunner {
1274
1296
  vendorSupportsEffortLevel: undefined,
1275
1297
  modelEffortLevel: undefined,
1276
1298
  allCandidates: candidates,
1299
+ credentialAvailability,
1277
1300
  selectionInfo: this.createSelectionInfo({
1278
1301
  aiConfiguration: configuration,
1279
1302
  modelsConsidered,
@@ -1309,7 +1332,7 @@ export class AIPromptRunner {
1309
1332
  // Get selected vendor entity
1310
1333
  let selectedVendor;
1311
1334
  if (selected.vendorId) {
1312
- selectedVendor = AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, selected.vendorId));
1335
+ selectedVendor = AIEngine.Instance.VendorsByID.get(NormalizeUUID(selected.vendorId));
1313
1336
  }
1314
1337
  return {
1315
1338
  model: selected.model,
@@ -1318,6 +1341,7 @@ export class AIPromptRunner {
1318
1341
  vendorSupportsEffortLevel: selected.supportsEffortLevel,
1319
1342
  modelEffortLevel: selected.effortLevel, // Pass through model-specific effort level
1320
1343
  allCandidates: candidates,
1344
+ credentialAvailability,
1321
1345
  selectionInfo: this.createSelectionInfo({
1322
1346
  aiConfiguration: configuration,
1323
1347
  modelsConsidered,
@@ -1385,7 +1409,7 @@ export class AIPromptRunner {
1385
1409
  * Returns candidates for the single model if it's active and compatible.
1386
1410
  */
1387
1411
  buildCandidatesForExplicitModel(explicitModelId, prompt, preferredVendorId) {
1388
- const model = AIEngine.Instance.Models.find(m => UUIDsEqual(m.ID, explicitModelId));
1412
+ const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(explicitModelId));
1389
1413
  if (!model || !model.IsActive) {
1390
1414
  return [];
1391
1415
  }
@@ -1480,7 +1504,7 @@ export class AIPromptRunner {
1480
1504
  return 0;
1481
1505
  const modelsWithPower = promptModels
1482
1506
  .map(pm => {
1483
- const model = AIEngine.Instance.Models.find(m => UUIDsEqual(m.ID, pm.ModelID));
1507
+ const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1484
1508
  return { powerRank: model?.PowerRank ?? 0, priority: pm.Priority || 1 };
1485
1509
  });
1486
1510
  const totalWeight = modelsWithPower.reduce((sum, m) => sum + m.priority, 0);
@@ -1510,7 +1534,7 @@ export class AIPromptRunner {
1510
1534
  */
1511
1535
  buildCandidatesForGeneralSelection(prompt, configurationId, preferredVendorId, verbose) {
1512
1536
  const preferredVendorName = preferredVendorId ?
1513
- AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, preferredVendorId))?.Name : undefined;
1537
+ AIEngine.Instance.VendorsByID.get(NormalizeUUID(preferredVendorId))?.Name : undefined;
1514
1538
  // Get prompt models for configuration
1515
1539
  const promptModels = this.getPromptModelsForConfiguration(prompt, configurationId);
1516
1540
  const candidates = [];
@@ -1582,7 +1606,7 @@ export class AIPromptRunner {
1582
1606
  const pm = promptModels[i];
1583
1607
  // Compute priority as inverse of array position so highest-priority (first) gets the largest number
1584
1608
  const computedPriority = promptModels.length - i;
1585
- const model = AIEngine.Instance.Models.find(m => UUIDsEqual(m.ID, pm.ModelID));
1609
+ const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1586
1610
  if (!model || !model.IsActive)
1587
1611
  continue;
1588
1612
  if (pm.VendorID) {
@@ -1604,8 +1628,9 @@ export class AIPromptRunner {
1604
1628
  * Helper: Create candidate for specific vendor from AIPromptModel.
1605
1629
  */
1606
1630
  createCandidateForSpecificVendor(model, promptModel, computedPriority = 0) {
1607
- const modelVendor = AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.ModelID, promptModel.ModelID) &&
1608
- UUIDsEqual(mv.VendorID, promptModel.VendorID) &&
1631
+ // Use the model's precomputed ModelVendors (grouped at engine load) instead of scanning
1632
+ // the global ModelVendors array — model.ID === promptModel.ModelID here.
1633
+ const modelVendor = model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, promptModel.VendorID) &&
1609
1634
  mv.Status === 'Active' &&
1610
1635
  this.isInferenceProvider(mv));
1611
1636
  if (!modelVendor)
@@ -1627,9 +1652,8 @@ export class AIPromptRunner {
1627
1652
  * Helper: Create candidates for all vendors of a model, sorted by vendor priority.
1628
1653
  */
1629
1654
  createCandidatesForAllVendors(model, computedPriority = 0) {
1630
- const vendors = AIEngine.Instance.ModelVendors
1631
- .filter(mv => UUIDsEqual(mv.ModelID, model.ID) &&
1632
- mv.Status === 'Active' &&
1655
+ const vendors = model.ModelVendors
1656
+ .filter(mv => mv.Status === 'Active' &&
1633
1657
  this.isInferenceProvider(mv))
1634
1658
  .sort((a, b) => (b.Priority || 0) - (a.Priority || 0));
1635
1659
  const candidates = [];
@@ -1691,7 +1715,7 @@ export class AIPromptRunner {
1691
1715
  */
1692
1716
  addPromptSpecificCandidates(candidates, promptModels, preferredVendorId) {
1693
1717
  for (const pm of promptModels) {
1694
- const model = AIEngine.Instance.Models.find(m => UUIDsEqual(m.ID, pm.ModelID));
1718
+ const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1695
1719
  if (model && model.IsActive) {
1696
1720
  const modelCandidates = this.createCandidatesForModel(model, 5000, 'prompt-model', preferredVendorId, pm.Priority);
1697
1721
  candidates.push(...modelCandidates);
@@ -1714,7 +1738,7 @@ export class AIPromptRunner {
1714
1738
  LogStatus(`Adding ${parentModels.length} models from parent config "${parentConfig.Name}" as fallback`);
1715
1739
  }
1716
1740
  for (const pm of parentModels) {
1717
- const model = AIEngine.Instance.Models.find(m => UUIDsEqual(m.ID, pm.ModelID));
1741
+ const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1718
1742
  if (model && model.IsActive) {
1719
1743
  // Decrease base priority for each level up the chain (3000, 2500, 2000, etc.)
1720
1744
  const basePriority = 3000 - (i * 500);
@@ -1731,7 +1755,7 @@ export class AIPromptRunner {
1731
1755
  LogStatus(`Adding ${nullConfigModels.length} NULL configuration models as universal fallback`);
1732
1756
  }
1733
1757
  for (const pm of nullConfigModels) {
1734
- const model = AIEngine.Instance.Models.find(m => UUIDsEqual(m.ID, pm.ModelID));
1758
+ const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1735
1759
  if (model && model.IsActive) {
1736
1760
  const modelCandidates = this.createCandidatesForModel(model, 1000, // Lowest priority tier
1737
1761
  'prompt-model', preferredVendorId, pm.Priority);
@@ -1759,8 +1783,7 @@ export class AIPromptRunner {
1759
1783
  return AIEngine.Instance.Models.filter(m => m.IsActive &&
1760
1784
  (!prompt.AIModelTypeID || UUIDsEqual(m.AIModelTypeID, prompt.AIModelTypeID)) &&
1761
1785
  (!preferredVendorName ||
1762
- AIEngine.Instance.ModelVendors.some(mv => UUIDsEqual(mv.ModelID, m.ID) &&
1763
- mv.Status === 'Active' &&
1786
+ m.ModelVendors.some(mv => mv.Status === 'Active' &&
1764
1787
  mv.Vendor === preferredVendorName &&
1765
1788
  this.isInferenceProvider(mv))));
1766
1789
  }
@@ -1801,9 +1824,11 @@ export class AIPromptRunner {
1801
1824
  */
1802
1825
  createCandidatesForModel(model, basePriority, source, preferredVendorId, promptModelPriority) {
1803
1826
  const modelCandidates = [];
1804
- // Get all vendors for this model - filter for inference providers only
1805
- const modelVendors = AIEngine.Instance.ModelVendors
1806
- .filter(mv => UUIDsEqual(mv.ModelID, model.ID) && mv.Status === 'Active' && this.isInferenceProvider(mv))
1827
+ // Get all vendors for this model - filter for inference providers only.
1828
+ // Uses the model's precomputed ModelVendors (grouped at engine load) rather than scanning
1829
+ // the global ModelVendors array.
1830
+ const modelVendors = model.ModelVendors
1831
+ .filter(mv => mv.Status === 'Active' && this.isInferenceProvider(mv))
1807
1832
  .sort((a, b) => b.Priority - a.Priority);
1808
1833
  // First, add preferred vendor if it exists
1809
1834
  if (preferredVendorId) {
@@ -1865,34 +1890,6 @@ export class AIPromptRunner {
1865
1890
  Object.assign(info, data);
1866
1891
  return info;
1867
1892
  }
1868
- /**
1869
- * Converts model selection info into ModelVendorCandidate array for retry logic.
1870
- * Extracts only the valid candidates (those with available API keys) from the selection info.
1871
- *
1872
- * @param selectionInfo - Model selection information containing considered models
1873
- * @returns Array of valid model-vendor candidates sorted by priority
1874
- */
1875
- buildCandidatesFromSelectionInfo(selectionInfo) {
1876
- const validModels = selectionInfo.extractValidCandidates();
1877
- return validModels.map(considered => {
1878
- // Find matching model vendor for driver and API info
1879
- const modelVendor = considered.vendor
1880
- ? AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.ModelID, considered.model.ID) &&
1881
- UUIDsEqual(mv.VendorID, considered.vendor.ID))
1882
- : undefined;
1883
- return {
1884
- model: considered.model,
1885
- vendorId: considered.vendor?.ID,
1886
- vendorName: considered.vendor?.Name,
1887
- driverClass: modelVendor?.DriverClass || considered.model.DriverClass,
1888
- apiName: modelVendor?.APIName || considered.model.APIName,
1889
- supportsEffortLevel: modelVendor?.SupportsEffortLevel ?? considered.model.SupportsEffortLevel ?? false,
1890
- isPreferredVendor: false, // Can't determine from selection info alone
1891
- priority: considered.priority,
1892
- source: (selectionInfo.selectionStrategy === 'ByPower' ? 'power-rank' : 'model-type')
1893
- };
1894
- }).sort((a, b) => b.priority - a.priority); // Sort by priority descending
1895
- }
1896
1893
  /**
1897
1894
  * Enhanced version of selectModelWithAPIKey that tracks all considered models
1898
1895
  * for model selection reporting. Uses the hierarchical credential resolution
@@ -1908,8 +1905,33 @@ export class AIPromptRunner {
1908
1905
  // Key format: "driverClass:modelId:vendorId" to properly cache credential hierarchy
1909
1906
  const credentialCache = new Map();
1910
1907
  const consideredModels = [];
1911
- // Check ALL candidates to build complete list of valid and invalid options
1908
+ // DECISION (performance): candidates are ordered by priority, and we only need the
1909
+ // highest-priority candidate that has working credentials. So once we find that first
1910
+ // hit, we STOP credential-probing the remaining candidates and record them as
1911
+ // "not-evaluated" rather than running a `hasCredentialsAvailable` check (which does
1912
+ // env-var lookups + binding scans) for every configured model on every prompt run.
1913
+ // The remaining candidates are still kept in `consideredModels` (and in the returned
1914
+ // `allCandidates` from selectModel, which is the FULL ordered list) so failover and the
1915
+ // ordering are unaffected — only the per-candidate availability *telemetry* for the tail
1916
+ // is skipped. Callers that need a complete availability report (e.g. an admin diagnostic)
1917
+ // can set `AIPromptParams.forceFullModelEvaluation = true` to probe every candidate.
1918
+ const forceFullEval = params?.forceFullModelEvaluation === true;
1919
+ let selected;
1912
1920
  for (const candidate of candidates) {
1921
+ const vendorEntity = candidate.vendorId
1922
+ ? AIEngine.Instance.VendorsByID.get(NormalizeUUID(candidate.vendorId))
1923
+ : undefined;
1924
+ // Short-circuit: a usable candidate is already selected and full evaluation wasn't requested.
1925
+ if (selected && !forceFullEval) {
1926
+ consideredModels.push({
1927
+ model: candidate.model,
1928
+ vendor: vendorEntity,
1929
+ priority: candidate.priority,
1930
+ available: false,
1931
+ unavailableReason: AIPromptRunner.NOT_EVALUATED_REASON
1932
+ });
1933
+ continue;
1934
+ }
1913
1935
  // Build cache key including model and vendor for proper credential resolution
1914
1936
  const cacheKey = `${candidate.driverClass}:${candidate.model.ID}:${candidate.vendorId || 'default'}`;
1915
1937
  // Check cache first
@@ -1922,22 +1944,20 @@ export class AIPromptRunner {
1922
1944
  hasCredentials = this.hasCredentialsAvailable(candidate.driverClass, promptId, candidate.model.ID, candidate.vendorId, params);
1923
1945
  credentialCache.set(cacheKey, hasCredentials);
1924
1946
  }
1925
- // Get vendor entity from AIEngine cache if vendorId is available
1926
- let vendorEntity;
1927
- if (candidate.vendorId) {
1928
- vendorEntity = AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, candidate.vendorId));
1929
- }
1930
1947
  // Track this model as considered with availability status
1931
- consideredModels.push({
1948
+ const considered = {
1932
1949
  model: candidate.model,
1933
1950
  vendor: vendorEntity,
1934
1951
  priority: candidate.priority,
1935
1952
  available: hasCredentials,
1936
1953
  unavailableReason: hasCredentials ? undefined : `No credentials configured for driver ${candidate.driverClass}`
1937
- });
1954
+ };
1955
+ consideredModels.push(considered);
1956
+ // Record the first available candidate as the selection (highest priority with credentials)
1957
+ if (hasCredentials && !selected) {
1958
+ selected = considered;
1959
+ }
1938
1960
  }
1939
- // Select the first available candidate (highest priority with API key)
1940
- const selected = consideredModels.find(m => m.available);
1941
1961
  const selectedCandidate = selected ? candidates.find(c => UUIDsEqual(c.model.ID, selected.model.ID) &&
1942
1962
  UUIDsEqual(c.vendorId, selected.vendor?.ID)) : null;
1943
1963
  if (selectedCandidate) {
@@ -1961,7 +1981,7 @@ export class AIPromptRunner {
1961
1981
  maxErrorLength: params?.maxErrorLength
1962
1982
  });
1963
1983
  }
1964
- return { selected: selectedCandidate, consideredModels };
1984
+ return { selected: selectedCandidate, consideredModels, credentialAvailability: credentialCache };
1965
1985
  }
1966
1986
  /**
1967
1987
  * Builds a descriptive error message when no model could be selected for a prompt.
@@ -1990,6 +2010,75 @@ export class AIPromptRunner {
1990
2010
  /**
1991
2011
  * Creates an AIPromptRun entity for execution tracking
1992
2012
  */
2013
+ /**
2014
+ * Resolves the scalar inference parameters for a run: each value is the per-request override
2015
+ * from `additionalParameters` when supplied, otherwise the prompt's configured default. This
2016
+ * is the single source of truth for parameter precedence so {@link executeModel} (ChatParams)
2017
+ * and {@link createPromptRun} (the persisted record) stay in lockstep. Stop sequences and
2018
+ * assistant prefill are intentionally excluded — their representations differ per target.
2019
+ */
2020
+ resolveScalarInferenceParams(prompt, additionalParameters) {
2021
+ const pick = (override, promptDefault) => override !== undefined ? override : (promptDefault != null ? promptDefault : undefined);
2022
+ const ap = additionalParameters;
2023
+ return {
2024
+ temperature: pick(ap?.temperature, prompt.Temperature),
2025
+ topP: pick(ap?.topP, prompt.TopP),
2026
+ topK: pick(ap?.topK, prompt.TopK),
2027
+ minP: pick(ap?.minP, prompt.MinP),
2028
+ frequencyPenalty: pick(ap?.frequencyPenalty, prompt.FrequencyPenalty),
2029
+ presencePenalty: pick(ap?.presencePenalty, prompt.PresencePenalty),
2030
+ seed: pick(ap?.seed, prompt.Seed),
2031
+ includeLogProbs: pick(ap?.includeLogProbs, prompt.IncludeLogProbs),
2032
+ topLogProbs: pick(ap?.topLogProbs, prompt.TopLogProbs),
2033
+ };
2034
+ }
2035
+ /**
2036
+ * Queues a fire-and-forget `Save()` for a prompt-run entity. Saves for the same instance are
2037
+ * chained (via {@link _promptRunSaveChains}) so the initial INSERT always completes before any
2038
+ * finalize UPDATE — guaranteeing a slow INSERT can't overwrite the finalized row. The whole
2039
+ * chain runs independently of the execution flow (callers do NOT await it), so the model call
2040
+ * is never delayed by a DB write.
2041
+ *
2042
+ * Save failures are logged (non-fatal): the AIPromptRun record is observability, not part of the
2043
+ * prompt's success contract, so a rare persistence failure must not fail the prompt. The chained
2044
+ * promise is tracked in {@link _pendingPromptRunSaves} so {@link WaitForPendingPromptRunSaves}
2045
+ * can flush them when determinism is required (e.g. tests, or a caller that needs the rows
2046
+ * durably written). Returns that promise.
2047
+ */
2048
+ queuePromptRunSave(promptRun) {
2049
+ const previous = this._promptRunSaveChains.get(promptRun) ?? Promise.resolve(true);
2050
+ const current = previous
2051
+ .then(async () => {
2052
+ const ok = await promptRun.Save();
2053
+ if (!ok) {
2054
+ this.logError(`Failed to save AIPromptRun ${promptRun.ID || '(unsaved)'}: ${promptRun.LatestResult?.CompleteMessage || 'Unknown error'}`, {
2055
+ category: 'PromptRunSave',
2056
+ metadata: { promptRunId: promptRun.ID }
2057
+ });
2058
+ }
2059
+ return ok;
2060
+ })
2061
+ .catch((err) => {
2062
+ // Infrastructure-level throw (network, etc.) — log and swallow so the fire-and-forget
2063
+ // promise never surfaces as an unhandled rejection.
2064
+ this.logError(err instanceof Error ? err : new Error(String(err)), {
2065
+ category: 'PromptRunSave',
2066
+ metadata: { promptRunId: promptRun.ID }
2067
+ });
2068
+ return false;
2069
+ });
2070
+ this._promptRunSaveChains.set(promptRun, current);
2071
+ this._pendingPromptRunSaves.push(current);
2072
+ return current;
2073
+ }
2074
+ /**
2075
+ * Awaits all in-flight prompt-run saves queued by this runner instance. The normal execution
2076
+ * path does NOT call this — prompt-run persistence is intentionally fire-and-forget. Exposed for
2077
+ * tests and for callers that need the AIPromptRun rows durably written before proceeding.
2078
+ */
2079
+ async WaitForPendingPromptRunSaves() {
2080
+ await Promise.allSettled(this._pendingPromptRunSaves);
2081
+ }
1993
2082
  async createPromptRun(prompt, model, params, systemPromptText, startTime, vendorId, modelSelectionInfo) {
1994
2083
  const provider = params.provider ?? Metadata.Provider;
1995
2084
  const promptRun = await provider.GetEntityObject('MJ: AI Prompt Runs', params.contextUser);
@@ -2005,7 +2094,7 @@ export class AIPromptRunner {
2005
2094
  promptRun.Status = 'Running';
2006
2095
  promptRun.Cancelled = false;
2007
2096
  promptRun.CacheHit = false;
2008
- promptRun.StreamingEnabled = false;
2097
+ promptRun.StreamingEnabled = !!params.onStreaming;
2009
2098
  promptRun.WasSelectedResult = false;
2010
2099
  // Set model selection tracking fields
2011
2100
  if (modelSelectionInfo) {
@@ -2058,8 +2147,8 @@ export class AIPromptRunner {
2058
2147
  }
2059
2148
  else {
2060
2149
  // Fallback: grab the highest priority AI Model Vendor record for this model (inference providers only)
2061
- const modelVendors = AIEngine.Instance.ModelVendors
2062
- .filter((mv) => UUIDsEqual(mv.ModelID, model.ID) && mv.Status === 'Active' && this.isInferenceProvider(mv))
2150
+ const modelVendors = model.ModelVendors
2151
+ .filter((mv) => mv.Status === 'Active' && this.isInferenceProvider(mv))
2063
2152
  .sort((a, b) => b.Priority - a.Priority);
2064
2153
  if (modelVendors.length > 0) {
2065
2154
  promptRun.VendorID = modelVendors[0].VendorID;
@@ -2091,62 +2180,35 @@ export class AIPromptRunner {
2091
2180
  if (prompt.ResponseFormat && prompt.ResponseFormat !== 'Any') {
2092
2181
  promptRun.ResponseFormat = prompt.ResponseFormat;
2093
2182
  }
2094
- // Save the actual values that will be used (either from prompt defaults or additionalParameters)
2095
- // First, apply defaults from prompt entity
2096
- if (prompt.Temperature != null)
2097
- promptRun.Temperature = prompt.Temperature;
2098
- if (prompt.TopP != null)
2099
- promptRun.TopP = prompt.TopP;
2100
- if (prompt.TopK != null)
2101
- promptRun.TopK = prompt.TopK;
2102
- if (prompt.MinP != null)
2103
- promptRun.MinP = prompt.MinP;
2104
- if (prompt.FrequencyPenalty != null)
2105
- promptRun.FrequencyPenalty = prompt.FrequencyPenalty;
2106
- if (prompt.PresencePenalty != null)
2107
- promptRun.PresencePenalty = prompt.PresencePenalty;
2108
- if (prompt.Seed != null)
2109
- promptRun.Seed = prompt.Seed;
2183
+ // Save the actual values that will be used (prompt defaults overridden by additionalParameters).
2184
+ // Uses the shared resolver so the persisted record matches what executeModel sends to the model.
2185
+ const resolvedParams = this.resolveScalarInferenceParams(prompt, params.additionalParameters);
2186
+ if (resolvedParams.temperature !== undefined)
2187
+ promptRun.Temperature = resolvedParams.temperature;
2188
+ if (resolvedParams.topP !== undefined)
2189
+ promptRun.TopP = resolvedParams.topP;
2190
+ if (resolvedParams.topK !== undefined)
2191
+ promptRun.TopK = resolvedParams.topK;
2192
+ if (resolvedParams.minP !== undefined)
2193
+ promptRun.MinP = resolvedParams.minP;
2194
+ if (resolvedParams.frequencyPenalty !== undefined)
2195
+ promptRun.FrequencyPenalty = resolvedParams.frequencyPenalty;
2196
+ if (resolvedParams.presencePenalty !== undefined)
2197
+ promptRun.PresencePenalty = resolvedParams.presencePenalty;
2198
+ if (resolvedParams.seed !== undefined)
2199
+ promptRun.Seed = resolvedParams.seed;
2200
+ if (resolvedParams.includeLogProbs !== undefined)
2201
+ promptRun.LogProbs = resolvedParams.includeLogProbs;
2202
+ if (resolvedParams.topLogProbs !== undefined)
2203
+ promptRun.TopLogProbs = resolvedParams.topLogProbs;
2204
+ // Stop sequences + assistant prefill: stored from the prompt, with the additionalParameters
2205
+ // array (JSON-encoded) taking precedence when supplied.
2110
2206
  if (prompt.StopSequences)
2111
2207
  promptRun.StopSequences = prompt.StopSequences;
2112
2208
  if (prompt.AssistantPrefill)
2113
2209
  promptRun.AssistantPrefill = prompt.AssistantPrefill;
2114
- if (prompt.IncludeLogProbs != null)
2115
- promptRun.LogProbs = prompt.IncludeLogProbs;
2116
- if (prompt.TopLogProbs != null)
2117
- promptRun.TopLogProbs = prompt.TopLogProbs;
2118
- // Then override with additionalParameters if provided
2119
- if (params.additionalParameters) {
2120
- if (params.additionalParameters.temperature !== undefined) {
2121
- promptRun.Temperature = params.additionalParameters.temperature;
2122
- }
2123
- if (params.additionalParameters.topP !== undefined) {
2124
- promptRun.TopP = params.additionalParameters.topP;
2125
- }
2126
- if (params.additionalParameters.topK !== undefined) {
2127
- promptRun.TopK = params.additionalParameters.topK;
2128
- }
2129
- if (params.additionalParameters.minP !== undefined) {
2130
- promptRun.MinP = params.additionalParameters.minP;
2131
- }
2132
- if (params.additionalParameters.frequencyPenalty !== undefined) {
2133
- promptRun.FrequencyPenalty = params.additionalParameters.frequencyPenalty;
2134
- }
2135
- if (params.additionalParameters.presencePenalty !== undefined) {
2136
- promptRun.PresencePenalty = params.additionalParameters.presencePenalty;
2137
- }
2138
- if (params.additionalParameters.seed !== undefined) {
2139
- promptRun.Seed = params.additionalParameters.seed;
2140
- }
2141
- if (params.additionalParameters.stopSequences !== undefined && params.additionalParameters.stopSequences.length > 0) {
2142
- promptRun.StopSequences = JSON.stringify(params.additionalParameters.stopSequences);
2143
- }
2144
- if (params.additionalParameters.includeLogProbs !== undefined) {
2145
- promptRun.LogProbs = params.additionalParameters.includeLogProbs;
2146
- }
2147
- if (params.additionalParameters.topLogProbs !== undefined) {
2148
- promptRun.TopLogProbs = params.additionalParameters.topLogProbs;
2149
- }
2210
+ if (params.additionalParameters?.stopSequences !== undefined && params.additionalParameters.stopSequences.length > 0) {
2211
+ promptRun.StopSequences = JSON.stringify(params.additionalParameters.stopSequences);
2150
2212
  }
2151
2213
  // Store the input data/context as JSON in Messages field
2152
2214
  if (params.data || params.templateData || systemPromptText) {
@@ -2179,21 +2241,13 @@ export class AIPromptRunner {
2179
2241
  promptRun.ValidationAttemptCount = 0; // Will be updated during execution
2180
2242
  promptRun.SuccessfulValidationCount = 0;
2181
2243
  promptRun.FinalValidationPassed = false; // Will be updated after execution
2182
- const saveResult = await promptRun.Save();
2183
- if (!saveResult) {
2184
- const error = `Failed to save AIPromptRun: ${promptRun.LatestResult?.CompleteMessage || 'Unknown error'}`;
2185
- this.logError(error, {
2186
- category: 'PromptRunCreation',
2187
- metadata: {
2188
- promptId: prompt.ID,
2189
- modelId: model.ID,
2190
- vendorId
2191
- },
2192
- maxErrorLength: params.maxErrorLength
2193
- });
2194
- throw new Error(error);
2195
- }
2196
- // Invoke callback if provided
2244
+ // Persist the initial 'Running' record fire-and-forget. The ID was already assigned by
2245
+ // NewRecord() above, so callers (and the onPromptRunCreated callback) have it immediately —
2246
+ // we don't block the model call on the INSERT. The finalize UPDATE chains after this INSERT
2247
+ // via the instance-keyed save queue, so ordering is guaranteed.
2248
+ this.queuePromptRunSave(promptRun);
2249
+ // Invoke callback if provided. The ID is available without awaiting the save (client-generated
2250
+ // by NewRecord()), so agent-run/step linking that depends on it works immediately.
2197
2251
  if (params.onPromptRunCreated) {
2198
2252
  try {
2199
2253
  await params.onPromptRunCreated(promptRun.ID);
@@ -2283,7 +2337,7 @@ export class AIPromptRunner {
2283
2337
  * - updatePromptRunWithFailoverFailure: Records failed failover metadata
2284
2338
  * - createFailoverErrorResult: Creates standardized error response
2285
2339
  */
2286
- async executeModelWithFailover(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, allCandidates, promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel) {
2340
+ async executeModelWithFailover(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, allCandidates, promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability) {
2287
2341
  // Get failover configuration (used for errorScope filtering)
2288
2342
  const failoverConfig = this.getFailoverConfiguration(prompt);
2289
2343
  // If no candidates provided or failover disabled, execute normally with first model
@@ -2293,10 +2347,45 @@ export class AIPromptRunner {
2293
2347
  // Track failover attempts
2294
2348
  const failoverAttempts = [];
2295
2349
  let lastError = null;
2350
+ // Cache credential availability per driver:model:vendor for the duration of this failover
2351
+ // scan so we don't repeat env-var / binding lookups while walking the candidate list.
2352
+ //
2353
+ // PERF: seed it with the probes model SELECTION already performed (same key format). Selection
2354
+ // walks the priority list until it finds the first credentialed candidate, so this map holds
2355
+ // the prefix it rejected (known false) PLUS the selected candidate (known true) — which is
2356
+ // exactly the segment failover re-walks on the happy path. Reusing those results means the
2357
+ // common case (and any caller looping failover) does ZERO redundant hasCredentialsAvailable
2358
+ // calls. The not-evaluated tail is intentionally absent, so failover still lazily probes it
2359
+ // only if a real failure forces it to walk down there.
2360
+ const failoverCredentialCache = credentialAvailability
2361
+ ? new Map(credentialAvailability)
2362
+ : new Map();
2363
+ const candidateHasCredentials = (c) => {
2364
+ const key = `${c.driverClass}:${c.model.ID}:${c.vendorId || 'default'}`;
2365
+ let has = failoverCredentialCache.get(key);
2366
+ if (has === undefined) {
2367
+ has = this.hasCredentialsAvailable(c.driverClass, prompt.ID, c.model.ID, c.vendorId, params);
2368
+ failoverCredentialCache.set(key, has);
2369
+ }
2370
+ return has;
2371
+ };
2372
+ let skippedForCredentials = 0;
2296
2373
  // Iterate through all candidates in priority order with instant failover
2297
2374
  for (let i = 0; i < allCandidates.length; i++) {
2298
2375
  const candidate = allCandidates[i];
2299
2376
  const attemptStartTime = Date.now();
2377
+ // Skip candidates with no credentials configured. `allCandidates` is intentionally the
2378
+ // FULL priority-ordered list (see the DECISION note in selectModelWithAPIKeyTracked),
2379
+ // so it can include vendors that have no API key in this environment. Firing a live
2380
+ // request at one of those produces a misleading "401 invalid API key" — and because an
2381
+ // Authentication error is treated as fatal, it would halt failover before any
2382
+ // credentialed candidate is ever reached. Skipping here makes failover land on the
2383
+ // first candidate that can actually authenticate (mirroring model selection's own
2384
+ // highest-priority-with-credentials rule).
2385
+ if (!candidateHasCredentials(candidate)) {
2386
+ skippedForCredentials++;
2387
+ continue;
2388
+ }
2300
2389
  try {
2301
2390
  // Log the attempt if not the first one
2302
2391
  if (i > 0) {
@@ -2365,6 +2454,11 @@ export class AIPromptRunner {
2365
2454
  if (promptRun && failoverAttempts.length > 0) {
2366
2455
  this.updatePromptRunWithFailoverFailure(promptRun, failoverAttempts);
2367
2456
  }
2457
+ // If every candidate was skipped for missing credentials we never attempted a call and
2458
+ // have no underlying error to report — surface an actionable message instead of null.
2459
+ if (!lastError && failoverAttempts.length === 0 && skippedForCredentials > 0) {
2460
+ lastError = new Error(`No API credentials configured for any of the ${skippedForCredentials} candidate model-vendor combination(s) for prompt "${prompt.Name}".`);
2461
+ }
2368
2462
  return this.createFailoverErrorResult(lastError, failoverAttempts);
2369
2463
  }
2370
2464
  /**
@@ -2500,7 +2594,11 @@ export class AIPromptRunner {
2500
2594
  };
2501
2595
  }
2502
2596
  /**
2503
- * Executes the AI model with the rendered prompt
2597
+ * Executes the AI model with the rendered prompt.
2598
+ *
2599
+ * `protected` so the parallel coordinator subclass reuses this exact code path — credential
2600
+ * resolution, driver selection, ChatParams construction, prefill, media handling, and streaming
2601
+ * all live here ONCE. Do not duplicate this logic elsewhere.
2504
2602
  */
2505
2603
  async executeModel(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel) {
2506
2604
  // define these variables here to ensure they're available in the catch block
@@ -2530,7 +2628,7 @@ export class AIPromptRunner {
2530
2628
  supportsEffortLevel = model.SupportsEffortLevel ?? false;
2531
2629
  if (vendorId) {
2532
2630
  // Find the AIModelVendor record for this specific vendor - must be an inference provider
2533
- const modelVendor = AIEngine.Instance.ModelVendors.find((mv) => UUIDsEqual(mv.ModelID, model.ID) && UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active' && this.isInferenceProvider(mv));
2631
+ const modelVendor = model.ModelVendors.find((mv) => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active' && this.isInferenceProvider(mv));
2534
2632
  if (modelVendor) {
2535
2633
  driverClass = modelVendor.DriverClass || driverClass;
2536
2634
  apiName = modelVendor.APIName || apiName;
@@ -2554,63 +2652,34 @@ export class AIPromptRunner {
2554
2652
  }
2555
2653
  chatParams.model = apiName;
2556
2654
  chatParams.cancellationToken = cancellationToken;
2557
- // Apply defaults from prompt entity first (if they exist)
2558
- // These can be overridden by additionalParameters
2559
- if (prompt.Temperature != null)
2560
- chatParams.temperature = prompt.Temperature;
2561
- if (prompt.TopP != null)
2562
- chatParams.topP = prompt.TopP;
2563
- if (prompt.TopK != null)
2564
- chatParams.topK = prompt.TopK;
2565
- if (prompt.MinP != null)
2566
- chatParams.minP = prompt.MinP;
2567
- if (prompt.FrequencyPenalty != null)
2568
- chatParams.frequencyPenalty = prompt.FrequencyPenalty;
2569
- if (prompt.PresencePenalty != null)
2570
- chatParams.presencePenalty = prompt.PresencePenalty;
2571
- if (prompt.Seed != null)
2572
- chatParams.seed = prompt.Seed;
2655
+ // Apply scalar inference params (prompt defaults overridden by additionalParameters) via the
2656
+ // shared resolver so ChatParams and the persisted AIPromptRun never drift.
2657
+ const resolvedParams = this.resolveScalarInferenceParams(prompt, params.additionalParameters);
2658
+ if (resolvedParams.temperature !== undefined)
2659
+ chatParams.temperature = resolvedParams.temperature;
2660
+ if (resolvedParams.topP !== undefined)
2661
+ chatParams.topP = resolvedParams.topP;
2662
+ if (resolvedParams.topK !== undefined)
2663
+ chatParams.topK = resolvedParams.topK;
2664
+ if (resolvedParams.minP !== undefined)
2665
+ chatParams.minP = resolvedParams.minP;
2666
+ if (resolvedParams.frequencyPenalty !== undefined)
2667
+ chatParams.frequencyPenalty = resolvedParams.frequencyPenalty;
2668
+ if (resolvedParams.presencePenalty !== undefined)
2669
+ chatParams.presencePenalty = resolvedParams.presencePenalty;
2670
+ if (resolvedParams.seed !== undefined)
2671
+ chatParams.seed = resolvedParams.seed;
2672
+ if (resolvedParams.includeLogProbs !== undefined)
2673
+ chatParams.includeLogProbs = resolvedParams.includeLogProbs;
2674
+ if (resolvedParams.topLogProbs !== undefined)
2675
+ chatParams.topLogProbs = resolvedParams.topLogProbs;
2676
+ // Stop sequences are handled separately: the prompt value is comma-delimited and gated by
2677
+ // driver support; additionalParameters supplies a ready-made array that overrides it.
2573
2678
  if (prompt.StopSequences && this.shouldApplyStopSequences(prompt, model, vendorId, llm)) {
2574
- // Parse comma-delimited stop sequences
2575
2679
  chatParams.stopSequences = prompt.StopSequences.split(',').map((s) => s.replace(AIPromptRunner.STOP_SEQUENCE_TRIM_REGEX, '')).filter((s) => s.length > 0);
2576
2680
  }
2577
- if (prompt.IncludeLogProbs != null)
2578
- chatParams.includeLogProbs = prompt.IncludeLogProbs;
2579
- if (prompt.TopLogProbs != null)
2580
- chatParams.topLogProbs = prompt.TopLogProbs;
2581
- // Apply additional parameters if provided (these override prompt defaults)
2582
- if (params.additionalParameters) {
2583
- // Apply chat-specific parameters from additionalParameters
2584
- if (params.additionalParameters.temperature !== undefined) {
2585
- chatParams.temperature = params.additionalParameters.temperature;
2586
- }
2587
- if (params.additionalParameters.topP !== undefined) {
2588
- chatParams.topP = params.additionalParameters.topP;
2589
- }
2590
- if (params.additionalParameters.topK !== undefined) {
2591
- chatParams.topK = params.additionalParameters.topK;
2592
- }
2593
- if (params.additionalParameters.minP !== undefined) {
2594
- chatParams.minP = params.additionalParameters.minP;
2595
- }
2596
- if (params.additionalParameters.frequencyPenalty !== undefined) {
2597
- chatParams.frequencyPenalty = params.additionalParameters.frequencyPenalty;
2598
- }
2599
- if (params.additionalParameters.presencePenalty !== undefined) {
2600
- chatParams.presencePenalty = params.additionalParameters.presencePenalty;
2601
- }
2602
- if (params.additionalParameters.seed !== undefined) {
2603
- chatParams.seed = params.additionalParameters.seed;
2604
- }
2605
- if (params.additionalParameters.stopSequences !== undefined) {
2606
- chatParams.stopSequences = params.additionalParameters.stopSequences;
2607
- }
2608
- if (params.additionalParameters.includeLogProbs !== undefined) {
2609
- chatParams.includeLogProbs = params.additionalParameters.includeLogProbs;
2610
- }
2611
- if (params.additionalParameters.topLogProbs !== undefined) {
2612
- chatParams.topLogProbs = params.additionalParameters.topLogProbs;
2613
- }
2681
+ if (params.additionalParameters?.stopSequences !== undefined) {
2682
+ chatParams.stopSequences = params.additionalParameters.stopSequences;
2614
2683
  }
2615
2684
  // Apply effortLevel with precedence hierarchy
2616
2685
  // 1. params.effortLevel (runtime override - highest priority)
@@ -2671,6 +2740,17 @@ export class AIPromptRunner {
2671
2740
  this.stripUnsupportedMediaBlocks(llm, chatParams, model, verbose, params);
2672
2741
  // Apply assistant prefill (native or fallback) based on prompt config and provider support
2673
2742
  this.applyAssistantPrefill(chatParams, prompt, model, vendorId, llm);
2743
+ // Streaming: wire the prompt-level onStreaming callback into the LLM call. This is the SINGLE
2744
+ // place streaming is configured for prompt execution, so the single-model path and the parallel
2745
+ // path (which bridges its per-task callbacks into params.onStreaming) stream through identical
2746
+ // code — no second streaming implementation that can drift.
2747
+ if (params.onStreaming) {
2748
+ const onStreaming = params.onStreaming;
2749
+ chatParams.streaming = true;
2750
+ chatParams.streamingCallbacks = {
2751
+ OnContent: (chunk, isComplete) => onStreaming({ content: chunk, isComplete, modelName: model.Name }),
2752
+ };
2753
+ }
2674
2754
  // Execute the model with cancellation support
2675
2755
  if (cancellationToken) {
2676
2756
  // If cancellation token is provided, wrap the execution to handle cancellation
@@ -2819,8 +2899,13 @@ export class AIPromptRunner {
2819
2899
  const lower = mimeType.toLowerCase();
2820
2900
  return caps.SupportedMimeTypes.some((pattern) => {
2821
2901
  const p = pattern.toLowerCase();
2902
+ // Wildcard on EITHER side must match (the requested mime is often a modality
2903
+ // probe like 'image/*' — e.g. an image_url block with no explicit mimeType —
2904
+ // and must match a driver that declares any concrete 'image/<x>' type).
2822
2905
  if (p.endsWith('/*'))
2823
2906
  return lower.startsWith(p.slice(0, -1));
2907
+ if (lower.endsWith('/*'))
2908
+ return p.startsWith(lower.slice(0, -1));
2824
2909
  return lower === p;
2825
2910
  });
2826
2911
  }
@@ -2976,7 +3061,7 @@ export class AIPromptRunner {
2976
3061
  }
2977
3062
  // Vendor-level override (null = inherit)
2978
3063
  if (vendorId) {
2979
- const modelVendor = AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.ModelID, model.ID) && UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
3064
+ const modelVendor = model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
2980
3065
  if (modelVendor?.SupportsPrefill != null) {
2981
3066
  supportsPrefill = modelVendor.SupportsPrefill;
2982
3067
  }
@@ -2990,7 +3075,7 @@ export class AIPromptRunner {
2990
3075
  */
2991
3076
  resolvePrefillFallbackText(model, vendorId) {
2992
3077
  // Start with model type default
2993
- const modelType = AIEngine.Instance.ModelTypes.find(mt => UUIDsEqual(mt.ID, model.AIModelTypeID));
3078
+ const modelType = AIEngine.Instance.ModelTypesByID.get(NormalizeUUID(model.AIModelTypeID));
2994
3079
  let fallbackText = modelType?.PrefillFallbackText ?? null;
2995
3080
  // Model-level override
2996
3081
  if (model.PrefillFallbackText != null) {
@@ -2998,7 +3083,7 @@ export class AIPromptRunner {
2998
3083
  }
2999
3084
  // Vendor-level override
3000
3085
  if (vendorId) {
3001
- const modelVendor = AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.ModelID, model.ID) && UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
3086
+ const modelVendor = model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
3002
3087
  if (modelVendor?.PrefillFallbackText != null) {
3003
3088
  fallbackText = modelVendor.PrefillFallbackText;
3004
3089
  }
@@ -3045,7 +3130,7 @@ export class AIPromptRunner {
3045
3130
  /**
3046
3131
  * Executes the model with retry logic for validation failures
3047
3132
  */
3048
- async executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel) {
3133
+ async executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability) {
3049
3134
  const validationAttempts = [];
3050
3135
  const maxRetries = Math.max(0, prompt.MaxRetries || 0);
3051
3136
  let lastError = null;
@@ -3065,7 +3150,8 @@ export class AIPromptRunner {
3065
3150
  }
3066
3151
  // Execute the AI model with failover support
3067
3152
  const modelResult = await this.executeModelWithFailover(selectedModel, renderedPromptText, prompt, params, promptRun.VendorID, params.conversationMessages, params.templateMessageRole || 'system', params.cancellationToken, allCandidates, // Pass the candidates from initial selection
3068
- promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel);
3153
+ promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability // Reuse credential probes from selection
3154
+ );
3069
3155
  // Check for fatal errors - don't attempt validation/retry on these
3070
3156
  // Fatal errors (like ContextLengthExceeded when all models exhausted) cannot be resolved by retrying
3071
3157
  if (!modelResult.success && modelResult.errorInfo?.severity === 'Fatal') {
@@ -3230,7 +3316,7 @@ export class AIPromptRunner {
3230
3316
  const filteredCandidates = allCandidates.filter(c => (c.vendorId || 'default') !== failedVendorId);
3231
3317
  const removedCount = beforeCount - filteredCandidates.length;
3232
3318
  if (removedCount > 0) {
3233
- const vendorName = AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, failedVendorId))?.Name || failedVendorId;
3319
+ const vendorName = AIEngine.Instance.VendorsByID.get(NormalizeUUID(failedVendorId))?.Name || failedVendorId;
3234
3320
  const remainingCount = filteredCandidates.length;
3235
3321
  // Log appropriate message based on error type
3236
3322
  let reason;
@@ -3271,7 +3357,7 @@ export class AIPromptRunner {
3271
3357
  if (shouldRetry) {
3272
3358
  const modelName = currentModel.Name;
3273
3359
  const vendorName = currentVendorId
3274
- ? AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, currentVendorId))?.Name || 'default'
3360
+ ? AIEngine.Instance.VendorsByID.get(NormalizeUUID(currentVendorId))?.Name || 'default'
3275
3361
  : 'default';
3276
3362
  this.logStatus(` ⏳ Rate limit hit - retrying ${modelName} (${vendorName}) with backoff (attempt ${rateLimitRetryCount}/${maxRetries})`, true);
3277
3363
  this.logFailoverAttempt(prompt.ID, failoverAttempt, true);
@@ -3746,28 +3832,28 @@ export class AIPromptRunner {
3746
3832
  jsonToParse = CleanJSON(rawOutput);
3747
3833
  }
3748
3834
  catch (cleanError) {
3749
- if (params.verbose) {
3750
- this.logError(cleanError, {
3751
- category: 'JSONCleaningFailed',
3752
- metadata: {
3753
- originalError: originalError.message,
3754
- rawOutput: rawOutput.substring(0, 500)
3755
- },
3756
- maxErrorLength: params.maxErrorLength
3757
- });
3758
- }
3835
+ this.logError(cleanError, {
3836
+ category: 'JSONCleaningFailed',
3837
+ metadata: {
3838
+ originalError: originalError.message,
3839
+ rawOutput: rawOutput.substring(0, 500)
3840
+ },
3841
+ maxErrorLength: params.maxErrorLength
3842
+ });
3759
3843
  }
3760
3844
  const json5Result = JSON5.parse(jsonToParse);
3761
- if (params.verbose) {
3762
- this.logStatus(' ✅ JSON5 successfully parsed the malformed JSON', true, params);
3763
- }
3845
+ this.logStatus(' ✅ JSON5 successfully parsed the malformed JSON', true, params);
3846
+ currentPromptRun._jsonRepairInfo = {
3847
+ repaired: true,
3848
+ method: 'JSON5',
3849
+ originalError: originalError.message,
3850
+ rawOutputPrefix: rawOutput.substring(0, 200)
3851
+ };
3764
3852
  return json5Result;
3765
3853
  }
3766
3854
  catch (json5Error) {
3767
3855
  // Step 2: Use AI to repair the JSON
3768
- if (params.verbose) {
3769
- this.logStatus(' 🤖 JSON5 failed, attempting AI-based JSON repair...', true, params);
3770
- }
3856
+ this.logStatus(' 🤖 JSON5 failed, attempting AI-based JSON repair...', true, params);
3771
3857
  try {
3772
3858
  // Find the "Repair JSON" prompt in the "MJ: System" category
3773
3859
  const repairPrompt = AIEngine.Instance.Prompts.find(p => p.Name.trim().toLowerCase() === 'repair json' && p.Category.trim().toLowerCase() === 'mj: system');
@@ -3797,39 +3883,61 @@ export class AIPromptRunner {
3797
3883
  }
3798
3884
  // if we get here, we successfully repaired the JSON!!!
3799
3885
  this.logStatus(' ✅ AI successfully repaired the JSON', true, params);
3886
+ currentPromptRun._jsonRepairInfo = {
3887
+ repaired: true,
3888
+ method: 'AIRepair',
3889
+ originalError: originalError.message,
3890
+ rawOutputPrefix: rawOutput.substring(0, 200),
3891
+ repairPromptRunId: repairResult.promptRun?.ID
3892
+ };
3800
3893
  return repairedJSON;
3801
3894
  }
3802
3895
  catch (aiRepairError) {
3803
- // Both repair attempts failed
3804
- if (params.verbose) {
3805
- this.logError(aiRepairError, {
3806
- category: 'JSONRepairFailed',
3807
- metadata: {
3808
- originalError: originalError.message,
3809
- json5Error: json5Error.message,
3810
- aiError: aiRepairError.message,
3811
- rawOutput: rawOutput.substring(0, 500)
3812
- },
3813
- maxErrorLength: params.maxErrorLength
3814
- });
3815
- }
3896
+ // Both repair attempts failed — always log, this is unexpected LLM behavior
3897
+ this.logError(aiRepairError, {
3898
+ category: 'JSONRepairFailed',
3899
+ metadata: {
3900
+ originalError: originalError.message,
3901
+ json5Error: json5Error.message,
3902
+ aiError: aiRepairError.message,
3903
+ rawOutput: rawOutput.substring(0, 500)
3904
+ },
3905
+ maxErrorLength: params.maxErrorLength
3906
+ });
3816
3907
  throw new Error(`JSON repair failed after both JSON5 and AI attempts: ${originalError.message}`);
3817
3908
  }
3818
3909
  }
3819
3910
  }
3911
+ /**
3912
+ * Returns the parsed form of a prompt's `OutputExample` JSON, memoized by content.
3913
+ * Parsing happens at most once per distinct example string for the life of the process;
3914
+ * parse failures are cached too (so malformed examples aren't re-parsed every attempt).
3915
+ */
3916
+ getParsedOutputExample(outputExample) {
3917
+ const cached = AIPromptRunner._outputExampleCache.get(outputExample);
3918
+ if (cached) {
3919
+ return cached;
3920
+ }
3921
+ let entry;
3922
+ try {
3923
+ entry = { parsed: JSON.parse(outputExample) };
3924
+ }
3925
+ catch (parseError) {
3926
+ entry = { error: parseError instanceof Error ? parseError.message : String(parseError) };
3927
+ }
3928
+ AIPromptRunner._outputExampleCache.set(outputExample, entry);
3929
+ return entry;
3930
+ }
3820
3931
  /**
3821
3932
  * Validates parsed result against JSON schema derived from OutputExample
3822
3933
  */
3823
3934
  async validateAgainstSchema(parsedResult, outputExample, promptId) {
3824
3935
  const validationErrors = [];
3825
3936
  try {
3826
- // Parse the output example
3827
- let exampleObject;
3828
- try {
3829
- exampleObject = JSON.parse(outputExample);
3830
- }
3831
- catch (parseError) {
3832
- const error = new ValidationErrorInfo('outputExample', `Invalid OutputExample JSON: ${parseError.message}`, outputExample, ValidationErrorType.Failure);
3937
+ // Parse the output example (cached by content — it's a static string reused across runs/retries)
3938
+ const { parsed: exampleObject, error: exampleParseError } = this.getParsedOutputExample(outputExample);
3939
+ if (exampleParseError) {
3940
+ const error = new ValidationErrorInfo('outputExample', `Invalid OutputExample JSON: ${exampleParseError}`, outputExample, ValidationErrorType.Failure);
3833
3941
  validationErrors.push(error);
3834
3942
  return validationErrors;
3835
3943
  }
@@ -3915,6 +4023,12 @@ export class AIPromptRunner {
3915
4023
  */
3916
4024
  async updatePromptRun(promptRun, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens) {
3917
4025
  try {
4026
+ // Ensure the initial 'Running' INSERT (and its BaseEntity.finalizeSave post-save reload, which does
4027
+ // init() + SetMany(insertedRow)) has fully landed BEFORE we mutate the final state. Mutating while the
4028
+ // INSERT is still in flight lets the reload revert these values, and the chained UPDATE then persists
4029
+ // the stale 'Running' row (the same race fixed in the agent-run-step queue and action-execution-log).
4030
+ // The model call between create and update almost always covers this; a fast-failing prompt could not.
4031
+ await this._promptRunSaveChains.get(promptRun);
3918
4032
  promptRun.CompletedAt = endTime;
3919
4033
  promptRun.ExecutionTimeMS = executionTimeMS;
3920
4034
  // Determine what to save as the result
@@ -4058,7 +4172,8 @@ export class AIPromptRunner {
4058
4172
  type: e.Type,
4059
4173
  value: e.Value
4060
4174
  })) || [],
4061
- validationDecision: this.getValidationDecisionDescription(parsedResult.validationResult?.Success || false, validationAttempts.length, promptRun.ValidationBehavior || 'Warn')
4175
+ validationDecision: this.getValidationDecisionDescription(parsedResult.validationResult?.Success || false, validationAttempts.length, promptRun.ValidationBehavior || 'Warn'),
4176
+ jsonRepairInfo: promptRun._jsonRepairInfo || null
4062
4177
  });
4063
4178
  }
4064
4179
  else {
@@ -4068,6 +4183,12 @@ export class AIPromptRunner {
4068
4183
  promptRun.FinalValidationPassed = parsedResult.validationResult?.Success !== false;
4069
4184
  promptRun.LastAttemptAt = endTime;
4070
4185
  promptRun.TotalRetryDurationMS = 0;
4186
+ // Even without validation, persist JSON repair info if a repair occurred
4187
+ if (promptRun._jsonRepairInfo) {
4188
+ promptRun.ValidationSummary = JSON.stringify({
4189
+ jsonRepairInfo: promptRun._jsonRepairInfo
4190
+ });
4191
+ }
4071
4192
  }
4072
4193
  // Set Success flag based on validation result
4073
4194
  promptRun.Success = modelResult.success && (parsedResult.validationResult?.Success !== false);
@@ -4093,28 +4214,10 @@ export class AIPromptRunner {
4093
4214
  if (promptRun.Cost !== undefined) {
4094
4215
  promptRun.TotalCost = promptRun.Cost;
4095
4216
  }
4096
- const saveResult = await promptRun.Save();
4097
- if (!saveResult) {
4098
- // Safely extract error message using CompleteMessage getter
4099
- let errorMsg = 'Unknown error';
4100
- try {
4101
- if (promptRun.LatestResult?.CompleteMessage) {
4102
- errorMsg = typeof promptRun.LatestResult.CompleteMessage === 'string'
4103
- ? promptRun.LatestResult.CompleteMessage
4104
- : String(promptRun.LatestResult.CompleteMessage);
4105
- }
4106
- }
4107
- catch (msgError) {
4108
- errorMsg = 'Error accessing error message';
4109
- }
4110
- this.logError(`Failed to update AIPromptRun with results: ${errorMsg}`, {
4111
- category: 'PromptRunUpdate',
4112
- metadata: {
4113
- promptRunId: promptRun.ID,
4114
- updateError: errorMsg
4115
- }
4116
- });
4117
- }
4217
+ // Finalize fire-and-forget. Chains after the initial INSERT (same entity instance) so the
4218
+ // 'Running' INSERT can never overwrite this finalized 'Completed'/'Failed' state. The
4219
+ // execution flow does NOT await this — see queuePromptRunSave / WaitForPendingPromptRunSaves.
4220
+ this.queuePromptRunSave(promptRun);
4118
4221
  }
4119
4222
  catch (error) {
4120
4223
  this.logError(error, {