@memberjunction/ai-prompts 5.41.0 → 5.43.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/AIModelRunner.d.ts +7 -17
- package/dist/AIModelRunner.d.ts.map +1 -1
- package/dist/AIModelRunner.js +12 -39
- package/dist/AIModelRunner.js.map +1 -1
- package/dist/AIPromptRunner.d.ts +39 -91
- package/dist/AIPromptRunner.d.ts.map +1 -1
- package/dist/AIPromptRunner.js +164 -153
- package/dist/AIPromptRunner.js.map +1 -1
- package/dist/ParallelExecution.d.ts +33 -2
- package/dist/ParallelExecution.d.ts.map +1 -1
- package/dist/ParallelExecutionCoordinator.d.ts +26 -27
- package/dist/ParallelExecutionCoordinator.d.ts.map +1 -1
- package/dist/ParallelExecutionCoordinator.js +72 -97
- package/dist/ParallelExecutionCoordinator.js.map +1 -1
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -0
- package/dist/index.js.map +1 -1
- package/package.json +11 -11
package/dist/AIPromptRunner.js
CHANGED
|
@@ -1,12 +1,11 @@
|
|
|
1
1
|
import { BaseLLM, ChatParams, ChatMessageRole, GetAIAPIKey, ErrorAnalyzer, ResolveFileInputStrategy } from '@memberjunction/ai';
|
|
2
2
|
import { AIModelRunner } from './AIModelRunner.js';
|
|
3
3
|
import { AIModelSelectionInfo } from '@memberjunction/ai-core-plus';
|
|
4
|
-
import { LogErrorEx, LogStatus, LogStatusEx, IsVerboseLoggingEnabled, Metadata } from '@memberjunction/core';
|
|
4
|
+
import { BaseEntitySaveQueue, LogErrorEx, LogStatus, LogStatusEx, IsVerboseLoggingEnabled, Metadata } from '@memberjunction/core';
|
|
5
5
|
import { CleanJSON, MJGlobal, JSONValidator, ValidationResult, ValidationErrorInfo, ValidationErrorType, UUIDsEqual, NormalizeUUID } from '@memberjunction/global';
|
|
6
6
|
import { CredentialEngine } from '@memberjunction/credentials';
|
|
7
7
|
import { TemplateEngineServer } from '@memberjunction/templates';
|
|
8
8
|
import { ExecutionPlanner } from './ExecutionPlanner.js';
|
|
9
|
-
import { ParallelExecutionCoordinator } from './ParallelExecutionCoordinator.js';
|
|
10
9
|
import { AIEngine } from '@memberjunction/aiengine';
|
|
11
10
|
import { AIEngineBase } from '@memberjunction/ai-engine-base';
|
|
12
11
|
import { SystemPlaceholderManager } from '@memberjunction/ai-core-plus';
|
|
@@ -55,22 +54,48 @@ export class AIPromptRunner {
|
|
|
55
54
|
constructor() {
|
|
56
55
|
this._provider = null;
|
|
57
56
|
/**
|
|
58
|
-
*
|
|
59
|
-
*
|
|
60
|
-
*
|
|
61
|
-
*
|
|
62
|
-
*
|
|
57
|
+
* Fire-and-forget AIPromptRun persistence. Prompt-run logging never blocks the execution path on a
|
|
58
|
+
* DB round-trip; the shared {@link BaseEntitySaveQueue} sequences saves for the SAME entity (the
|
|
59
|
+
* initial 'Running' INSERT always completes before the finalize UPDATE, and the finalize mutation
|
|
60
|
+
* runs INSIDE the post-INSERT task so a slow INSERT can never clobber the finalized row). Failures
|
|
61
|
+
* stay in this runner's structured log stream via the queue's `onError` hook.
|
|
63
62
|
*/
|
|
64
|
-
this.
|
|
65
|
-
|
|
66
|
-
|
|
63
|
+
this._promptRunQueue = new BaseEntitySaveQueue({
|
|
64
|
+
onError: (message) => this.logError(message, { category: 'PromptRunSave' }),
|
|
65
|
+
});
|
|
67
66
|
this._metadata = this._provider ?? new Metadata();
|
|
68
67
|
this._templateEngine = TemplateEngineServer.Instance;
|
|
69
68
|
this._executionPlanner = new ExecutionPlanner();
|
|
70
|
-
this._parallelCoordinator = new ParallelExecutionCoordinator();
|
|
71
69
|
this._jsonValidator = new JSONValidator();
|
|
72
70
|
this._modelRunner = new AIModelRunner();
|
|
73
71
|
}
|
|
72
|
+
/** ClassFactory key the parallel coordinator self-registers under (see {@link ParallelCoordinator}). */
|
|
73
|
+
static { this.PARALLEL_COORDINATOR_KEY = 'ParallelExecutionCoordinator'; }
|
|
74
|
+
/**
|
|
75
|
+
* Lazily resolves the parallel execution coordinator.
|
|
76
|
+
*
|
|
77
|
+
* The coordinator is a SUBCLASS of AIPromptRunner — it inherits {@link executeModel} and the rest
|
|
78
|
+
* of the battle-tested execution path so there is a single source of truth for credential / driver
|
|
79
|
+
* / ChatParams / streaming resolution. That subclass relationship means the base cannot statically
|
|
80
|
+
* `new` it without a hard circular import, so we resolve it through the ClassFactory instead (the
|
|
81
|
+
* coordinator self-registers via `@RegisterClass(AIPromptRunner, PARALLEL_COORDINATOR_KEY)`).
|
|
82
|
+
* Created once per runner and reused; the runner's Provider override is propagated. Throws a clear
|
|
83
|
+
* error if the coordinator class was never loaded/registered — otherwise ClassFactory would silently
|
|
84
|
+
* fall back to a plain AIPromptRunner that lacks the parallel methods.
|
|
85
|
+
*/
|
|
86
|
+
get ParallelCoordinator() {
|
|
87
|
+
if (!this._parallelCoordinator) {
|
|
88
|
+
const instance = MJGlobal.Instance.ClassFactory.CreateInstance(AIPromptRunner, AIPromptRunner.PARALLEL_COORDINATOR_KEY);
|
|
89
|
+
if (!instance || typeof instance.executeTasksInParallel !== 'function') {
|
|
90
|
+
throw new Error(`ParallelExecutionCoordinator is not registered with the ClassFactory. Ensure ` +
|
|
91
|
+
`'@memberjunction/ai-prompts' is fully loaded (it is exported from the package index and ` +
|
|
92
|
+
`picked up by the class-registration manifest).`);
|
|
93
|
+
}
|
|
94
|
+
instance.Provider = this.Provider;
|
|
95
|
+
this._parallelCoordinator = instance;
|
|
96
|
+
}
|
|
97
|
+
return this._parallelCoordinator;
|
|
98
|
+
}
|
|
74
99
|
/**
|
|
75
100
|
* Access the underlying AIModelRunner for embedding and other non-LLM model calls.
|
|
76
101
|
* Use this when you need tracked embedding execution with AIPromptRun record creation.
|
|
@@ -457,25 +482,20 @@ export class AIPromptRunner {
|
|
|
457
482
|
let renderedPromptText = '';
|
|
458
483
|
// For hierarchical prompts, we need to create the parent prompt run first to get its ID
|
|
459
484
|
let parentPromptRun;
|
|
460
|
-
let selectedModel;
|
|
461
485
|
let childTemplateRenderingResult;
|
|
462
|
-
let
|
|
486
|
+
let selection;
|
|
463
487
|
// Handle different prompt execution modes
|
|
464
488
|
if (params.childPrompts && params.childPrompts.length > 0) {
|
|
465
489
|
// Hierarchical template composition mode - render child templates first, then compose
|
|
466
|
-
//this.logStatus(`🌳 Composing prompt with ${params.childPrompts.length} child templates in hierarchical mode`, true, params);
|
|
467
490
|
// Determine which prompt to use for model selection
|
|
468
491
|
let modelSelectionPrompt = prompt;
|
|
469
492
|
if (params.modelSelectionPrompt) {
|
|
470
493
|
modelSelectionPrompt = params.modelSelectionPrompt;
|
|
471
|
-
//this.logStatus(`🎯 Using prompt "${modelSelectionPrompt.Name}" for model selection instead of parent prompt`, true, params);
|
|
472
494
|
}
|
|
473
|
-
// Select model using the appropriate prompt
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
if (!selectedModel) {
|
|
478
|
-
throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, modelSelectionInfo));
|
|
495
|
+
// Select model using the appropriate prompt — capture the FULL result
|
|
496
|
+
selection = await this.selectModel(modelSelectionPrompt, params.override?.modelId, params.contextUser, params.configurationId, params.override?.vendorId, params);
|
|
497
|
+
if (!selection.model) {
|
|
498
|
+
throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, selection.selectionInfo));
|
|
479
499
|
}
|
|
480
500
|
// Check if we have a system prompt override
|
|
481
501
|
if (params.systemPromptOverride) {
|
|
@@ -490,7 +510,7 @@ export class AIPromptRunner {
|
|
|
490
510
|
renderedPromptText = await this.renderPromptWithChildTemplates(prompt, params, childTemplateRenderingResult.renderedTemplates);
|
|
491
511
|
}
|
|
492
512
|
// Create parent prompt run for the final composed prompt execution
|
|
493
|
-
parentPromptRun = await this.createPromptRun(prompt,
|
|
513
|
+
parentPromptRun = await this.createPromptRun(prompt, selection.model, params, renderedPromptText, startTime, params.override?.vendorId, selection.selectionInfo);
|
|
494
514
|
}
|
|
495
515
|
else if (prompt.TemplateID && (!params.conversationMessages || params.templateMessageRole !== 'none')) {
|
|
496
516
|
// Check if we have a system prompt override
|
|
@@ -520,30 +540,28 @@ export class AIPromptRunner {
|
|
|
520
540
|
if (params.cancellationToken?.aborted) {
|
|
521
541
|
throw new Error('Prompt execution was cancelled during template rendering');
|
|
522
542
|
}
|
|
523
|
-
// If no model was selected yet (
|
|
524
|
-
if (!
|
|
543
|
+
// If no model was selected yet (non-hierarchical case), select one now — capture the FULL result
|
|
544
|
+
if (!selection?.model) {
|
|
525
545
|
let modelSelectionPrompt = prompt;
|
|
526
546
|
if (params.modelSelectionPrompt) {
|
|
527
547
|
modelSelectionPrompt = params.modelSelectionPrompt;
|
|
528
548
|
this.logStatus(`🎯 Using prompt "${modelSelectionPrompt.Name}" for model selection instead of main prompt`, true, params);
|
|
529
549
|
}
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
if (!selectedModel) {
|
|
534
|
-
throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, modelSelectionInfo));
|
|
550
|
+
selection = await this.selectModel(modelSelectionPrompt, params.override?.modelId, params.contextUser, params.configurationId, params.override?.vendorId, params);
|
|
551
|
+
if (!selection.model) {
|
|
552
|
+
throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, selection.selectionInfo));
|
|
535
553
|
}
|
|
536
554
|
}
|
|
537
555
|
// Check if we need parallel execution based on ParallelizationMode
|
|
538
556
|
const shouldUseParallelExecution = prompt.ParallelizationMode && prompt.ParallelizationMode !== 'None';
|
|
539
557
|
let result;
|
|
540
558
|
if (shouldUseParallelExecution) {
|
|
541
|
-
// Use parallel execution path
|
|
542
|
-
result = await this.executePromptInParallel(prompt, renderedPromptText, params, startTime, parentPromptRun,
|
|
559
|
+
// Use parallel execution path — pass full selection through
|
|
560
|
+
result = await this.executePromptInParallel(prompt, renderedPromptText, params, startTime, parentPromptRun, selection);
|
|
543
561
|
}
|
|
544
562
|
else {
|
|
545
|
-
// Use traditional single execution path
|
|
546
|
-
result = await this.executeSinglePrompt(prompt, renderedPromptText, params, startTime, parentPromptRun,
|
|
563
|
+
// Use traditional single execution path — pass full selection through
|
|
564
|
+
result = await this.executeSinglePrompt(prompt, renderedPromptText, params, startTime, parentPromptRun, selection);
|
|
547
565
|
}
|
|
548
566
|
// Note: With template composition, we only execute once so no rollup calculations needed
|
|
549
567
|
// The final composed prompt is executed as a single operation
|
|
@@ -613,33 +631,22 @@ export class AIPromptRunner {
|
|
|
613
631
|
* @param startTime - Execution start time
|
|
614
632
|
* @returns Promise<AIPromptRunResult<T>> - The execution result
|
|
615
633
|
*/
|
|
616
|
-
async executeSinglePrompt(prompt, renderedPromptText, params, startTime, existingPromptRun,
|
|
634
|
+
async executeSinglePrompt(prompt, renderedPromptText, params, startTime, existingPromptRun, existingSelection) {
|
|
617
635
|
// Check for cancellation before model selection
|
|
618
636
|
if (params.cancellationToken?.aborted) {
|
|
619
637
|
throw new Error('Prompt execution was cancelled before model selection');
|
|
620
638
|
}
|
|
621
|
-
// Use existing
|
|
622
|
-
let selectedModel =
|
|
623
|
-
let modelSelectionInfo =
|
|
624
|
-
let vendorDriverClass;
|
|
625
|
-
let vendorApiName;
|
|
626
|
-
let vendorSupportsEffortLevel;
|
|
627
|
-
let modelEffortLevel;
|
|
628
|
-
let allCandidates = [];
|
|
629
|
-
|
|
630
|
-
|
|
631
|
-
|
|
632
|
-
const modelID = modelSelectionInfo.modelSelected.ID;
|
|
633
|
-
const modelVendor = AIEngine.Instance.ModelVendorsByModelID.get(NormalizeUUID(modelID))
|
|
634
|
-
?.find(mv => UUIDsEqual(mv.VendorID, vendorID));
|
|
635
|
-
if (modelVendor) {
|
|
636
|
-
vendorDriverClass = modelVendor.DriverClass;
|
|
637
|
-
vendorApiName = modelVendor.APIName;
|
|
638
|
-
vendorSupportsEffortLevel = modelVendor.SupportsEffortLevel;
|
|
639
|
-
}
|
|
640
|
-
// Extract valid candidates from selection info for retry logic
|
|
641
|
-
allCandidates = this.buildCandidatesFromSelectionInfo(modelSelectionInfo);
|
|
642
|
-
}
|
|
639
|
+
// Use existing selection if provided (hierarchical case) or select now
|
|
640
|
+
let selectedModel = existingSelection?.model ?? undefined;
|
|
641
|
+
let modelSelectionInfo = existingSelection?.selectionInfo;
|
|
642
|
+
let vendorDriverClass = existingSelection?.vendorDriverClass;
|
|
643
|
+
let vendorApiName = existingSelection?.vendorApiName;
|
|
644
|
+
let vendorSupportsEffortLevel = existingSelection?.vendorSupportsEffortLevel;
|
|
645
|
+
let modelEffortLevel = existingSelection?.modelEffortLevel;
|
|
646
|
+
let allCandidates = existingSelection?.allCandidates ?? [];
|
|
647
|
+
// Credential probes already done during selection — reused by failover so it doesn't
|
|
648
|
+
// recompute hasCredentialsAvailable for the prefix it walks before the selected candidate.
|
|
649
|
+
let credentialAvailability = existingSelection?.credentialAvailability;
|
|
643
650
|
if (!selectedModel) {
|
|
644
651
|
// Determine which prompt to use for model selection
|
|
645
652
|
let modelSelectionPrompt = prompt;
|
|
@@ -655,6 +662,7 @@ export class AIPromptRunner {
|
|
|
655
662
|
modelEffortLevel = modelResult.modelEffortLevel;
|
|
656
663
|
modelSelectionInfo = modelResult.selectionInfo;
|
|
657
664
|
allCandidates = modelResult.allCandidates || [];
|
|
665
|
+
credentialAvailability = modelResult.credentialAvailability;
|
|
658
666
|
if (!selectedModel) {
|
|
659
667
|
throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, modelSelectionInfo));
|
|
660
668
|
}
|
|
@@ -670,7 +678,8 @@ export class AIPromptRunner {
|
|
|
670
678
|
throw new Error('Prompt execution was cancelled before model execution');
|
|
671
679
|
}
|
|
672
680
|
// Execute with retry logic for validation failures
|
|
673
|
-
const { modelResult, parsedResult, validationAttempts, cumulativeTokens } = await this.executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel // Pass model-specific effort level
|
|
681
|
+
const { modelResult, parsedResult, validationAttempts, cumulativeTokens } = await this.executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, // Pass model-specific effort level
|
|
682
|
+
credentialAvailability // Reuse credential probes from selection
|
|
674
683
|
);
|
|
675
684
|
// Calculate execution metrics
|
|
676
685
|
const endTime = new Date();
|
|
@@ -730,7 +739,7 @@ export class AIPromptRunner {
|
|
|
730
739
|
* @param startTime - Execution start time
|
|
731
740
|
* @returns Promise<AIPromptRunResult<T>> - The aggregated execution result
|
|
732
741
|
*/
|
|
733
|
-
async executePromptInParallel(prompt, renderedPromptText, params, startTime, existingPromptRun,
|
|
742
|
+
async executePromptInParallel(prompt, renderedPromptText, params, startTime, existingPromptRun, existingSelection) {
|
|
734
743
|
// Check for cancellation before starting parallel execution
|
|
735
744
|
if (params.cancellationToken?.aborted) {
|
|
736
745
|
throw new Error('Parallel execution was cancelled before starting');
|
|
@@ -738,21 +747,21 @@ export class AIPromptRunner {
|
|
|
738
747
|
// Load AI Engine to get models and prompt models
|
|
739
748
|
await AIEngine.Instance.Config(false, params.contextUser);
|
|
740
749
|
let executionTasks;
|
|
741
|
-
// If a model is already selected (from hierarchical template composition),
|
|
750
|
+
// If a model is already selected (from hierarchical template composition),
|
|
742
751
|
// create a single task with that model instead of using the planner
|
|
743
|
-
if (
|
|
752
|
+
if (existingSelection?.model) {
|
|
744
753
|
// Create a single execution task with the pre-selected model
|
|
745
754
|
executionTasks = [{
|
|
746
755
|
taskId: 'pre-selected',
|
|
747
|
-
model:
|
|
748
|
-
vendorDriverClass:
|
|
749
|
-
vendorApiName:
|
|
756
|
+
model: existingSelection.model,
|
|
757
|
+
vendorDriverClass: existingSelection.vendorDriverClass,
|
|
758
|
+
vendorApiName: existingSelection.vendorApiName,
|
|
750
759
|
messages: params.conversationMessages || [],
|
|
751
760
|
promptText: renderedPromptText,
|
|
752
761
|
templateMessageRole: params.templateMessageRole || 'system',
|
|
753
762
|
contextUser: params.contextUser
|
|
754
763
|
}];
|
|
755
|
-
this.logStatus(` Using pre-selected model "${
|
|
764
|
+
this.logStatus(` Using pre-selected model "${existingSelection.model.Name}" for parallel execution`, true, params);
|
|
756
765
|
}
|
|
757
766
|
else {
|
|
758
767
|
// Normal parallel execution path - let the planner decide
|
|
@@ -777,7 +786,7 @@ export class AIPromptRunner {
|
|
|
777
786
|
throw new Error('Parallel execution was cancelled before task execution');
|
|
778
787
|
}
|
|
779
788
|
// Execute tasks in parallel
|
|
780
|
-
const parallelResult = await this.
|
|
789
|
+
const parallelResult = await this.ParallelCoordinator.executeTasksInParallel(params, executionTasks, undefined, undefined, params.cancellationToken, undefined, params.agentRunId);
|
|
781
790
|
if (!parallelResult.success) {
|
|
782
791
|
throw new Error(`Parallel execution failed: ${parallelResult.errors.join(', ')}`);
|
|
783
792
|
}
|
|
@@ -793,7 +802,7 @@ export class AIPromptRunner {
|
|
|
793
802
|
method: 'PromptSelector',
|
|
794
803
|
selectorPromptId: prompt.ResultSelectorPromptID,
|
|
795
804
|
};
|
|
796
|
-
const aiSelectedResult = await this.
|
|
805
|
+
const aiSelectedResult = await this.ParallelCoordinator.selectBestResult(successfulResults, selectionConfig, undefined, params.cancellationToken);
|
|
797
806
|
if (aiSelectedResult) {
|
|
798
807
|
selectedResult = aiSelectedResult;
|
|
799
808
|
}
|
|
@@ -822,7 +831,7 @@ export class AIPromptRunner {
|
|
|
822
831
|
}
|
|
823
832
|
// Use existing prompt run if provided (hierarchical case) or create new one
|
|
824
833
|
// Use the model selection info if provided (from hierarchical execution)
|
|
825
|
-
const consolidatedPromptRun = existingPromptRun || await this.createPromptRun(prompt, selectedResult.task.model, params, renderedPromptText, startTime, params.override?.vendorId,
|
|
834
|
+
const consolidatedPromptRun = existingPromptRun || await this.createPromptRun(prompt, selectedResult.task.model, params, renderedPromptText, startTime, params.override?.vendorId, existingSelection?.selectionInfo);
|
|
826
835
|
// Update with parallel execution metadata
|
|
827
836
|
const endTime = new Date();
|
|
828
837
|
consolidatedPromptRun.CompletedAt = endTime;
|
|
@@ -875,8 +884,10 @@ export class AIPromptRunner {
|
|
|
875
884
|
// Set Status and WasSelectedResult for parallel execution
|
|
876
885
|
consolidatedPromptRun.Status = parallelResult.successCount > 0 ? 'Completed' : 'Failed';
|
|
877
886
|
consolidatedPromptRun.WasSelectedResult = true; // This is the consolidated result chosen by judge
|
|
878
|
-
// Persist the consolidated run fire-and-forget; chains after its INSERT via
|
|
879
|
-
|
|
887
|
+
// Persist the consolidated run fire-and-forget; the finalize UPDATE chains after its INSERT via
|
|
888
|
+
// the save queue. These fields are set after all the awaited parallel work, so the INSERT has long
|
|
889
|
+
// landed — a plain Update (no post-INSERT callback) is race-safe here.
|
|
890
|
+
this._promptRunQueue.Update(consolidatedPromptRun);
|
|
880
891
|
// Create additional results from all other successful results (excluding the best one)
|
|
881
892
|
const additionalResults = [];
|
|
882
893
|
// Sort successful results by ranking (if available) or keep original order
|
|
@@ -944,11 +955,11 @@ export class AIPromptRunner {
|
|
|
944
955
|
modelInfo: {
|
|
945
956
|
modelId: selectedResult.task.model.ID,
|
|
946
957
|
modelName: selectedResult.task.model.Name,
|
|
947
|
-
vendorId:
|
|
958
|
+
vendorId: existingSelection?.selectionInfo?.vendorSelected?.ID, // VendorID not directly available on AIModel
|
|
948
959
|
vendorName: selectedResult.task.model.Vendor,
|
|
949
960
|
},
|
|
950
961
|
judgeMetadata: selectedResult.judgeMetadata,
|
|
951
|
-
modelSelectionInfo:
|
|
962
|
+
modelSelectionInfo: existingSelection?.selectionInfo, // Include model selection info if provided
|
|
952
963
|
};
|
|
953
964
|
}
|
|
954
965
|
/**
|
|
@@ -1275,7 +1286,7 @@ export class AIPromptRunner {
|
|
|
1275
1286
|
// });
|
|
1276
1287
|
// }
|
|
1277
1288
|
// Select the first candidate with available credentials and track all attempts
|
|
1278
|
-
const { selected, consideredModels } = await this.selectModelWithAPIKeyTracked(candidates, prompt.ID, params);
|
|
1289
|
+
const { selected, consideredModels, credentialAvailability } = await this.selectModelWithAPIKeyTracked(candidates, prompt.ID, params);
|
|
1279
1290
|
// Merge considered models into our tracking
|
|
1280
1291
|
modelsConsidered.push(...consideredModels);
|
|
1281
1292
|
if (!selected) {
|
|
@@ -1287,6 +1298,7 @@ export class AIPromptRunner {
|
|
|
1287
1298
|
vendorSupportsEffortLevel: undefined,
|
|
1288
1299
|
modelEffortLevel: undefined,
|
|
1289
1300
|
allCandidates: candidates,
|
|
1301
|
+
credentialAvailability,
|
|
1290
1302
|
selectionInfo: this.createSelectionInfo({
|
|
1291
1303
|
aiConfiguration: configuration,
|
|
1292
1304
|
modelsConsidered,
|
|
@@ -1331,6 +1343,7 @@ export class AIPromptRunner {
|
|
|
1331
1343
|
vendorSupportsEffortLevel: selected.supportsEffortLevel,
|
|
1332
1344
|
modelEffortLevel: selected.effortLevel, // Pass through model-specific effort level
|
|
1333
1345
|
allCandidates: candidates,
|
|
1346
|
+
credentialAvailability,
|
|
1334
1347
|
selectionInfo: this.createSelectionInfo({
|
|
1335
1348
|
aiConfiguration: configuration,
|
|
1336
1349
|
modelsConsidered,
|
|
@@ -1879,33 +1892,6 @@ export class AIPromptRunner {
|
|
|
1879
1892
|
Object.assign(info, data);
|
|
1880
1893
|
return info;
|
|
1881
1894
|
}
|
|
1882
|
-
/**
|
|
1883
|
-
* Converts model selection info into ModelVendorCandidate array for retry logic.
|
|
1884
|
-
* Extracts only the valid candidates (those with available API keys) from the selection info.
|
|
1885
|
-
*
|
|
1886
|
-
* @param selectionInfo - Model selection information containing considered models
|
|
1887
|
-
* @returns Array of valid model-vendor candidates sorted by priority
|
|
1888
|
-
*/
|
|
1889
|
-
buildCandidatesFromSelectionInfo(selectionInfo) {
|
|
1890
|
-
const validModels = selectionInfo.extractValidCandidates();
|
|
1891
|
-
return validModels.map(considered => {
|
|
1892
|
-
// Find matching model vendor for driver and API info
|
|
1893
|
-
const modelVendor = considered.vendor
|
|
1894
|
-
? considered.model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, considered.vendor.ID))
|
|
1895
|
-
: undefined;
|
|
1896
|
-
return {
|
|
1897
|
-
model: considered.model,
|
|
1898
|
-
vendorId: considered.vendor?.ID,
|
|
1899
|
-
vendorName: considered.vendor?.Name,
|
|
1900
|
-
driverClass: modelVendor?.DriverClass || considered.model.DriverClass,
|
|
1901
|
-
apiName: modelVendor?.APIName || considered.model.APIName,
|
|
1902
|
-
supportsEffortLevel: modelVendor?.SupportsEffortLevel ?? considered.model.SupportsEffortLevel ?? false,
|
|
1903
|
-
isPreferredVendor: false, // Can't determine from selection info alone
|
|
1904
|
-
priority: considered.priority,
|
|
1905
|
-
source: (selectionInfo.selectionStrategy === 'ByPower' ? 'power-rank' : 'model-type')
|
|
1906
|
-
};
|
|
1907
|
-
}).sort((a, b) => b.priority - a.priority); // Sort by priority descending
|
|
1908
|
-
}
|
|
1909
1895
|
/**
|
|
1910
1896
|
* Enhanced version of selectModelWithAPIKey that tracks all considered models
|
|
1911
1897
|
* for model selection reporting. Uses the hierarchical credential resolution
|
|
@@ -1997,7 +1983,7 @@ export class AIPromptRunner {
|
|
|
1997
1983
|
maxErrorLength: params?.maxErrorLength
|
|
1998
1984
|
});
|
|
1999
1985
|
}
|
|
2000
|
-
return { selected: selectedCandidate, consideredModels };
|
|
1986
|
+
return { selected: selectedCandidate, consideredModels, credentialAvailability: credentialCache };
|
|
2001
1987
|
}
|
|
2002
1988
|
/**
|
|
2003
1989
|
* Builds a descriptive error message when no model could be selected for a prompt.
|
|
@@ -2049,51 +2035,12 @@ export class AIPromptRunner {
|
|
|
2049
2035
|
};
|
|
2050
2036
|
}
|
|
2051
2037
|
/**
|
|
2052
|
-
*
|
|
2053
|
-
*
|
|
2054
|
-
*
|
|
2055
|
-
* chain runs independently of the execution flow (callers do NOT await it), so the model call
|
|
2056
|
-
* is never delayed by a DB write.
|
|
2057
|
-
*
|
|
2058
|
-
* Save failures are logged (non-fatal): the AIPromptRun record is observability, not part of the
|
|
2059
|
-
* prompt's success contract, so a rare persistence failure must not fail the prompt. The chained
|
|
2060
|
-
* promise is tracked in {@link _pendingPromptRunSaves} so {@link WaitForPendingPromptRunSaves}
|
|
2061
|
-
* can flush them when determinism is required (e.g. tests, or a caller that needs the rows
|
|
2062
|
-
* durably written). Returns that promise.
|
|
2063
|
-
*/
|
|
2064
|
-
queuePromptRunSave(promptRun) {
|
|
2065
|
-
const previous = this._promptRunSaveChains.get(promptRun) ?? Promise.resolve(true);
|
|
2066
|
-
const current = previous
|
|
2067
|
-
.then(async () => {
|
|
2068
|
-
const ok = await promptRun.Save();
|
|
2069
|
-
if (!ok) {
|
|
2070
|
-
this.logError(`Failed to save AIPromptRun ${promptRun.ID || '(unsaved)'}: ${promptRun.LatestResult?.CompleteMessage || 'Unknown error'}`, {
|
|
2071
|
-
category: 'PromptRunSave',
|
|
2072
|
-
metadata: { promptRunId: promptRun.ID }
|
|
2073
|
-
});
|
|
2074
|
-
}
|
|
2075
|
-
return ok;
|
|
2076
|
-
})
|
|
2077
|
-
.catch((err) => {
|
|
2078
|
-
// Infrastructure-level throw (network, etc.) — log and swallow so the fire-and-forget
|
|
2079
|
-
// promise never surfaces as an unhandled rejection.
|
|
2080
|
-
this.logError(err instanceof Error ? err : new Error(String(err)), {
|
|
2081
|
-
category: 'PromptRunSave',
|
|
2082
|
-
metadata: { promptRunId: promptRun.ID }
|
|
2083
|
-
});
|
|
2084
|
-
return false;
|
|
2085
|
-
});
|
|
2086
|
-
this._promptRunSaveChains.set(promptRun, current);
|
|
2087
|
-
this._pendingPromptRunSaves.push(current);
|
|
2088
|
-
return current;
|
|
2089
|
-
}
|
|
2090
|
-
/**
|
|
2091
|
-
* Awaits all in-flight prompt-run saves queued by this runner instance. The normal execution
|
|
2092
|
-
* path does NOT call this — prompt-run persistence is intentionally fire-and-forget. Exposed for
|
|
2093
|
-
* tests and for callers that need the AIPromptRun rows durably written before proceeding.
|
|
2038
|
+
* Awaits all in-flight prompt-run saves queued by this runner instance. The normal execution path
|
|
2039
|
+
* does NOT call this — prompt-run persistence is intentionally fire-and-forget. Exposed for tests
|
|
2040
|
+
* and for callers that need the AIPromptRun rows durably written before proceeding.
|
|
2094
2041
|
*/
|
|
2095
2042
|
async WaitForPendingPromptRunSaves() {
|
|
2096
|
-
await
|
|
2043
|
+
await this._promptRunQueue.Flush();
|
|
2097
2044
|
}
|
|
2098
2045
|
async createPromptRun(prompt, model, params, systemPromptText, startTime, vendorId, modelSelectionInfo) {
|
|
2099
2046
|
const provider = params.provider ?? Metadata.Provider;
|
|
@@ -2110,7 +2057,7 @@ export class AIPromptRunner {
|
|
|
2110
2057
|
promptRun.Status = 'Running';
|
|
2111
2058
|
promptRun.Cancelled = false;
|
|
2112
2059
|
promptRun.CacheHit = false;
|
|
2113
|
-
promptRun.StreamingEnabled =
|
|
2060
|
+
promptRun.StreamingEnabled = !!params.onStreaming;
|
|
2114
2061
|
promptRun.WasSelectedResult = false;
|
|
2115
2062
|
// Set model selection tracking fields
|
|
2116
2063
|
if (modelSelectionInfo) {
|
|
@@ -2261,7 +2208,7 @@ export class AIPromptRunner {
|
|
|
2261
2208
|
// NewRecord() above, so callers (and the onPromptRunCreated callback) have it immediately —
|
|
2262
2209
|
// we don't block the model call on the INSERT. The finalize UPDATE chains after this INSERT
|
|
2263
2210
|
// via the instance-keyed save queue, so ordering is guaranteed.
|
|
2264
|
-
this.
|
|
2211
|
+
this._promptRunQueue.Insert(promptRun);
|
|
2265
2212
|
// Invoke callback if provided. The ID is available without awaiting the save (client-generated
|
|
2266
2213
|
// by NewRecord()), so agent-run/step linking that depends on it works immediately.
|
|
2267
2214
|
if (params.onPromptRunCreated) {
|
|
@@ -2353,7 +2300,7 @@ export class AIPromptRunner {
|
|
|
2353
2300
|
* - updatePromptRunWithFailoverFailure: Records failed failover metadata
|
|
2354
2301
|
* - createFailoverErrorResult: Creates standardized error response
|
|
2355
2302
|
*/
|
|
2356
|
-
async executeModelWithFailover(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, allCandidates, promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel) {
|
|
2303
|
+
async executeModelWithFailover(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, allCandidates, promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability) {
|
|
2357
2304
|
// Get failover configuration (used for errorScope filtering)
|
|
2358
2305
|
const failoverConfig = this.getFailoverConfiguration(prompt);
|
|
2359
2306
|
// If no candidates provided or failover disabled, execute normally with first model
|
|
@@ -2363,10 +2310,45 @@ export class AIPromptRunner {
|
|
|
2363
2310
|
// Track failover attempts
|
|
2364
2311
|
const failoverAttempts = [];
|
|
2365
2312
|
let lastError = null;
|
|
2313
|
+
// Cache credential availability per driver:model:vendor for the duration of this failover
|
|
2314
|
+
// scan so we don't repeat env-var / binding lookups while walking the candidate list.
|
|
2315
|
+
//
|
|
2316
|
+
// PERF: seed it with the probes model SELECTION already performed (same key format). Selection
|
|
2317
|
+
// walks the priority list until it finds the first credentialed candidate, so this map holds
|
|
2318
|
+
// the prefix it rejected (known false) PLUS the selected candidate (known true) — which is
|
|
2319
|
+
// exactly the segment failover re-walks on the happy path. Reusing those results means the
|
|
2320
|
+
// common case (and any caller looping failover) does ZERO redundant hasCredentialsAvailable
|
|
2321
|
+
// calls. The not-evaluated tail is intentionally absent, so failover still lazily probes it
|
|
2322
|
+
// only if a real failure forces it to walk down there.
|
|
2323
|
+
const failoverCredentialCache = credentialAvailability
|
|
2324
|
+
? new Map(credentialAvailability)
|
|
2325
|
+
: new Map();
|
|
2326
|
+
const candidateHasCredentials = (c) => {
|
|
2327
|
+
const key = `${c.driverClass}:${c.model.ID}:${c.vendorId || 'default'}`;
|
|
2328
|
+
let has = failoverCredentialCache.get(key);
|
|
2329
|
+
if (has === undefined) {
|
|
2330
|
+
has = this.hasCredentialsAvailable(c.driverClass, prompt.ID, c.model.ID, c.vendorId, params);
|
|
2331
|
+
failoverCredentialCache.set(key, has);
|
|
2332
|
+
}
|
|
2333
|
+
return has;
|
|
2334
|
+
};
|
|
2335
|
+
let skippedForCredentials = 0;
|
|
2366
2336
|
// Iterate through all candidates in priority order with instant failover
|
|
2367
2337
|
for (let i = 0; i < allCandidates.length; i++) {
|
|
2368
2338
|
const candidate = allCandidates[i];
|
|
2369
2339
|
const attemptStartTime = Date.now();
|
|
2340
|
+
// Skip candidates with no credentials configured. `allCandidates` is intentionally the
|
|
2341
|
+
// FULL priority-ordered list (see the DECISION note in selectModelWithAPIKeyTracked),
|
|
2342
|
+
// so it can include vendors that have no API key in this environment. Firing a live
|
|
2343
|
+
// request at one of those produces a misleading "401 invalid API key" — and because an
|
|
2344
|
+
// Authentication error is treated as fatal, it would halt failover before any
|
|
2345
|
+
// credentialed candidate is ever reached. Skipping here makes failover land on the
|
|
2346
|
+
// first candidate that can actually authenticate (mirroring model selection's own
|
|
2347
|
+
// highest-priority-with-credentials rule).
|
|
2348
|
+
if (!candidateHasCredentials(candidate)) {
|
|
2349
|
+
skippedForCredentials++;
|
|
2350
|
+
continue;
|
|
2351
|
+
}
|
|
2370
2352
|
try {
|
|
2371
2353
|
// Log the attempt if not the first one
|
|
2372
2354
|
if (i > 0) {
|
|
@@ -2435,6 +2417,11 @@ export class AIPromptRunner {
|
|
|
2435
2417
|
if (promptRun && failoverAttempts.length > 0) {
|
|
2436
2418
|
this.updatePromptRunWithFailoverFailure(promptRun, failoverAttempts);
|
|
2437
2419
|
}
|
|
2420
|
+
// If every candidate was skipped for missing credentials we never attempted a call and
|
|
2421
|
+
// have no underlying error to report — surface an actionable message instead of null.
|
|
2422
|
+
if (!lastError && failoverAttempts.length === 0 && skippedForCredentials > 0) {
|
|
2423
|
+
lastError = new Error(`No API credentials configured for any of the ${skippedForCredentials} candidate model-vendor combination(s) for prompt "${prompt.Name}".`);
|
|
2424
|
+
}
|
|
2438
2425
|
return this.createFailoverErrorResult(lastError, failoverAttempts);
|
|
2439
2426
|
}
|
|
2440
2427
|
/**
|
|
@@ -2570,7 +2557,11 @@ export class AIPromptRunner {
|
|
|
2570
2557
|
};
|
|
2571
2558
|
}
|
|
2572
2559
|
/**
|
|
2573
|
-
* Executes the AI model with the rendered prompt
|
|
2560
|
+
* Executes the AI model with the rendered prompt.
|
|
2561
|
+
*
|
|
2562
|
+
* `protected` so the parallel coordinator subclass reuses this exact code path — credential
|
|
2563
|
+
* resolution, driver selection, ChatParams construction, prefill, media handling, and streaming
|
|
2564
|
+
* all live here ONCE. Do not duplicate this logic elsewhere.
|
|
2574
2565
|
*/
|
|
2575
2566
|
async executeModel(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel) {
|
|
2576
2567
|
// define these variables here to ensure they're available in the catch block
|
|
@@ -2712,6 +2703,17 @@ export class AIPromptRunner {
|
|
|
2712
2703
|
this.stripUnsupportedMediaBlocks(llm, chatParams, model, verbose, params);
|
|
2713
2704
|
// Apply assistant prefill (native or fallback) based on prompt config and provider support
|
|
2714
2705
|
this.applyAssistantPrefill(chatParams, prompt, model, vendorId, llm);
|
|
2706
|
+
// Streaming: wire the prompt-level onStreaming callback into the LLM call. This is the SINGLE
|
|
2707
|
+
// place streaming is configured for prompt execution, so the single-model path and the parallel
|
|
2708
|
+
// path (which bridges its per-task callbacks into params.onStreaming) stream through identical
|
|
2709
|
+
// code — no second streaming implementation that can drift.
|
|
2710
|
+
if (params.onStreaming) {
|
|
2711
|
+
const onStreaming = params.onStreaming;
|
|
2712
|
+
chatParams.streaming = true;
|
|
2713
|
+
chatParams.streamingCallbacks = {
|
|
2714
|
+
OnContent: (chunk, isComplete) => onStreaming({ content: chunk, isComplete, modelName: model.Name }),
|
|
2715
|
+
};
|
|
2716
|
+
}
|
|
2715
2717
|
// Execute the model with cancellation support
|
|
2716
2718
|
if (cancellationToken) {
|
|
2717
2719
|
// If cancellation token is provided, wrap the execution to handle cancellation
|
|
@@ -3091,7 +3093,7 @@ export class AIPromptRunner {
|
|
|
3091
3093
|
/**
|
|
3092
3094
|
* Executes the model with retry logic for validation failures
|
|
3093
3095
|
*/
|
|
3094
|
-
async executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel) {
|
|
3096
|
+
async executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability) {
|
|
3095
3097
|
const validationAttempts = [];
|
|
3096
3098
|
const maxRetries = Math.max(0, prompt.MaxRetries || 0);
|
|
3097
3099
|
let lastError = null;
|
|
@@ -3111,7 +3113,8 @@ export class AIPromptRunner {
|
|
|
3111
3113
|
}
|
|
3112
3114
|
// Execute the AI model with failover support
|
|
3113
3115
|
const modelResult = await this.executeModelWithFailover(selectedModel, renderedPromptText, prompt, params, promptRun.VendorID, params.conversationMessages, params.templateMessageRole || 'system', params.cancellationToken, allCandidates, // Pass the candidates from initial selection
|
|
3114
|
-
promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel
|
|
3116
|
+
promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability // Reuse credential probes from selection
|
|
3117
|
+
);
|
|
3115
3118
|
// Check for fatal errors - don't attempt validation/retry on these
|
|
3116
3119
|
// Fatal errors (like ContextLengthExceeded when all models exhausted) cannot be resolved by retrying
|
|
3117
3120
|
if (!modelResult.success && modelResult.errorInfo?.severity === 'Fatal') {
|
|
@@ -3982,6 +3985,18 @@ export class AIPromptRunner {
|
|
|
3982
3985
|
* Updates the AIPromptRun entity with execution results
|
|
3983
3986
|
*/
|
|
3984
3987
|
async updatePromptRun(promptRun, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens) {
|
|
3988
|
+
// Fire-and-forget finalize UPDATE. The field mutations run INSIDE the post-INSERT task (after the
|
|
3989
|
+
// 'Running' INSERT + its finalizeSave reload land), so the reload can never revert them and the
|
|
3990
|
+
// chained UPDATE persists the finalized state — the "stuck at Running" race is structurally
|
|
3991
|
+
// impossible. The execution flow does NOT await the save.
|
|
3992
|
+
this._promptRunQueue.Update(promptRun, () => this.applyFinalizedPromptRunFields(promptRun, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens));
|
|
3993
|
+
}
|
|
3994
|
+
/**
|
|
3995
|
+
* Populates a prompt-run's finalized fields (result, tokens, cost, timing, rollups) from the model
|
|
3996
|
+
* result. Runs INSIDE the post-INSERT save task — see {@link updatePromptRun}. Errors here are
|
|
3997
|
+
* logged (non-fatal): the AIPromptRun is observability, not part of the prompt's success contract.
|
|
3998
|
+
*/
|
|
3999
|
+
applyFinalizedPromptRunFields(promptRun, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens) {
|
|
3985
4000
|
try {
|
|
3986
4001
|
promptRun.CompletedAt = endTime;
|
|
3987
4002
|
promptRun.ExecutionTimeMS = executionTimeMS;
|
|
@@ -4168,10 +4183,6 @@ export class AIPromptRunner {
|
|
|
4168
4183
|
if (promptRun.Cost !== undefined) {
|
|
4169
4184
|
promptRun.TotalCost = promptRun.Cost;
|
|
4170
4185
|
}
|
|
4171
|
-
// Finalize fire-and-forget. Chains after the initial INSERT (same entity instance) so the
|
|
4172
|
-
// 'Running' INSERT can never overwrite this finalized 'Completed'/'Failed' state. The
|
|
4173
|
-
// execution flow does NOT await this — see queuePromptRunSave / WaitForPendingPromptRunSaves.
|
|
4174
|
-
this.queuePromptRunSave(promptRun);
|
|
4175
4186
|
}
|
|
4176
4187
|
catch (error) {
|
|
4177
4188
|
this.logError(error, {
|