@memberjunction/ai-prompts 5.40.2 → 5.42.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -0
- package/dist/AIModelRunner.d.ts +24 -0
- package/dist/AIModelRunner.d.ts.map +1 -1
- package/dist/AIModelRunner.js +48 -7
- package/dist/AIModelRunner.js.map +1 -1
- package/dist/AIPromptRunner.d.ts +87 -67
- package/dist/AIPromptRunner.d.ts.map +1 -1
- package/dist/AIPromptRunner.js +440 -337
- package/dist/AIPromptRunner.js.map +1 -1
- package/dist/ExecutionPlanner.d.ts.map +1 -1
- package/dist/ExecutionPlanner.js +6 -14
- package/dist/ExecutionPlanner.js.map +1 -1
- package/dist/ParallelExecution.d.ts +33 -2
- package/dist/ParallelExecution.d.ts.map +1 -1
- package/dist/ParallelExecutionCoordinator.d.ts +26 -27
- package/dist/ParallelExecutionCoordinator.d.ts.map +1 -1
- package/dist/ParallelExecutionCoordinator.js +72 -97
- package/dist/ParallelExecutionCoordinator.js.map +1 -1
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -0
- package/dist/index.js.map +1 -1
- package/package.json +11 -11
package/dist/AIPromptRunner.js
CHANGED
|
@@ -6,7 +6,6 @@ import { CleanJSON, MJGlobal, JSONValidator, ValidationResult, ValidationErrorIn
|
|
|
6
6
|
import { CredentialEngine } from '@memberjunction/credentials';
|
|
7
7
|
import { TemplateEngineServer } from '@memberjunction/templates';
|
|
8
8
|
import { ExecutionPlanner } from './ExecutionPlanner.js';
|
|
9
|
-
import { ParallelExecutionCoordinator } from './ParallelExecutionCoordinator.js';
|
|
10
9
|
import { AIEngine } from '@memberjunction/aiengine';
|
|
11
10
|
import { AIEngineBase } from '@memberjunction/ai-engine-base';
|
|
12
11
|
import { SystemPlaceholderManager } from '@memberjunction/ai-core-plus';
|
|
@@ -26,6 +25,21 @@ function mimeFromBlockType(type) {
|
|
|
26
25
|
}
|
|
27
26
|
}
|
|
28
27
|
export class AIPromptRunner {
|
|
28
|
+
/**
|
|
29
|
+
* Process-wide cache of parsed `OutputExample` JSON, keyed by the raw example string.
|
|
30
|
+
* A prompt's OutputExample is a static string reused across every run and every validation
|
|
31
|
+
* retry, so re-parsing it each time is pure waste. Keyed by content (not prompt ID) so two
|
|
32
|
+
* prompts sharing an identical example share one parsed entry and an edited example never
|
|
33
|
+
* serves a stale parse. Stores `{ parsed }` on success or `{ error }` on failure so we cache
|
|
34
|
+
* the failure too rather than re-throwing-and-reparsing bad JSON every attempt.
|
|
35
|
+
*/
|
|
36
|
+
static { this._outputExampleCache = new Map(); }
|
|
37
|
+
/**
|
|
38
|
+
* Marker used in `AIModelSelectionInfo.modelsConsidered[].unavailableReason` for candidates
|
|
39
|
+
* that were intentionally NOT credential-checked because a higher-priority candidate had
|
|
40
|
+
* already been selected. See the DECISION note in {@link selectModelWithAPIKeyTracked}.
|
|
41
|
+
*/
|
|
42
|
+
static { this.NOT_EVALUATED_REASON = 'Not evaluated (a higher-priority candidate was already selected; set AIPromptParams.forceFullModelEvaluation to probe all)'; }
|
|
29
43
|
/**
|
|
30
44
|
* Optional metadata provider override. Callers should set
|
|
31
45
|
* `instance.Provider = providerToUse` before invoking run methods
|
|
@@ -39,13 +53,49 @@ export class AIPromptRunner {
|
|
|
39
53
|
}
|
|
40
54
|
constructor() {
|
|
41
55
|
this._provider = null;
|
|
56
|
+
/**
|
|
57
|
+
* Instance-keyed chain of in-flight AIPromptRun saves. Mirrors the BaseAgent step-save pattern:
|
|
58
|
+
* prompt-run persistence is fire-and-forget so the execution path never blocks on a DB
|
|
59
|
+
* round-trip, but saves for the SAME entity are sequenced — the initial 'Running' INSERT always
|
|
60
|
+
* completes before the finalize UPDATE, so a slow INSERT can never clobber the finalized row.
|
|
61
|
+
* Keyed by the entity INSTANCE (stable), not its ID. See {@link queuePromptRunSave}.
|
|
62
|
+
*/
|
|
63
|
+
this._promptRunSaveChains = new Map();
|
|
64
|
+
/** All queued prompt-run save promises, for optional flushing via {@link WaitForPendingPromptRunSaves}. */
|
|
65
|
+
this._pendingPromptRunSaves = [];
|
|
42
66
|
this._metadata = this._provider ?? new Metadata();
|
|
43
67
|
this._templateEngine = TemplateEngineServer.Instance;
|
|
44
68
|
this._executionPlanner = new ExecutionPlanner();
|
|
45
|
-
this._parallelCoordinator = new ParallelExecutionCoordinator();
|
|
46
69
|
this._jsonValidator = new JSONValidator();
|
|
47
70
|
this._modelRunner = new AIModelRunner();
|
|
48
71
|
}
|
|
72
|
+
/** ClassFactory key the parallel coordinator self-registers under (see {@link ParallelCoordinator}). */
|
|
73
|
+
static { this.PARALLEL_COORDINATOR_KEY = 'ParallelExecutionCoordinator'; }
|
|
74
|
+
/**
|
|
75
|
+
* Lazily resolves the parallel execution coordinator.
|
|
76
|
+
*
|
|
77
|
+
* The coordinator is a SUBCLASS of AIPromptRunner — it inherits {@link executeModel} and the rest
|
|
78
|
+
* of the battle-tested execution path so there is a single source of truth for credential / driver
|
|
79
|
+
* / ChatParams / streaming resolution. That subclass relationship means the base cannot statically
|
|
80
|
+
* `new` it without a hard circular import, so we resolve it through the ClassFactory instead (the
|
|
81
|
+
* coordinator self-registers via `@RegisterClass(AIPromptRunner, PARALLEL_COORDINATOR_KEY)`).
|
|
82
|
+
* Created once per runner and reused; the runner's Provider override is propagated. Throws a clear
|
|
83
|
+
* error if the coordinator class was never loaded/registered — otherwise ClassFactory would silently
|
|
84
|
+
* fall back to a plain AIPromptRunner that lacks the parallel methods.
|
|
85
|
+
*/
|
|
86
|
+
get ParallelCoordinator() {
|
|
87
|
+
if (!this._parallelCoordinator) {
|
|
88
|
+
const instance = MJGlobal.Instance.ClassFactory.CreateInstance(AIPromptRunner, AIPromptRunner.PARALLEL_COORDINATOR_KEY);
|
|
89
|
+
if (!instance || typeof instance.executeTasksInParallel !== 'function') {
|
|
90
|
+
throw new Error(`ParallelExecutionCoordinator is not registered with the ClassFactory. Ensure ` +
|
|
91
|
+
`'@memberjunction/ai-prompts' is fully loaded (it is exported from the package index and ` +
|
|
92
|
+
`picked up by the class-registration manifest).`);
|
|
93
|
+
}
|
|
94
|
+
instance.Provider = this.Provider;
|
|
95
|
+
this._parallelCoordinator = instance;
|
|
96
|
+
}
|
|
97
|
+
return this._parallelCoordinator;
|
|
98
|
+
}
|
|
49
99
|
/**
|
|
50
100
|
* Access the underlying AIModelRunner for embedding and other non-LLM model calls.
|
|
51
101
|
* Use this when you need tracked embedding execution with AIPromptRun record creation.
|
|
@@ -116,19 +166,15 @@ export class AIPromptRunner {
|
|
|
116
166
|
});
|
|
117
167
|
}
|
|
118
168
|
/**
|
|
119
|
-
* Checks if a model vendor is configured as an inference provider
|
|
169
|
+
* Checks if a model vendor is configured as an inference provider.
|
|
170
|
+
* Delegates to the memoized {@link AIEngine.IsInferenceProvider} helper so the
|
|
171
|
+
* "Inference Provider" vendor-type lookup happens once per engine load rather than on
|
|
172
|
+
* every candidate in every selection pass.
|
|
120
173
|
* @param modelVendor The model vendor to check
|
|
121
174
|
* @returns true if the vendor is an inference provider
|
|
122
175
|
*/
|
|
123
176
|
isInferenceProvider(modelVendor) {
|
|
124
|
-
|
|
125
|
-
const inferenceProviderType = AIEngine.Instance.VendorTypeDefinitions.find(vt => vt.Name === 'Inference Provider');
|
|
126
|
-
if (!inferenceProviderType) {
|
|
127
|
-
// Fallback to checking if it's not a model developer (should rarely happen)
|
|
128
|
-
const modelDeveloperType = AIEngine.Instance.VendorTypeDefinitions.find(vt => vt.Name === 'Model Developer');
|
|
129
|
-
return !UUIDsEqual(modelVendor.TypeID, modelDeveloperType?.ID);
|
|
130
|
-
}
|
|
131
|
-
return UUIDsEqual(modelVendor.TypeID, inferenceProviderType.ID);
|
|
177
|
+
return AIEngine.Instance.IsInferenceProvider(modelVendor);
|
|
132
178
|
}
|
|
133
179
|
/**
|
|
134
180
|
* Resolves credentials for AI model execution using a hierarchical resolution system.
|
|
@@ -171,7 +217,8 @@ export class AIPromptRunner {
|
|
|
171
217
|
}
|
|
172
218
|
// Priority 3: ModelVendor bindings - with failover
|
|
173
219
|
if (modelId && vendorId) {
|
|
174
|
-
const modelVendor = AIEngine.Instance.
|
|
220
|
+
const modelVendor = AIEngine.Instance.ModelVendorsByModelID.get(NormalizeUUID(modelId))
|
|
221
|
+
?.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
|
|
175
222
|
if (modelVendor) {
|
|
176
223
|
const bindings = AIEngine.Instance.GetCredentialBindingsForTarget('ModelVendor', modelVendor.ID);
|
|
177
224
|
const result = await this.tryCredentialBindingsWithFailover(bindings, 'AICredentialBinding(ModelVendor)', params, verbose);
|
|
@@ -189,7 +236,7 @@ export class AIPromptRunner {
|
|
|
189
236
|
// Priority 5: Type-based default credential
|
|
190
237
|
// If the vendor declares a CredentialTypeID, try to find a default credential of that type
|
|
191
238
|
if (vendorId) {
|
|
192
|
-
const vendor = AIEngine.Instance.
|
|
239
|
+
const vendor = AIEngine.Instance.VendorsByID.get(NormalizeUUID(vendorId));
|
|
193
240
|
if (vendor?.CredentialTypeID) {
|
|
194
241
|
const defaultCredential = this.findDefaultCredentialByType(vendor.CredentialTypeID);
|
|
195
242
|
if (defaultCredential) {
|
|
@@ -348,7 +395,8 @@ export class AIPromptRunner {
|
|
|
348
395
|
}
|
|
349
396
|
// Priority 3: ModelVendor bindings
|
|
350
397
|
if (modelId && vendorId) {
|
|
351
|
-
const modelVendor = AIEngine.Instance.
|
|
398
|
+
const modelVendor = AIEngine.Instance.ModelVendorsByModelID.get(NormalizeUUID(modelId))
|
|
399
|
+
?.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
|
|
352
400
|
if (modelVendor && AIEngine.Instance.HasCredentialBindings('ModelVendor', modelVendor.ID)) {
|
|
353
401
|
return true;
|
|
354
402
|
}
|
|
@@ -361,7 +409,7 @@ export class AIPromptRunner {
|
|
|
361
409
|
}
|
|
362
410
|
// Priority 5: Type-based default credential
|
|
363
411
|
if (vendorId) {
|
|
364
|
-
const vendor = AIEngine.Instance.
|
|
412
|
+
const vendor = AIEngine.Instance.VendorsByID.get(NormalizeUUID(vendorId));
|
|
365
413
|
if (vendor?.CredentialTypeID) {
|
|
366
414
|
const defaultCredential = this.findDefaultCredentialByType(vendor.CredentialTypeID);
|
|
367
415
|
if (defaultCredential) {
|
|
@@ -434,25 +482,20 @@ export class AIPromptRunner {
|
|
|
434
482
|
let renderedPromptText = '';
|
|
435
483
|
// For hierarchical prompts, we need to create the parent prompt run first to get its ID
|
|
436
484
|
let parentPromptRun;
|
|
437
|
-
let selectedModel;
|
|
438
485
|
let childTemplateRenderingResult;
|
|
439
|
-
let
|
|
486
|
+
let selection;
|
|
440
487
|
// Handle different prompt execution modes
|
|
441
488
|
if (params.childPrompts && params.childPrompts.length > 0) {
|
|
442
489
|
// Hierarchical template composition mode - render child templates first, then compose
|
|
443
|
-
//this.logStatus(`🌳 Composing prompt with ${params.childPrompts.length} child templates in hierarchical mode`, true, params);
|
|
444
490
|
// Determine which prompt to use for model selection
|
|
445
491
|
let modelSelectionPrompt = prompt;
|
|
446
492
|
if (params.modelSelectionPrompt) {
|
|
447
493
|
modelSelectionPrompt = params.modelSelectionPrompt;
|
|
448
|
-
//this.logStatus(`🎯 Using prompt "${modelSelectionPrompt.Name}" for model selection instead of parent prompt`, true, params);
|
|
449
494
|
}
|
|
450
|
-
// Select model using the appropriate prompt
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
if (!selectedModel) {
|
|
455
|
-
throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, modelSelectionInfo));
|
|
495
|
+
// Select model using the appropriate prompt — capture the FULL result
|
|
496
|
+
selection = await this.selectModel(modelSelectionPrompt, params.override?.modelId, params.contextUser, params.configurationId, params.override?.vendorId, params);
|
|
497
|
+
if (!selection.model) {
|
|
498
|
+
throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, selection.selectionInfo));
|
|
456
499
|
}
|
|
457
500
|
// Check if we have a system prompt override
|
|
458
501
|
if (params.systemPromptOverride) {
|
|
@@ -467,7 +510,7 @@ export class AIPromptRunner {
|
|
|
467
510
|
renderedPromptText = await this.renderPromptWithChildTemplates(prompt, params, childTemplateRenderingResult.renderedTemplates);
|
|
468
511
|
}
|
|
469
512
|
// Create parent prompt run for the final composed prompt execution
|
|
470
|
-
parentPromptRun = await this.createPromptRun(prompt,
|
|
513
|
+
parentPromptRun = await this.createPromptRun(prompt, selection.model, params, renderedPromptText, startTime, params.override?.vendorId, selection.selectionInfo);
|
|
471
514
|
}
|
|
472
515
|
else if (prompt.TemplateID && (!params.conversationMessages || params.templateMessageRole !== 'none')) {
|
|
473
516
|
// Check if we have a system prompt override
|
|
@@ -497,30 +540,28 @@ export class AIPromptRunner {
|
|
|
497
540
|
if (params.cancellationToken?.aborted) {
|
|
498
541
|
throw new Error('Prompt execution was cancelled during template rendering');
|
|
499
542
|
}
|
|
500
|
-
// If no model was selected yet (
|
|
501
|
-
if (!
|
|
543
|
+
// If no model was selected yet (non-hierarchical case), select one now — capture the FULL result
|
|
544
|
+
if (!selection?.model) {
|
|
502
545
|
let modelSelectionPrompt = prompt;
|
|
503
546
|
if (params.modelSelectionPrompt) {
|
|
504
547
|
modelSelectionPrompt = params.modelSelectionPrompt;
|
|
505
548
|
this.logStatus(`🎯 Using prompt "${modelSelectionPrompt.Name}" for model selection instead of main prompt`, true, params);
|
|
506
549
|
}
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
if (!selectedModel) {
|
|
511
|
-
throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, modelSelectionInfo));
|
|
550
|
+
selection = await this.selectModel(modelSelectionPrompt, params.override?.modelId, params.contextUser, params.configurationId, params.override?.vendorId, params);
|
|
551
|
+
if (!selection.model) {
|
|
552
|
+
throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, selection.selectionInfo));
|
|
512
553
|
}
|
|
513
554
|
}
|
|
514
555
|
// Check if we need parallel execution based on ParallelizationMode
|
|
515
556
|
const shouldUseParallelExecution = prompt.ParallelizationMode && prompt.ParallelizationMode !== 'None';
|
|
516
557
|
let result;
|
|
517
558
|
if (shouldUseParallelExecution) {
|
|
518
|
-
// Use parallel execution path
|
|
519
|
-
result = await this.executePromptInParallel(prompt, renderedPromptText, params, startTime, parentPromptRun,
|
|
559
|
+
// Use parallel execution path — pass full selection through
|
|
560
|
+
result = await this.executePromptInParallel(prompt, renderedPromptText, params, startTime, parentPromptRun, selection);
|
|
520
561
|
}
|
|
521
562
|
else {
|
|
522
|
-
// Use traditional single execution path
|
|
523
|
-
result = await this.executeSinglePrompt(prompt, renderedPromptText, params, startTime, parentPromptRun,
|
|
563
|
+
// Use traditional single execution path — pass full selection through
|
|
564
|
+
result = await this.executeSinglePrompt(prompt, renderedPromptText, params, startTime, parentPromptRun, selection);
|
|
524
565
|
}
|
|
525
566
|
// Note: With template composition, we only execute once so no rollup calculations needed
|
|
526
567
|
// The final composed prompt is executed as a single operation
|
|
@@ -590,33 +631,22 @@ export class AIPromptRunner {
|
|
|
590
631
|
* @param startTime - Execution start time
|
|
591
632
|
* @returns Promise<AIPromptRunResult<T>> - The execution result
|
|
592
633
|
*/
|
|
593
|
-
async executeSinglePrompt(prompt, renderedPromptText, params, startTime, existingPromptRun,
|
|
634
|
+
async executeSinglePrompt(prompt, renderedPromptText, params, startTime, existingPromptRun, existingSelection) {
|
|
594
635
|
// Check for cancellation before model selection
|
|
595
636
|
if (params.cancellationToken?.aborted) {
|
|
596
637
|
throw new Error('Prompt execution was cancelled before model selection');
|
|
597
638
|
}
|
|
598
|
-
// Use existing
|
|
599
|
-
let selectedModel =
|
|
600
|
-
let modelSelectionInfo =
|
|
601
|
-
let vendorDriverClass;
|
|
602
|
-
let vendorApiName;
|
|
603
|
-
let vendorSupportsEffortLevel;
|
|
604
|
-
let modelEffortLevel;
|
|
605
|
-
let allCandidates = [];
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
const modelID = modelSelectionInfo.modelSelected.ID;
|
|
610
|
-
const modelVendor = AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, vendorID) &&
|
|
611
|
-
UUIDsEqual(mv.ModelID, modelID));
|
|
612
|
-
if (modelVendor) {
|
|
613
|
-
vendorDriverClass = modelVendor.DriverClass;
|
|
614
|
-
vendorApiName = modelVendor.APIName;
|
|
615
|
-
vendorSupportsEffortLevel = modelVendor.SupportsEffortLevel;
|
|
616
|
-
}
|
|
617
|
-
// Extract valid candidates from selection info for retry logic
|
|
618
|
-
allCandidates = this.buildCandidatesFromSelectionInfo(modelSelectionInfo);
|
|
619
|
-
}
|
|
639
|
+
// Use existing selection if provided (hierarchical case) or select now
|
|
640
|
+
let selectedModel = existingSelection?.model ?? undefined;
|
|
641
|
+
let modelSelectionInfo = existingSelection?.selectionInfo;
|
|
642
|
+
let vendorDriverClass = existingSelection?.vendorDriverClass;
|
|
643
|
+
let vendorApiName = existingSelection?.vendorApiName;
|
|
644
|
+
let vendorSupportsEffortLevel = existingSelection?.vendorSupportsEffortLevel;
|
|
645
|
+
let modelEffortLevel = existingSelection?.modelEffortLevel;
|
|
646
|
+
let allCandidates = existingSelection?.allCandidates ?? [];
|
|
647
|
+
// Credential probes already done during selection — reused by failover so it doesn't
|
|
648
|
+
// recompute hasCredentialsAvailable for the prefix it walks before the selected candidate.
|
|
649
|
+
let credentialAvailability = existingSelection?.credentialAvailability;
|
|
620
650
|
if (!selectedModel) {
|
|
621
651
|
// Determine which prompt to use for model selection
|
|
622
652
|
let modelSelectionPrompt = prompt;
|
|
@@ -632,6 +662,7 @@ export class AIPromptRunner {
|
|
|
632
662
|
modelEffortLevel = modelResult.modelEffortLevel;
|
|
633
663
|
modelSelectionInfo = modelResult.selectionInfo;
|
|
634
664
|
allCandidates = modelResult.allCandidates || [];
|
|
665
|
+
credentialAvailability = modelResult.credentialAvailability;
|
|
635
666
|
if (!selectedModel) {
|
|
636
667
|
throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, modelSelectionInfo));
|
|
637
668
|
}
|
|
@@ -647,7 +678,8 @@ export class AIPromptRunner {
|
|
|
647
678
|
throw new Error('Prompt execution was cancelled before model execution');
|
|
648
679
|
}
|
|
649
680
|
// Execute with retry logic for validation failures
|
|
650
|
-
const { modelResult, parsedResult, validationAttempts, cumulativeTokens } = await this.executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel // Pass model-specific effort level
|
|
681
|
+
const { modelResult, parsedResult, validationAttempts, cumulativeTokens } = await this.executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, // Pass model-specific effort level
|
|
682
|
+
credentialAvailability // Reuse credential probes from selection
|
|
651
683
|
);
|
|
652
684
|
// Calculate execution metrics
|
|
653
685
|
const endTime = new Date();
|
|
@@ -707,7 +739,7 @@ export class AIPromptRunner {
|
|
|
707
739
|
* @param startTime - Execution start time
|
|
708
740
|
* @returns Promise<AIPromptRunResult<T>> - The aggregated execution result
|
|
709
741
|
*/
|
|
710
|
-
async executePromptInParallel(prompt, renderedPromptText, params, startTime, existingPromptRun,
|
|
742
|
+
async executePromptInParallel(prompt, renderedPromptText, params, startTime, existingPromptRun, existingSelection) {
|
|
711
743
|
// Check for cancellation before starting parallel execution
|
|
712
744
|
if (params.cancellationToken?.aborted) {
|
|
713
745
|
throw new Error('Parallel execution was cancelled before starting');
|
|
@@ -715,21 +747,21 @@ export class AIPromptRunner {
|
|
|
715
747
|
// Load AI Engine to get models and prompt models
|
|
716
748
|
await AIEngine.Instance.Config(false, params.contextUser);
|
|
717
749
|
let executionTasks;
|
|
718
|
-
// If a model is already selected (from hierarchical template composition),
|
|
750
|
+
// If a model is already selected (from hierarchical template composition),
|
|
719
751
|
// create a single task with that model instead of using the planner
|
|
720
|
-
if (
|
|
752
|
+
if (existingSelection?.model) {
|
|
721
753
|
// Create a single execution task with the pre-selected model
|
|
722
754
|
executionTasks = [{
|
|
723
755
|
taskId: 'pre-selected',
|
|
724
|
-
model:
|
|
725
|
-
vendorDriverClass:
|
|
726
|
-
vendorApiName:
|
|
756
|
+
model: existingSelection.model,
|
|
757
|
+
vendorDriverClass: existingSelection.vendorDriverClass,
|
|
758
|
+
vendorApiName: existingSelection.vendorApiName,
|
|
727
759
|
messages: params.conversationMessages || [],
|
|
728
760
|
promptText: renderedPromptText,
|
|
729
761
|
templateMessageRole: params.templateMessageRole || 'system',
|
|
730
762
|
contextUser: params.contextUser
|
|
731
763
|
}];
|
|
732
|
-
this.logStatus(` Using pre-selected model "${
|
|
764
|
+
this.logStatus(` Using pre-selected model "${existingSelection.model.Name}" for parallel execution`, true, params);
|
|
733
765
|
}
|
|
734
766
|
else {
|
|
735
767
|
// Normal parallel execution path - let the planner decide
|
|
@@ -754,7 +786,7 @@ export class AIPromptRunner {
|
|
|
754
786
|
throw new Error('Parallel execution was cancelled before task execution');
|
|
755
787
|
}
|
|
756
788
|
// Execute tasks in parallel
|
|
757
|
-
const parallelResult = await this.
|
|
789
|
+
const parallelResult = await this.ParallelCoordinator.executeTasksInParallel(params, executionTasks, undefined, undefined, params.cancellationToken, undefined, params.agentRunId);
|
|
758
790
|
if (!parallelResult.success) {
|
|
759
791
|
throw new Error(`Parallel execution failed: ${parallelResult.errors.join(', ')}`);
|
|
760
792
|
}
|
|
@@ -770,7 +802,7 @@ export class AIPromptRunner {
|
|
|
770
802
|
method: 'PromptSelector',
|
|
771
803
|
selectorPromptId: prompt.ResultSelectorPromptID,
|
|
772
804
|
};
|
|
773
|
-
const aiSelectedResult = await this.
|
|
805
|
+
const aiSelectedResult = await this.ParallelCoordinator.selectBestResult(successfulResults, selectionConfig, undefined, params.cancellationToken);
|
|
774
806
|
if (aiSelectedResult) {
|
|
775
807
|
selectedResult = aiSelectedResult;
|
|
776
808
|
}
|
|
@@ -799,7 +831,7 @@ export class AIPromptRunner {
|
|
|
799
831
|
}
|
|
800
832
|
// Use existing prompt run if provided (hierarchical case) or create new one
|
|
801
833
|
// Use the model selection info if provided (from hierarchical execution)
|
|
802
|
-
const consolidatedPromptRun = existingPromptRun || await this.createPromptRun(prompt, selectedResult.task.model, params, renderedPromptText, startTime, params.override?.vendorId,
|
|
834
|
+
const consolidatedPromptRun = existingPromptRun || await this.createPromptRun(prompt, selectedResult.task.model, params, renderedPromptText, startTime, params.override?.vendorId, existingSelection?.selectionInfo);
|
|
803
835
|
// Update with parallel execution metadata
|
|
804
836
|
const endTime = new Date();
|
|
805
837
|
consolidatedPromptRun.CompletedAt = endTime;
|
|
@@ -852,18 +884,8 @@ export class AIPromptRunner {
|
|
|
852
884
|
// Set Status and WasSelectedResult for parallel execution
|
|
853
885
|
consolidatedPromptRun.Status = parallelResult.successCount > 0 ? 'Completed' : 'Failed';
|
|
854
886
|
consolidatedPromptRun.WasSelectedResult = true; // This is the consolidated result chosen by judge
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
this.logError(`Failed to save consolidated AIPromptRun: ${consolidatedPromptRun.LatestResult?.CompleteMessage || 'Unknown error'}`, {
|
|
858
|
-
category: 'ConsolidatedPromptRunSave',
|
|
859
|
-
metadata: {
|
|
860
|
-
promptRunId: consolidatedPromptRun.ID,
|
|
861
|
-
executionTasks: executionTasks.length,
|
|
862
|
-
successfulResults: successfulResults.length
|
|
863
|
-
},
|
|
864
|
-
maxErrorLength: params.maxErrorLength
|
|
865
|
-
});
|
|
866
|
-
}
|
|
887
|
+
// Persist the consolidated run fire-and-forget; chains after its INSERT via the save queue.
|
|
888
|
+
this.queuePromptRunSave(consolidatedPromptRun);
|
|
867
889
|
// Create additional results from all other successful results (excluding the best one)
|
|
868
890
|
const additionalResults = [];
|
|
869
891
|
// Sort successful results by ranking (if available) or keep original order
|
|
@@ -931,11 +953,11 @@ export class AIPromptRunner {
|
|
|
931
953
|
modelInfo: {
|
|
932
954
|
modelId: selectedResult.task.model.ID,
|
|
933
955
|
modelName: selectedResult.task.model.Name,
|
|
934
|
-
vendorId:
|
|
956
|
+
vendorId: existingSelection?.selectionInfo?.vendorSelected?.ID, // VendorID not directly available on AIModel
|
|
935
957
|
vendorName: selectedResult.task.model.Vendor,
|
|
936
958
|
},
|
|
937
959
|
judgeMetadata: selectedResult.judgeMetadata,
|
|
938
|
-
modelSelectionInfo:
|
|
960
|
+
modelSelectionInfo: existingSelection?.selectionInfo, // Include model selection info if provided
|
|
939
961
|
};
|
|
940
962
|
}
|
|
941
963
|
/**
|
|
@@ -1225,7 +1247,7 @@ export class AIPromptRunner {
|
|
|
1225
1247
|
}
|
|
1226
1248
|
// Get configuration info if provided
|
|
1227
1249
|
if (configurationId) {
|
|
1228
|
-
configuration = AIEngine.Instance.
|
|
1250
|
+
configuration = AIEngine.Instance.ConfigurationsByID.get(NormalizeUUID(configurationId));
|
|
1229
1251
|
configurationName = configuration?.Name;
|
|
1230
1252
|
}
|
|
1231
1253
|
// Build unified list of model-vendor candidates
|
|
@@ -1262,7 +1284,7 @@ export class AIPromptRunner {
|
|
|
1262
1284
|
// });
|
|
1263
1285
|
// }
|
|
1264
1286
|
// Select the first candidate with available credentials and track all attempts
|
|
1265
|
-
const { selected, consideredModels } = await this.selectModelWithAPIKeyTracked(candidates, prompt.ID, params);
|
|
1287
|
+
const { selected, consideredModels, credentialAvailability } = await this.selectModelWithAPIKeyTracked(candidates, prompt.ID, params);
|
|
1266
1288
|
// Merge considered models into our tracking
|
|
1267
1289
|
modelsConsidered.push(...consideredModels);
|
|
1268
1290
|
if (!selected) {
|
|
@@ -1274,6 +1296,7 @@ export class AIPromptRunner {
|
|
|
1274
1296
|
vendorSupportsEffortLevel: undefined,
|
|
1275
1297
|
modelEffortLevel: undefined,
|
|
1276
1298
|
allCandidates: candidates,
|
|
1299
|
+
credentialAvailability,
|
|
1277
1300
|
selectionInfo: this.createSelectionInfo({
|
|
1278
1301
|
aiConfiguration: configuration,
|
|
1279
1302
|
modelsConsidered,
|
|
@@ -1309,7 +1332,7 @@ export class AIPromptRunner {
|
|
|
1309
1332
|
// Get selected vendor entity
|
|
1310
1333
|
let selectedVendor;
|
|
1311
1334
|
if (selected.vendorId) {
|
|
1312
|
-
selectedVendor = AIEngine.Instance.
|
|
1335
|
+
selectedVendor = AIEngine.Instance.VendorsByID.get(NormalizeUUID(selected.vendorId));
|
|
1313
1336
|
}
|
|
1314
1337
|
return {
|
|
1315
1338
|
model: selected.model,
|
|
@@ -1318,6 +1341,7 @@ export class AIPromptRunner {
|
|
|
1318
1341
|
vendorSupportsEffortLevel: selected.supportsEffortLevel,
|
|
1319
1342
|
modelEffortLevel: selected.effortLevel, // Pass through model-specific effort level
|
|
1320
1343
|
allCandidates: candidates,
|
|
1344
|
+
credentialAvailability,
|
|
1321
1345
|
selectionInfo: this.createSelectionInfo({
|
|
1322
1346
|
aiConfiguration: configuration,
|
|
1323
1347
|
modelsConsidered,
|
|
@@ -1385,7 +1409,7 @@ export class AIPromptRunner {
|
|
|
1385
1409
|
* Returns candidates for the single model if it's active and compatible.
|
|
1386
1410
|
*/
|
|
1387
1411
|
buildCandidatesForExplicitModel(explicitModelId, prompt, preferredVendorId) {
|
|
1388
|
-
const model = AIEngine.Instance.
|
|
1412
|
+
const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(explicitModelId));
|
|
1389
1413
|
if (!model || !model.IsActive) {
|
|
1390
1414
|
return [];
|
|
1391
1415
|
}
|
|
@@ -1480,7 +1504,7 @@ export class AIPromptRunner {
|
|
|
1480
1504
|
return 0;
|
|
1481
1505
|
const modelsWithPower = promptModels
|
|
1482
1506
|
.map(pm => {
|
|
1483
|
-
const model = AIEngine.Instance.
|
|
1507
|
+
const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
|
|
1484
1508
|
return { powerRank: model?.PowerRank ?? 0, priority: pm.Priority || 1 };
|
|
1485
1509
|
});
|
|
1486
1510
|
const totalWeight = modelsWithPower.reduce((sum, m) => sum + m.priority, 0);
|
|
@@ -1510,7 +1534,7 @@ export class AIPromptRunner {
|
|
|
1510
1534
|
*/
|
|
1511
1535
|
buildCandidatesForGeneralSelection(prompt, configurationId, preferredVendorId, verbose) {
|
|
1512
1536
|
const preferredVendorName = preferredVendorId ?
|
|
1513
|
-
AIEngine.Instance.
|
|
1537
|
+
AIEngine.Instance.VendorsByID.get(NormalizeUUID(preferredVendorId))?.Name : undefined;
|
|
1514
1538
|
// Get prompt models for configuration
|
|
1515
1539
|
const promptModels = this.getPromptModelsForConfiguration(prompt, configurationId);
|
|
1516
1540
|
const candidates = [];
|
|
@@ -1582,7 +1606,7 @@ export class AIPromptRunner {
|
|
|
1582
1606
|
const pm = promptModels[i];
|
|
1583
1607
|
// Compute priority as inverse of array position so highest-priority (first) gets the largest number
|
|
1584
1608
|
const computedPriority = promptModels.length - i;
|
|
1585
|
-
const model = AIEngine.Instance.
|
|
1609
|
+
const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
|
|
1586
1610
|
if (!model || !model.IsActive)
|
|
1587
1611
|
continue;
|
|
1588
1612
|
if (pm.VendorID) {
|
|
@@ -1604,8 +1628,9 @@ export class AIPromptRunner {
|
|
|
1604
1628
|
* Helper: Create candidate for specific vendor from AIPromptModel.
|
|
1605
1629
|
*/
|
|
1606
1630
|
createCandidateForSpecificVendor(model, promptModel, computedPriority = 0) {
|
|
1607
|
-
|
|
1608
|
-
|
|
1631
|
+
// Use the model's precomputed ModelVendors (grouped at engine load) instead of scanning
|
|
1632
|
+
// the global ModelVendors array — model.ID === promptModel.ModelID here.
|
|
1633
|
+
const modelVendor = model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, promptModel.VendorID) &&
|
|
1609
1634
|
mv.Status === 'Active' &&
|
|
1610
1635
|
this.isInferenceProvider(mv));
|
|
1611
1636
|
if (!modelVendor)
|
|
@@ -1627,9 +1652,8 @@ export class AIPromptRunner {
|
|
|
1627
1652
|
* Helper: Create candidates for all vendors of a model, sorted by vendor priority.
|
|
1628
1653
|
*/
|
|
1629
1654
|
createCandidatesForAllVendors(model, computedPriority = 0) {
|
|
1630
|
-
const vendors =
|
|
1631
|
-
.filter(mv =>
|
|
1632
|
-
mv.Status === 'Active' &&
|
|
1655
|
+
const vendors = model.ModelVendors
|
|
1656
|
+
.filter(mv => mv.Status === 'Active' &&
|
|
1633
1657
|
this.isInferenceProvider(mv))
|
|
1634
1658
|
.sort((a, b) => (b.Priority || 0) - (a.Priority || 0));
|
|
1635
1659
|
const candidates = [];
|
|
@@ -1691,7 +1715,7 @@ export class AIPromptRunner {
|
|
|
1691
1715
|
*/
|
|
1692
1716
|
addPromptSpecificCandidates(candidates, promptModels, preferredVendorId) {
|
|
1693
1717
|
for (const pm of promptModels) {
|
|
1694
|
-
const model = AIEngine.Instance.
|
|
1718
|
+
const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
|
|
1695
1719
|
if (model && model.IsActive) {
|
|
1696
1720
|
const modelCandidates = this.createCandidatesForModel(model, 5000, 'prompt-model', preferredVendorId, pm.Priority);
|
|
1697
1721
|
candidates.push(...modelCandidates);
|
|
@@ -1714,7 +1738,7 @@ export class AIPromptRunner {
|
|
|
1714
1738
|
LogStatus(`Adding ${parentModels.length} models from parent config "${parentConfig.Name}" as fallback`);
|
|
1715
1739
|
}
|
|
1716
1740
|
for (const pm of parentModels) {
|
|
1717
|
-
const model = AIEngine.Instance.
|
|
1741
|
+
const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
|
|
1718
1742
|
if (model && model.IsActive) {
|
|
1719
1743
|
// Decrease base priority for each level up the chain (3000, 2500, 2000, etc.)
|
|
1720
1744
|
const basePriority = 3000 - (i * 500);
|
|
@@ -1731,7 +1755,7 @@ export class AIPromptRunner {
|
|
|
1731
1755
|
LogStatus(`Adding ${nullConfigModels.length} NULL configuration models as universal fallback`);
|
|
1732
1756
|
}
|
|
1733
1757
|
for (const pm of nullConfigModels) {
|
|
1734
|
-
const model = AIEngine.Instance.
|
|
1758
|
+
const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
|
|
1735
1759
|
if (model && model.IsActive) {
|
|
1736
1760
|
const modelCandidates = this.createCandidatesForModel(model, 1000, // Lowest priority tier
|
|
1737
1761
|
'prompt-model', preferredVendorId, pm.Priority);
|
|
@@ -1759,8 +1783,7 @@ export class AIPromptRunner {
|
|
|
1759
1783
|
return AIEngine.Instance.Models.filter(m => m.IsActive &&
|
|
1760
1784
|
(!prompt.AIModelTypeID || UUIDsEqual(m.AIModelTypeID, prompt.AIModelTypeID)) &&
|
|
1761
1785
|
(!preferredVendorName ||
|
|
1762
|
-
|
|
1763
|
-
mv.Status === 'Active' &&
|
|
1786
|
+
m.ModelVendors.some(mv => mv.Status === 'Active' &&
|
|
1764
1787
|
mv.Vendor === preferredVendorName &&
|
|
1765
1788
|
this.isInferenceProvider(mv))));
|
|
1766
1789
|
}
|
|
@@ -1801,9 +1824,11 @@ export class AIPromptRunner {
|
|
|
1801
1824
|
*/
|
|
1802
1825
|
createCandidatesForModel(model, basePriority, source, preferredVendorId, promptModelPriority) {
|
|
1803
1826
|
const modelCandidates = [];
|
|
1804
|
-
// Get all vendors for this model - filter for inference providers only
|
|
1805
|
-
|
|
1806
|
-
|
|
1827
|
+
// Get all vendors for this model - filter for inference providers only.
|
|
1828
|
+
// Uses the model's precomputed ModelVendors (grouped at engine load) rather than scanning
|
|
1829
|
+
// the global ModelVendors array.
|
|
1830
|
+
const modelVendors = model.ModelVendors
|
|
1831
|
+
.filter(mv => mv.Status === 'Active' && this.isInferenceProvider(mv))
|
|
1807
1832
|
.sort((a, b) => b.Priority - a.Priority);
|
|
1808
1833
|
// First, add preferred vendor if it exists
|
|
1809
1834
|
if (preferredVendorId) {
|
|
@@ -1865,34 +1890,6 @@ export class AIPromptRunner {
|
|
|
1865
1890
|
Object.assign(info, data);
|
|
1866
1891
|
return info;
|
|
1867
1892
|
}
|
|
1868
|
-
/**
|
|
1869
|
-
* Converts model selection info into ModelVendorCandidate array for retry logic.
|
|
1870
|
-
* Extracts only the valid candidates (those with available API keys) from the selection info.
|
|
1871
|
-
*
|
|
1872
|
-
* @param selectionInfo - Model selection information containing considered models
|
|
1873
|
-
* @returns Array of valid model-vendor candidates sorted by priority
|
|
1874
|
-
*/
|
|
1875
|
-
buildCandidatesFromSelectionInfo(selectionInfo) {
|
|
1876
|
-
const validModels = selectionInfo.extractValidCandidates();
|
|
1877
|
-
return validModels.map(considered => {
|
|
1878
|
-
// Find matching model vendor for driver and API info
|
|
1879
|
-
const modelVendor = considered.vendor
|
|
1880
|
-
? AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.ModelID, considered.model.ID) &&
|
|
1881
|
-
UUIDsEqual(mv.VendorID, considered.vendor.ID))
|
|
1882
|
-
: undefined;
|
|
1883
|
-
return {
|
|
1884
|
-
model: considered.model,
|
|
1885
|
-
vendorId: considered.vendor?.ID,
|
|
1886
|
-
vendorName: considered.vendor?.Name,
|
|
1887
|
-
driverClass: modelVendor?.DriverClass || considered.model.DriverClass,
|
|
1888
|
-
apiName: modelVendor?.APIName || considered.model.APIName,
|
|
1889
|
-
supportsEffortLevel: modelVendor?.SupportsEffortLevel ?? considered.model.SupportsEffortLevel ?? false,
|
|
1890
|
-
isPreferredVendor: false, // Can't determine from selection info alone
|
|
1891
|
-
priority: considered.priority,
|
|
1892
|
-
source: (selectionInfo.selectionStrategy === 'ByPower' ? 'power-rank' : 'model-type')
|
|
1893
|
-
};
|
|
1894
|
-
}).sort((a, b) => b.priority - a.priority); // Sort by priority descending
|
|
1895
|
-
}
|
|
1896
1893
|
/**
|
|
1897
1894
|
* Enhanced version of selectModelWithAPIKey that tracks all considered models
|
|
1898
1895
|
* for model selection reporting. Uses the hierarchical credential resolution
|
|
@@ -1908,8 +1905,33 @@ export class AIPromptRunner {
|
|
|
1908
1905
|
// Key format: "driverClass:modelId:vendorId" to properly cache credential hierarchy
|
|
1909
1906
|
const credentialCache = new Map();
|
|
1910
1907
|
const consideredModels = [];
|
|
1911
|
-
//
|
|
1908
|
+
// DECISION (performance): candidates are ordered by priority, and we only need the
|
|
1909
|
+
// highest-priority candidate that has working credentials. So once we find that first
|
|
1910
|
+
// hit, we STOP credential-probing the remaining candidates and record them as
|
|
1911
|
+
// "not-evaluated" rather than running a `hasCredentialsAvailable` check (which does
|
|
1912
|
+
// env-var lookups + binding scans) for every configured model on every prompt run.
|
|
1913
|
+
// The remaining candidates are still kept in `consideredModels` (and in the returned
|
|
1914
|
+
// `allCandidates` from selectModel, which is the FULL ordered list) so failover and the
|
|
1915
|
+
// ordering are unaffected — only the per-candidate availability *telemetry* for the tail
|
|
1916
|
+
// is skipped. Callers that need a complete availability report (e.g. an admin diagnostic)
|
|
1917
|
+
// can set `AIPromptParams.forceFullModelEvaluation = true` to probe every candidate.
|
|
1918
|
+
const forceFullEval = params?.forceFullModelEvaluation === true;
|
|
1919
|
+
let selected;
|
|
1912
1920
|
for (const candidate of candidates) {
|
|
1921
|
+
const vendorEntity = candidate.vendorId
|
|
1922
|
+
? AIEngine.Instance.VendorsByID.get(NormalizeUUID(candidate.vendorId))
|
|
1923
|
+
: undefined;
|
|
1924
|
+
// Short-circuit: a usable candidate is already selected and full evaluation wasn't requested.
|
|
1925
|
+
if (selected && !forceFullEval) {
|
|
1926
|
+
consideredModels.push({
|
|
1927
|
+
model: candidate.model,
|
|
1928
|
+
vendor: vendorEntity,
|
|
1929
|
+
priority: candidate.priority,
|
|
1930
|
+
available: false,
|
|
1931
|
+
unavailableReason: AIPromptRunner.NOT_EVALUATED_REASON
|
|
1932
|
+
});
|
|
1933
|
+
continue;
|
|
1934
|
+
}
|
|
1913
1935
|
// Build cache key including model and vendor for proper credential resolution
|
|
1914
1936
|
const cacheKey = `${candidate.driverClass}:${candidate.model.ID}:${candidate.vendorId || 'default'}`;
|
|
1915
1937
|
// Check cache first
|
|
@@ -1922,22 +1944,20 @@ export class AIPromptRunner {
|
|
|
1922
1944
|
hasCredentials = this.hasCredentialsAvailable(candidate.driverClass, promptId, candidate.model.ID, candidate.vendorId, params);
|
|
1923
1945
|
credentialCache.set(cacheKey, hasCredentials);
|
|
1924
1946
|
}
|
|
1925
|
-
// Get vendor entity from AIEngine cache if vendorId is available
|
|
1926
|
-
let vendorEntity;
|
|
1927
|
-
if (candidate.vendorId) {
|
|
1928
|
-
vendorEntity = AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, candidate.vendorId));
|
|
1929
|
-
}
|
|
1930
1947
|
// Track this model as considered with availability status
|
|
1931
|
-
|
|
1948
|
+
const considered = {
|
|
1932
1949
|
model: candidate.model,
|
|
1933
1950
|
vendor: vendorEntity,
|
|
1934
1951
|
priority: candidate.priority,
|
|
1935
1952
|
available: hasCredentials,
|
|
1936
1953
|
unavailableReason: hasCredentials ? undefined : `No credentials configured for driver ${candidate.driverClass}`
|
|
1937
|
-
}
|
|
1954
|
+
};
|
|
1955
|
+
consideredModels.push(considered);
|
|
1956
|
+
// Record the first available candidate as the selection (highest priority with credentials)
|
|
1957
|
+
if (hasCredentials && !selected) {
|
|
1958
|
+
selected = considered;
|
|
1959
|
+
}
|
|
1938
1960
|
}
|
|
1939
|
-
// Select the first available candidate (highest priority with API key)
|
|
1940
|
-
const selected = consideredModels.find(m => m.available);
|
|
1941
1961
|
const selectedCandidate = selected ? candidates.find(c => UUIDsEqual(c.model.ID, selected.model.ID) &&
|
|
1942
1962
|
UUIDsEqual(c.vendorId, selected.vendor?.ID)) : null;
|
|
1943
1963
|
if (selectedCandidate) {
|
|
@@ -1961,7 +1981,7 @@ export class AIPromptRunner {
|
|
|
1961
1981
|
maxErrorLength: params?.maxErrorLength
|
|
1962
1982
|
});
|
|
1963
1983
|
}
|
|
1964
|
-
return { selected: selectedCandidate, consideredModels };
|
|
1984
|
+
return { selected: selectedCandidate, consideredModels, credentialAvailability: credentialCache };
|
|
1965
1985
|
}
|
|
1966
1986
|
/**
|
|
1967
1987
|
* Builds a descriptive error message when no model could be selected for a prompt.
|
|
@@ -1990,6 +2010,75 @@ export class AIPromptRunner {
|
|
|
1990
2010
|
/**
|
|
1991
2011
|
* Creates an AIPromptRun entity for execution tracking
|
|
1992
2012
|
*/
|
|
2013
|
+
/**
|
|
2014
|
+
* Resolves the scalar inference parameters for a run: each value is the per-request override
|
|
2015
|
+
* from `additionalParameters` when supplied, otherwise the prompt's configured default. This
|
|
2016
|
+
* is the single source of truth for parameter precedence so {@link executeModel} (ChatParams)
|
|
2017
|
+
* and {@link createPromptRun} (the persisted record) stay in lockstep. Stop sequences and
|
|
2018
|
+
* assistant prefill are intentionally excluded — their representations differ per target.
|
|
2019
|
+
*/
|
|
2020
|
+
resolveScalarInferenceParams(prompt, additionalParameters) {
|
|
2021
|
+
const pick = (override, promptDefault) => override !== undefined ? override : (promptDefault != null ? promptDefault : undefined);
|
|
2022
|
+
const ap = additionalParameters;
|
|
2023
|
+
return {
|
|
2024
|
+
temperature: pick(ap?.temperature, prompt.Temperature),
|
|
2025
|
+
topP: pick(ap?.topP, prompt.TopP),
|
|
2026
|
+
topK: pick(ap?.topK, prompt.TopK),
|
|
2027
|
+
minP: pick(ap?.minP, prompt.MinP),
|
|
2028
|
+
frequencyPenalty: pick(ap?.frequencyPenalty, prompt.FrequencyPenalty),
|
|
2029
|
+
presencePenalty: pick(ap?.presencePenalty, prompt.PresencePenalty),
|
|
2030
|
+
seed: pick(ap?.seed, prompt.Seed),
|
|
2031
|
+
includeLogProbs: pick(ap?.includeLogProbs, prompt.IncludeLogProbs),
|
|
2032
|
+
topLogProbs: pick(ap?.topLogProbs, prompt.TopLogProbs),
|
|
2033
|
+
};
|
|
2034
|
+
}
|
|
2035
|
+
/**
|
|
2036
|
+
* Queues a fire-and-forget `Save()` for a prompt-run entity. Saves for the same instance are
|
|
2037
|
+
* chained (via {@link _promptRunSaveChains}) so the initial INSERT always completes before any
|
|
2038
|
+
* finalize UPDATE — guaranteeing a slow INSERT can't overwrite the finalized row. The whole
|
|
2039
|
+
* chain runs independently of the execution flow (callers do NOT await it), so the model call
|
|
2040
|
+
* is never delayed by a DB write.
|
|
2041
|
+
*
|
|
2042
|
+
* Save failures are logged (non-fatal): the AIPromptRun record is observability, not part of the
|
|
2043
|
+
* prompt's success contract, so a rare persistence failure must not fail the prompt. The chained
|
|
2044
|
+
* promise is tracked in {@link _pendingPromptRunSaves} so {@link WaitForPendingPromptRunSaves}
|
|
2045
|
+
* can flush them when determinism is required (e.g. tests, or a caller that needs the rows
|
|
2046
|
+
* durably written). Returns that promise.
|
|
2047
|
+
*/
|
|
2048
|
+
queuePromptRunSave(promptRun) {
|
|
2049
|
+
const previous = this._promptRunSaveChains.get(promptRun) ?? Promise.resolve(true);
|
|
2050
|
+
const current = previous
|
|
2051
|
+
.then(async () => {
|
|
2052
|
+
const ok = await promptRun.Save();
|
|
2053
|
+
if (!ok) {
|
|
2054
|
+
this.logError(`Failed to save AIPromptRun ${promptRun.ID || '(unsaved)'}: ${promptRun.LatestResult?.CompleteMessage || 'Unknown error'}`, {
|
|
2055
|
+
category: 'PromptRunSave',
|
|
2056
|
+
metadata: { promptRunId: promptRun.ID }
|
|
2057
|
+
});
|
|
2058
|
+
}
|
|
2059
|
+
return ok;
|
|
2060
|
+
})
|
|
2061
|
+
.catch((err) => {
|
|
2062
|
+
// Infrastructure-level throw (network, etc.) — log and swallow so the fire-and-forget
|
|
2063
|
+
// promise never surfaces as an unhandled rejection.
|
|
2064
|
+
this.logError(err instanceof Error ? err : new Error(String(err)), {
|
|
2065
|
+
category: 'PromptRunSave',
|
|
2066
|
+
metadata: { promptRunId: promptRun.ID }
|
|
2067
|
+
});
|
|
2068
|
+
return false;
|
|
2069
|
+
});
|
|
2070
|
+
this._promptRunSaveChains.set(promptRun, current);
|
|
2071
|
+
this._pendingPromptRunSaves.push(current);
|
|
2072
|
+
return current;
|
|
2073
|
+
}
|
|
2074
|
+
/**
|
|
2075
|
+
* Awaits all in-flight prompt-run saves queued by this runner instance. The normal execution
|
|
2076
|
+
* path does NOT call this — prompt-run persistence is intentionally fire-and-forget. Exposed for
|
|
2077
|
+
* tests and for callers that need the AIPromptRun rows durably written before proceeding.
|
|
2078
|
+
*/
|
|
2079
|
+
async WaitForPendingPromptRunSaves() {
|
|
2080
|
+
await Promise.allSettled(this._pendingPromptRunSaves);
|
|
2081
|
+
}
|
|
1993
2082
|
async createPromptRun(prompt, model, params, systemPromptText, startTime, vendorId, modelSelectionInfo) {
|
|
1994
2083
|
const provider = params.provider ?? Metadata.Provider;
|
|
1995
2084
|
const promptRun = await provider.GetEntityObject('MJ: AI Prompt Runs', params.contextUser);
|
|
@@ -2005,7 +2094,7 @@ export class AIPromptRunner {
|
|
|
2005
2094
|
promptRun.Status = 'Running';
|
|
2006
2095
|
promptRun.Cancelled = false;
|
|
2007
2096
|
promptRun.CacheHit = false;
|
|
2008
|
-
promptRun.StreamingEnabled =
|
|
2097
|
+
promptRun.StreamingEnabled = !!params.onStreaming;
|
|
2009
2098
|
promptRun.WasSelectedResult = false;
|
|
2010
2099
|
// Set model selection tracking fields
|
|
2011
2100
|
if (modelSelectionInfo) {
|
|
@@ -2058,8 +2147,8 @@ export class AIPromptRunner {
|
|
|
2058
2147
|
}
|
|
2059
2148
|
else {
|
|
2060
2149
|
// Fallback: grab the highest priority AI Model Vendor record for this model (inference providers only)
|
|
2061
|
-
const modelVendors =
|
|
2062
|
-
.filter((mv) =>
|
|
2150
|
+
const modelVendors = model.ModelVendors
|
|
2151
|
+
.filter((mv) => mv.Status === 'Active' && this.isInferenceProvider(mv))
|
|
2063
2152
|
.sort((a, b) => b.Priority - a.Priority);
|
|
2064
2153
|
if (modelVendors.length > 0) {
|
|
2065
2154
|
promptRun.VendorID = modelVendors[0].VendorID;
|
|
@@ -2091,62 +2180,35 @@ export class AIPromptRunner {
|
|
|
2091
2180
|
if (prompt.ResponseFormat && prompt.ResponseFormat !== 'Any') {
|
|
2092
2181
|
promptRun.ResponseFormat = prompt.ResponseFormat;
|
|
2093
2182
|
}
|
|
2094
|
-
// Save the actual values that will be used (
|
|
2095
|
-
//
|
|
2096
|
-
|
|
2097
|
-
|
|
2098
|
-
|
|
2099
|
-
|
|
2100
|
-
|
|
2101
|
-
|
|
2102
|
-
|
|
2103
|
-
|
|
2104
|
-
|
|
2105
|
-
|
|
2106
|
-
|
|
2107
|
-
|
|
2108
|
-
|
|
2109
|
-
|
|
2183
|
+
// Save the actual values that will be used (prompt defaults overridden by additionalParameters).
|
|
2184
|
+
// Uses the shared resolver so the persisted record matches what executeModel sends to the model.
|
|
2185
|
+
const resolvedParams = this.resolveScalarInferenceParams(prompt, params.additionalParameters);
|
|
2186
|
+
if (resolvedParams.temperature !== undefined)
|
|
2187
|
+
promptRun.Temperature = resolvedParams.temperature;
|
|
2188
|
+
if (resolvedParams.topP !== undefined)
|
|
2189
|
+
promptRun.TopP = resolvedParams.topP;
|
|
2190
|
+
if (resolvedParams.topK !== undefined)
|
|
2191
|
+
promptRun.TopK = resolvedParams.topK;
|
|
2192
|
+
if (resolvedParams.minP !== undefined)
|
|
2193
|
+
promptRun.MinP = resolvedParams.minP;
|
|
2194
|
+
if (resolvedParams.frequencyPenalty !== undefined)
|
|
2195
|
+
promptRun.FrequencyPenalty = resolvedParams.frequencyPenalty;
|
|
2196
|
+
if (resolvedParams.presencePenalty !== undefined)
|
|
2197
|
+
promptRun.PresencePenalty = resolvedParams.presencePenalty;
|
|
2198
|
+
if (resolvedParams.seed !== undefined)
|
|
2199
|
+
promptRun.Seed = resolvedParams.seed;
|
|
2200
|
+
if (resolvedParams.includeLogProbs !== undefined)
|
|
2201
|
+
promptRun.LogProbs = resolvedParams.includeLogProbs;
|
|
2202
|
+
if (resolvedParams.topLogProbs !== undefined)
|
|
2203
|
+
promptRun.TopLogProbs = resolvedParams.topLogProbs;
|
|
2204
|
+
// Stop sequences + assistant prefill: stored from the prompt, with the additionalParameters
|
|
2205
|
+
// array (JSON-encoded) taking precedence when supplied.
|
|
2110
2206
|
if (prompt.StopSequences)
|
|
2111
2207
|
promptRun.StopSequences = prompt.StopSequences;
|
|
2112
2208
|
if (prompt.AssistantPrefill)
|
|
2113
2209
|
promptRun.AssistantPrefill = prompt.AssistantPrefill;
|
|
2114
|
-
if (
|
|
2115
|
-
promptRun.
|
|
2116
|
-
if (prompt.TopLogProbs != null)
|
|
2117
|
-
promptRun.TopLogProbs = prompt.TopLogProbs;
|
|
2118
|
-
// Then override with additionalParameters if provided
|
|
2119
|
-
if (params.additionalParameters) {
|
|
2120
|
-
if (params.additionalParameters.temperature !== undefined) {
|
|
2121
|
-
promptRun.Temperature = params.additionalParameters.temperature;
|
|
2122
|
-
}
|
|
2123
|
-
if (params.additionalParameters.topP !== undefined) {
|
|
2124
|
-
promptRun.TopP = params.additionalParameters.topP;
|
|
2125
|
-
}
|
|
2126
|
-
if (params.additionalParameters.topK !== undefined) {
|
|
2127
|
-
promptRun.TopK = params.additionalParameters.topK;
|
|
2128
|
-
}
|
|
2129
|
-
if (params.additionalParameters.minP !== undefined) {
|
|
2130
|
-
promptRun.MinP = params.additionalParameters.minP;
|
|
2131
|
-
}
|
|
2132
|
-
if (params.additionalParameters.frequencyPenalty !== undefined) {
|
|
2133
|
-
promptRun.FrequencyPenalty = params.additionalParameters.frequencyPenalty;
|
|
2134
|
-
}
|
|
2135
|
-
if (params.additionalParameters.presencePenalty !== undefined) {
|
|
2136
|
-
promptRun.PresencePenalty = params.additionalParameters.presencePenalty;
|
|
2137
|
-
}
|
|
2138
|
-
if (params.additionalParameters.seed !== undefined) {
|
|
2139
|
-
promptRun.Seed = params.additionalParameters.seed;
|
|
2140
|
-
}
|
|
2141
|
-
if (params.additionalParameters.stopSequences !== undefined && params.additionalParameters.stopSequences.length > 0) {
|
|
2142
|
-
promptRun.StopSequences = JSON.stringify(params.additionalParameters.stopSequences);
|
|
2143
|
-
}
|
|
2144
|
-
if (params.additionalParameters.includeLogProbs !== undefined) {
|
|
2145
|
-
promptRun.LogProbs = params.additionalParameters.includeLogProbs;
|
|
2146
|
-
}
|
|
2147
|
-
if (params.additionalParameters.topLogProbs !== undefined) {
|
|
2148
|
-
promptRun.TopLogProbs = params.additionalParameters.topLogProbs;
|
|
2149
|
-
}
|
|
2210
|
+
if (params.additionalParameters?.stopSequences !== undefined && params.additionalParameters.stopSequences.length > 0) {
|
|
2211
|
+
promptRun.StopSequences = JSON.stringify(params.additionalParameters.stopSequences);
|
|
2150
2212
|
}
|
|
2151
2213
|
// Store the input data/context as JSON in Messages field
|
|
2152
2214
|
if (params.data || params.templateData || systemPromptText) {
|
|
@@ -2179,21 +2241,13 @@ export class AIPromptRunner {
|
|
|
2179
2241
|
promptRun.ValidationAttemptCount = 0; // Will be updated during execution
|
|
2180
2242
|
promptRun.SuccessfulValidationCount = 0;
|
|
2181
2243
|
promptRun.FinalValidationPassed = false; // Will be updated after execution
|
|
2182
|
-
|
|
2183
|
-
|
|
2184
|
-
|
|
2185
|
-
|
|
2186
|
-
|
|
2187
|
-
|
|
2188
|
-
|
|
2189
|
-
modelId: model.ID,
|
|
2190
|
-
vendorId
|
|
2191
|
-
},
|
|
2192
|
-
maxErrorLength: params.maxErrorLength
|
|
2193
|
-
});
|
|
2194
|
-
throw new Error(error);
|
|
2195
|
-
}
|
|
2196
|
-
// Invoke callback if provided
|
|
2244
|
+
// Persist the initial 'Running' record fire-and-forget. The ID was already assigned by
|
|
2245
|
+
// NewRecord() above, so callers (and the onPromptRunCreated callback) have it immediately —
|
|
2246
|
+
// we don't block the model call on the INSERT. The finalize UPDATE chains after this INSERT
|
|
2247
|
+
// via the instance-keyed save queue, so ordering is guaranteed.
|
|
2248
|
+
this.queuePromptRunSave(promptRun);
|
|
2249
|
+
// Invoke callback if provided. The ID is available without awaiting the save (client-generated
|
|
2250
|
+
// by NewRecord()), so agent-run/step linking that depends on it works immediately.
|
|
2197
2251
|
if (params.onPromptRunCreated) {
|
|
2198
2252
|
try {
|
|
2199
2253
|
await params.onPromptRunCreated(promptRun.ID);
|
|
@@ -2283,7 +2337,7 @@ export class AIPromptRunner {
|
|
|
2283
2337
|
* - updatePromptRunWithFailoverFailure: Records failed failover metadata
|
|
2284
2338
|
* - createFailoverErrorResult: Creates standardized error response
|
|
2285
2339
|
*/
|
|
2286
|
-
async executeModelWithFailover(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, allCandidates, promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel) {
|
|
2340
|
+
async executeModelWithFailover(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, allCandidates, promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability) {
|
|
2287
2341
|
// Get failover configuration (used for errorScope filtering)
|
|
2288
2342
|
const failoverConfig = this.getFailoverConfiguration(prompt);
|
|
2289
2343
|
// If no candidates provided or failover disabled, execute normally with first model
|
|
@@ -2293,10 +2347,45 @@ export class AIPromptRunner {
|
|
|
2293
2347
|
// Track failover attempts
|
|
2294
2348
|
const failoverAttempts = [];
|
|
2295
2349
|
let lastError = null;
|
|
2350
|
+
// Cache credential availability per driver:model:vendor for the duration of this failover
|
|
2351
|
+
// scan so we don't repeat env-var / binding lookups while walking the candidate list.
|
|
2352
|
+
//
|
|
2353
|
+
// PERF: seed it with the probes model SELECTION already performed (same key format). Selection
|
|
2354
|
+
// walks the priority list until it finds the first credentialed candidate, so this map holds
|
|
2355
|
+
// the prefix it rejected (known false) PLUS the selected candidate (known true) — which is
|
|
2356
|
+
// exactly the segment failover re-walks on the happy path. Reusing those results means the
|
|
2357
|
+
// common case (and any caller looping failover) does ZERO redundant hasCredentialsAvailable
|
|
2358
|
+
// calls. The not-evaluated tail is intentionally absent, so failover still lazily probes it
|
|
2359
|
+
// only if a real failure forces it to walk down there.
|
|
2360
|
+
const failoverCredentialCache = credentialAvailability
|
|
2361
|
+
? new Map(credentialAvailability)
|
|
2362
|
+
: new Map();
|
|
2363
|
+
const candidateHasCredentials = (c) => {
|
|
2364
|
+
const key = `${c.driverClass}:${c.model.ID}:${c.vendorId || 'default'}`;
|
|
2365
|
+
let has = failoverCredentialCache.get(key);
|
|
2366
|
+
if (has === undefined) {
|
|
2367
|
+
has = this.hasCredentialsAvailable(c.driverClass, prompt.ID, c.model.ID, c.vendorId, params);
|
|
2368
|
+
failoverCredentialCache.set(key, has);
|
|
2369
|
+
}
|
|
2370
|
+
return has;
|
|
2371
|
+
};
|
|
2372
|
+
let skippedForCredentials = 0;
|
|
2296
2373
|
// Iterate through all candidates in priority order with instant failover
|
|
2297
2374
|
for (let i = 0; i < allCandidates.length; i++) {
|
|
2298
2375
|
const candidate = allCandidates[i];
|
|
2299
2376
|
const attemptStartTime = Date.now();
|
|
2377
|
+
// Skip candidates with no credentials configured. `allCandidates` is intentionally the
|
|
2378
|
+
// FULL priority-ordered list (see the DECISION note in selectModelWithAPIKeyTracked),
|
|
2379
|
+
// so it can include vendors that have no API key in this environment. Firing a live
|
|
2380
|
+
// request at one of those produces a misleading "401 invalid API key" — and because an
|
|
2381
|
+
// Authentication error is treated as fatal, it would halt failover before any
|
|
2382
|
+
// credentialed candidate is ever reached. Skipping here makes failover land on the
|
|
2383
|
+
// first candidate that can actually authenticate (mirroring model selection's own
|
|
2384
|
+
// highest-priority-with-credentials rule).
|
|
2385
|
+
if (!candidateHasCredentials(candidate)) {
|
|
2386
|
+
skippedForCredentials++;
|
|
2387
|
+
continue;
|
|
2388
|
+
}
|
|
2300
2389
|
try {
|
|
2301
2390
|
// Log the attempt if not the first one
|
|
2302
2391
|
if (i > 0) {
|
|
@@ -2365,6 +2454,11 @@ export class AIPromptRunner {
|
|
|
2365
2454
|
if (promptRun && failoverAttempts.length > 0) {
|
|
2366
2455
|
this.updatePromptRunWithFailoverFailure(promptRun, failoverAttempts);
|
|
2367
2456
|
}
|
|
2457
|
+
// If every candidate was skipped for missing credentials we never attempted a call and
|
|
2458
|
+
// have no underlying error to report — surface an actionable message instead of null.
|
|
2459
|
+
if (!lastError && failoverAttempts.length === 0 && skippedForCredentials > 0) {
|
|
2460
|
+
lastError = new Error(`No API credentials configured for any of the ${skippedForCredentials} candidate model-vendor combination(s) for prompt "${prompt.Name}".`);
|
|
2461
|
+
}
|
|
2368
2462
|
return this.createFailoverErrorResult(lastError, failoverAttempts);
|
|
2369
2463
|
}
|
|
2370
2464
|
/**
|
|
@@ -2500,7 +2594,11 @@ export class AIPromptRunner {
|
|
|
2500
2594
|
};
|
|
2501
2595
|
}
|
|
2502
2596
|
/**
|
|
2503
|
-
* Executes the AI model with the rendered prompt
|
|
2597
|
+
* Executes the AI model with the rendered prompt.
|
|
2598
|
+
*
|
|
2599
|
+
* `protected` so the parallel coordinator subclass reuses this exact code path — credential
|
|
2600
|
+
* resolution, driver selection, ChatParams construction, prefill, media handling, and streaming
|
|
2601
|
+
* all live here ONCE. Do not duplicate this logic elsewhere.
|
|
2504
2602
|
*/
|
|
2505
2603
|
async executeModel(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel) {
|
|
2506
2604
|
// define these variables here to ensure they're available in the catch block
|
|
@@ -2530,7 +2628,7 @@ export class AIPromptRunner {
|
|
|
2530
2628
|
supportsEffortLevel = model.SupportsEffortLevel ?? false;
|
|
2531
2629
|
if (vendorId) {
|
|
2532
2630
|
// Find the AIModelVendor record for this specific vendor - must be an inference provider
|
|
2533
|
-
const modelVendor =
|
|
2631
|
+
const modelVendor = model.ModelVendors.find((mv) => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active' && this.isInferenceProvider(mv));
|
|
2534
2632
|
if (modelVendor) {
|
|
2535
2633
|
driverClass = modelVendor.DriverClass || driverClass;
|
|
2536
2634
|
apiName = modelVendor.APIName || apiName;
|
|
@@ -2554,63 +2652,34 @@ export class AIPromptRunner {
|
|
|
2554
2652
|
}
|
|
2555
2653
|
chatParams.model = apiName;
|
|
2556
2654
|
chatParams.cancellationToken = cancellationToken;
|
|
2557
|
-
// Apply
|
|
2558
|
-
//
|
|
2559
|
-
|
|
2560
|
-
|
|
2561
|
-
|
|
2562
|
-
|
|
2563
|
-
|
|
2564
|
-
|
|
2565
|
-
|
|
2566
|
-
|
|
2567
|
-
|
|
2568
|
-
|
|
2569
|
-
|
|
2570
|
-
|
|
2571
|
-
|
|
2572
|
-
|
|
2655
|
+
// Apply scalar inference params (prompt defaults overridden by additionalParameters) via the
|
|
2656
|
+
// shared resolver so ChatParams and the persisted AIPromptRun never drift.
|
|
2657
|
+
const resolvedParams = this.resolveScalarInferenceParams(prompt, params.additionalParameters);
|
|
2658
|
+
if (resolvedParams.temperature !== undefined)
|
|
2659
|
+
chatParams.temperature = resolvedParams.temperature;
|
|
2660
|
+
if (resolvedParams.topP !== undefined)
|
|
2661
|
+
chatParams.topP = resolvedParams.topP;
|
|
2662
|
+
if (resolvedParams.topK !== undefined)
|
|
2663
|
+
chatParams.topK = resolvedParams.topK;
|
|
2664
|
+
if (resolvedParams.minP !== undefined)
|
|
2665
|
+
chatParams.minP = resolvedParams.minP;
|
|
2666
|
+
if (resolvedParams.frequencyPenalty !== undefined)
|
|
2667
|
+
chatParams.frequencyPenalty = resolvedParams.frequencyPenalty;
|
|
2668
|
+
if (resolvedParams.presencePenalty !== undefined)
|
|
2669
|
+
chatParams.presencePenalty = resolvedParams.presencePenalty;
|
|
2670
|
+
if (resolvedParams.seed !== undefined)
|
|
2671
|
+
chatParams.seed = resolvedParams.seed;
|
|
2672
|
+
if (resolvedParams.includeLogProbs !== undefined)
|
|
2673
|
+
chatParams.includeLogProbs = resolvedParams.includeLogProbs;
|
|
2674
|
+
if (resolvedParams.topLogProbs !== undefined)
|
|
2675
|
+
chatParams.topLogProbs = resolvedParams.topLogProbs;
|
|
2676
|
+
// Stop sequences are handled separately: the prompt value is comma-delimited and gated by
|
|
2677
|
+
// driver support; additionalParameters supplies a ready-made array that overrides it.
|
|
2573
2678
|
if (prompt.StopSequences && this.shouldApplyStopSequences(prompt, model, vendorId, llm)) {
|
|
2574
|
-
// Parse comma-delimited stop sequences
|
|
2575
2679
|
chatParams.stopSequences = prompt.StopSequences.split(',').map((s) => s.replace(AIPromptRunner.STOP_SEQUENCE_TRIM_REGEX, '')).filter((s) => s.length > 0);
|
|
2576
2680
|
}
|
|
2577
|
-
if (
|
|
2578
|
-
chatParams.
|
|
2579
|
-
if (prompt.TopLogProbs != null)
|
|
2580
|
-
chatParams.topLogProbs = prompt.TopLogProbs;
|
|
2581
|
-
// Apply additional parameters if provided (these override prompt defaults)
|
|
2582
|
-
if (params.additionalParameters) {
|
|
2583
|
-
// Apply chat-specific parameters from additionalParameters
|
|
2584
|
-
if (params.additionalParameters.temperature !== undefined) {
|
|
2585
|
-
chatParams.temperature = params.additionalParameters.temperature;
|
|
2586
|
-
}
|
|
2587
|
-
if (params.additionalParameters.topP !== undefined) {
|
|
2588
|
-
chatParams.topP = params.additionalParameters.topP;
|
|
2589
|
-
}
|
|
2590
|
-
if (params.additionalParameters.topK !== undefined) {
|
|
2591
|
-
chatParams.topK = params.additionalParameters.topK;
|
|
2592
|
-
}
|
|
2593
|
-
if (params.additionalParameters.minP !== undefined) {
|
|
2594
|
-
chatParams.minP = params.additionalParameters.minP;
|
|
2595
|
-
}
|
|
2596
|
-
if (params.additionalParameters.frequencyPenalty !== undefined) {
|
|
2597
|
-
chatParams.frequencyPenalty = params.additionalParameters.frequencyPenalty;
|
|
2598
|
-
}
|
|
2599
|
-
if (params.additionalParameters.presencePenalty !== undefined) {
|
|
2600
|
-
chatParams.presencePenalty = params.additionalParameters.presencePenalty;
|
|
2601
|
-
}
|
|
2602
|
-
if (params.additionalParameters.seed !== undefined) {
|
|
2603
|
-
chatParams.seed = params.additionalParameters.seed;
|
|
2604
|
-
}
|
|
2605
|
-
if (params.additionalParameters.stopSequences !== undefined) {
|
|
2606
|
-
chatParams.stopSequences = params.additionalParameters.stopSequences;
|
|
2607
|
-
}
|
|
2608
|
-
if (params.additionalParameters.includeLogProbs !== undefined) {
|
|
2609
|
-
chatParams.includeLogProbs = params.additionalParameters.includeLogProbs;
|
|
2610
|
-
}
|
|
2611
|
-
if (params.additionalParameters.topLogProbs !== undefined) {
|
|
2612
|
-
chatParams.topLogProbs = params.additionalParameters.topLogProbs;
|
|
2613
|
-
}
|
|
2681
|
+
if (params.additionalParameters?.stopSequences !== undefined) {
|
|
2682
|
+
chatParams.stopSequences = params.additionalParameters.stopSequences;
|
|
2614
2683
|
}
|
|
2615
2684
|
// Apply effortLevel with precedence hierarchy
|
|
2616
2685
|
// 1. params.effortLevel (runtime override - highest priority)
|
|
@@ -2671,6 +2740,17 @@ export class AIPromptRunner {
|
|
|
2671
2740
|
this.stripUnsupportedMediaBlocks(llm, chatParams, model, verbose, params);
|
|
2672
2741
|
// Apply assistant prefill (native or fallback) based on prompt config and provider support
|
|
2673
2742
|
this.applyAssistantPrefill(chatParams, prompt, model, vendorId, llm);
|
|
2743
|
+
// Streaming: wire the prompt-level onStreaming callback into the LLM call. This is the SINGLE
|
|
2744
|
+
// place streaming is configured for prompt execution, so the single-model path and the parallel
|
|
2745
|
+
// path (which bridges its per-task callbacks into params.onStreaming) stream through identical
|
|
2746
|
+
// code — no second streaming implementation that can drift.
|
|
2747
|
+
if (params.onStreaming) {
|
|
2748
|
+
const onStreaming = params.onStreaming;
|
|
2749
|
+
chatParams.streaming = true;
|
|
2750
|
+
chatParams.streamingCallbacks = {
|
|
2751
|
+
OnContent: (chunk, isComplete) => onStreaming({ content: chunk, isComplete, modelName: model.Name }),
|
|
2752
|
+
};
|
|
2753
|
+
}
|
|
2674
2754
|
// Execute the model with cancellation support
|
|
2675
2755
|
if (cancellationToken) {
|
|
2676
2756
|
// If cancellation token is provided, wrap the execution to handle cancellation
|
|
@@ -2819,8 +2899,13 @@ export class AIPromptRunner {
|
|
|
2819
2899
|
const lower = mimeType.toLowerCase();
|
|
2820
2900
|
return caps.SupportedMimeTypes.some((pattern) => {
|
|
2821
2901
|
const p = pattern.toLowerCase();
|
|
2902
|
+
// Wildcard on EITHER side must match (the requested mime is often a modality
|
|
2903
|
+
// probe like 'image/*' — e.g. an image_url block with no explicit mimeType —
|
|
2904
|
+
// and must match a driver that declares any concrete 'image/<x>' type).
|
|
2822
2905
|
if (p.endsWith('/*'))
|
|
2823
2906
|
return lower.startsWith(p.slice(0, -1));
|
|
2907
|
+
if (lower.endsWith('/*'))
|
|
2908
|
+
return p.startsWith(lower.slice(0, -1));
|
|
2824
2909
|
return lower === p;
|
|
2825
2910
|
});
|
|
2826
2911
|
}
|
|
@@ -2976,7 +3061,7 @@ export class AIPromptRunner {
|
|
|
2976
3061
|
}
|
|
2977
3062
|
// Vendor-level override (null = inherit)
|
|
2978
3063
|
if (vendorId) {
|
|
2979
|
-
const modelVendor =
|
|
3064
|
+
const modelVendor = model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
|
|
2980
3065
|
if (modelVendor?.SupportsPrefill != null) {
|
|
2981
3066
|
supportsPrefill = modelVendor.SupportsPrefill;
|
|
2982
3067
|
}
|
|
@@ -2990,7 +3075,7 @@ export class AIPromptRunner {
|
|
|
2990
3075
|
*/
|
|
2991
3076
|
resolvePrefillFallbackText(model, vendorId) {
|
|
2992
3077
|
// Start with model type default
|
|
2993
|
-
const modelType = AIEngine.Instance.
|
|
3078
|
+
const modelType = AIEngine.Instance.ModelTypesByID.get(NormalizeUUID(model.AIModelTypeID));
|
|
2994
3079
|
let fallbackText = modelType?.PrefillFallbackText ?? null;
|
|
2995
3080
|
// Model-level override
|
|
2996
3081
|
if (model.PrefillFallbackText != null) {
|
|
@@ -2998,7 +3083,7 @@ export class AIPromptRunner {
|
|
|
2998
3083
|
}
|
|
2999
3084
|
// Vendor-level override
|
|
3000
3085
|
if (vendorId) {
|
|
3001
|
-
const modelVendor =
|
|
3086
|
+
const modelVendor = model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
|
|
3002
3087
|
if (modelVendor?.PrefillFallbackText != null) {
|
|
3003
3088
|
fallbackText = modelVendor.PrefillFallbackText;
|
|
3004
3089
|
}
|
|
@@ -3045,7 +3130,7 @@ export class AIPromptRunner {
|
|
|
3045
3130
|
/**
|
|
3046
3131
|
* Executes the model with retry logic for validation failures
|
|
3047
3132
|
*/
|
|
3048
|
-
async executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel) {
|
|
3133
|
+
async executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability) {
|
|
3049
3134
|
const validationAttempts = [];
|
|
3050
3135
|
const maxRetries = Math.max(0, prompt.MaxRetries || 0);
|
|
3051
3136
|
let lastError = null;
|
|
@@ -3065,7 +3150,8 @@ export class AIPromptRunner {
|
|
|
3065
3150
|
}
|
|
3066
3151
|
// Execute the AI model with failover support
|
|
3067
3152
|
const modelResult = await this.executeModelWithFailover(selectedModel, renderedPromptText, prompt, params, promptRun.VendorID, params.conversationMessages, params.templateMessageRole || 'system', params.cancellationToken, allCandidates, // Pass the candidates from initial selection
|
|
3068
|
-
promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel
|
|
3153
|
+
promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability // Reuse credential probes from selection
|
|
3154
|
+
);
|
|
3069
3155
|
// Check for fatal errors - don't attempt validation/retry on these
|
|
3070
3156
|
// Fatal errors (like ContextLengthExceeded when all models exhausted) cannot be resolved by retrying
|
|
3071
3157
|
if (!modelResult.success && modelResult.errorInfo?.severity === 'Fatal') {
|
|
@@ -3230,7 +3316,7 @@ export class AIPromptRunner {
|
|
|
3230
3316
|
const filteredCandidates = allCandidates.filter(c => (c.vendorId || 'default') !== failedVendorId);
|
|
3231
3317
|
const removedCount = beforeCount - filteredCandidates.length;
|
|
3232
3318
|
if (removedCount > 0) {
|
|
3233
|
-
const vendorName = AIEngine.Instance.
|
|
3319
|
+
const vendorName = AIEngine.Instance.VendorsByID.get(NormalizeUUID(failedVendorId))?.Name || failedVendorId;
|
|
3234
3320
|
const remainingCount = filteredCandidates.length;
|
|
3235
3321
|
// Log appropriate message based on error type
|
|
3236
3322
|
let reason;
|
|
@@ -3271,7 +3357,7 @@ export class AIPromptRunner {
|
|
|
3271
3357
|
if (shouldRetry) {
|
|
3272
3358
|
const modelName = currentModel.Name;
|
|
3273
3359
|
const vendorName = currentVendorId
|
|
3274
|
-
? AIEngine.Instance.
|
|
3360
|
+
? AIEngine.Instance.VendorsByID.get(NormalizeUUID(currentVendorId))?.Name || 'default'
|
|
3275
3361
|
: 'default';
|
|
3276
3362
|
this.logStatus(` ⏳ Rate limit hit - retrying ${modelName} (${vendorName}) with backoff (attempt ${rateLimitRetryCount}/${maxRetries})`, true);
|
|
3277
3363
|
this.logFailoverAttempt(prompt.ID, failoverAttempt, true);
|
|
@@ -3746,28 +3832,28 @@ export class AIPromptRunner {
|
|
|
3746
3832
|
jsonToParse = CleanJSON(rawOutput);
|
|
3747
3833
|
}
|
|
3748
3834
|
catch (cleanError) {
|
|
3749
|
-
|
|
3750
|
-
|
|
3751
|
-
|
|
3752
|
-
|
|
3753
|
-
|
|
3754
|
-
|
|
3755
|
-
|
|
3756
|
-
|
|
3757
|
-
});
|
|
3758
|
-
}
|
|
3835
|
+
this.logError(cleanError, {
|
|
3836
|
+
category: 'JSONCleaningFailed',
|
|
3837
|
+
metadata: {
|
|
3838
|
+
originalError: originalError.message,
|
|
3839
|
+
rawOutput: rawOutput.substring(0, 500)
|
|
3840
|
+
},
|
|
3841
|
+
maxErrorLength: params.maxErrorLength
|
|
3842
|
+
});
|
|
3759
3843
|
}
|
|
3760
3844
|
const json5Result = JSON5.parse(jsonToParse);
|
|
3761
|
-
|
|
3762
|
-
|
|
3763
|
-
|
|
3845
|
+
this.logStatus(' ✅ JSON5 successfully parsed the malformed JSON', true, params);
|
|
3846
|
+
currentPromptRun._jsonRepairInfo = {
|
|
3847
|
+
repaired: true,
|
|
3848
|
+
method: 'JSON5',
|
|
3849
|
+
originalError: originalError.message,
|
|
3850
|
+
rawOutputPrefix: rawOutput.substring(0, 200)
|
|
3851
|
+
};
|
|
3764
3852
|
return json5Result;
|
|
3765
3853
|
}
|
|
3766
3854
|
catch (json5Error) {
|
|
3767
3855
|
// Step 2: Use AI to repair the JSON
|
|
3768
|
-
|
|
3769
|
-
this.logStatus(' 🤖 JSON5 failed, attempting AI-based JSON repair...', true, params);
|
|
3770
|
-
}
|
|
3856
|
+
this.logStatus(' 🤖 JSON5 failed, attempting AI-based JSON repair...', true, params);
|
|
3771
3857
|
try {
|
|
3772
3858
|
// Find the "Repair JSON" prompt in the "MJ: System" category
|
|
3773
3859
|
const repairPrompt = AIEngine.Instance.Prompts.find(p => p.Name.trim().toLowerCase() === 'repair json' && p.Category.trim().toLowerCase() === 'mj: system');
|
|
@@ -3797,39 +3883,61 @@ export class AIPromptRunner {
|
|
|
3797
3883
|
}
|
|
3798
3884
|
// if we get here, we successfully repaired the JSON!!!
|
|
3799
3885
|
this.logStatus(' ✅ AI successfully repaired the JSON', true, params);
|
|
3886
|
+
currentPromptRun._jsonRepairInfo = {
|
|
3887
|
+
repaired: true,
|
|
3888
|
+
method: 'AIRepair',
|
|
3889
|
+
originalError: originalError.message,
|
|
3890
|
+
rawOutputPrefix: rawOutput.substring(0, 200),
|
|
3891
|
+
repairPromptRunId: repairResult.promptRun?.ID
|
|
3892
|
+
};
|
|
3800
3893
|
return repairedJSON;
|
|
3801
3894
|
}
|
|
3802
3895
|
catch (aiRepairError) {
|
|
3803
|
-
// Both repair attempts failed
|
|
3804
|
-
|
|
3805
|
-
|
|
3806
|
-
|
|
3807
|
-
|
|
3808
|
-
|
|
3809
|
-
|
|
3810
|
-
|
|
3811
|
-
|
|
3812
|
-
|
|
3813
|
-
|
|
3814
|
-
});
|
|
3815
|
-
}
|
|
3896
|
+
// Both repair attempts failed — always log, this is unexpected LLM behavior
|
|
3897
|
+
this.logError(aiRepairError, {
|
|
3898
|
+
category: 'JSONRepairFailed',
|
|
3899
|
+
metadata: {
|
|
3900
|
+
originalError: originalError.message,
|
|
3901
|
+
json5Error: json5Error.message,
|
|
3902
|
+
aiError: aiRepairError.message,
|
|
3903
|
+
rawOutput: rawOutput.substring(0, 500)
|
|
3904
|
+
},
|
|
3905
|
+
maxErrorLength: params.maxErrorLength
|
|
3906
|
+
});
|
|
3816
3907
|
throw new Error(`JSON repair failed after both JSON5 and AI attempts: ${originalError.message}`);
|
|
3817
3908
|
}
|
|
3818
3909
|
}
|
|
3819
3910
|
}
|
|
3911
|
+
/**
|
|
3912
|
+
* Returns the parsed form of a prompt's `OutputExample` JSON, memoized by content.
|
|
3913
|
+
* Parsing happens at most once per distinct example string for the life of the process;
|
|
3914
|
+
* parse failures are cached too (so malformed examples aren't re-parsed every attempt).
|
|
3915
|
+
*/
|
|
3916
|
+
getParsedOutputExample(outputExample) {
|
|
3917
|
+
const cached = AIPromptRunner._outputExampleCache.get(outputExample);
|
|
3918
|
+
if (cached) {
|
|
3919
|
+
return cached;
|
|
3920
|
+
}
|
|
3921
|
+
let entry;
|
|
3922
|
+
try {
|
|
3923
|
+
entry = { parsed: JSON.parse(outputExample) };
|
|
3924
|
+
}
|
|
3925
|
+
catch (parseError) {
|
|
3926
|
+
entry = { error: parseError instanceof Error ? parseError.message : String(parseError) };
|
|
3927
|
+
}
|
|
3928
|
+
AIPromptRunner._outputExampleCache.set(outputExample, entry);
|
|
3929
|
+
return entry;
|
|
3930
|
+
}
|
|
3820
3931
|
/**
|
|
3821
3932
|
* Validates parsed result against JSON schema derived from OutputExample
|
|
3822
3933
|
*/
|
|
3823
3934
|
async validateAgainstSchema(parsedResult, outputExample, promptId) {
|
|
3824
3935
|
const validationErrors = [];
|
|
3825
3936
|
try {
|
|
3826
|
-
// Parse the output example
|
|
3827
|
-
|
|
3828
|
-
|
|
3829
|
-
|
|
3830
|
-
}
|
|
3831
|
-
catch (parseError) {
|
|
3832
|
-
const error = new ValidationErrorInfo('outputExample', `Invalid OutputExample JSON: ${parseError.message}`, outputExample, ValidationErrorType.Failure);
|
|
3937
|
+
// Parse the output example (cached by content — it's a static string reused across runs/retries)
|
|
3938
|
+
const { parsed: exampleObject, error: exampleParseError } = this.getParsedOutputExample(outputExample);
|
|
3939
|
+
if (exampleParseError) {
|
|
3940
|
+
const error = new ValidationErrorInfo('outputExample', `Invalid OutputExample JSON: ${exampleParseError}`, outputExample, ValidationErrorType.Failure);
|
|
3833
3941
|
validationErrors.push(error);
|
|
3834
3942
|
return validationErrors;
|
|
3835
3943
|
}
|
|
@@ -3915,6 +4023,12 @@ export class AIPromptRunner {
|
|
|
3915
4023
|
*/
|
|
3916
4024
|
async updatePromptRun(promptRun, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens) {
|
|
3917
4025
|
try {
|
|
4026
|
+
// Ensure the initial 'Running' INSERT (and its BaseEntity.finalizeSave post-save reload, which does
|
|
4027
|
+
// init() + SetMany(insertedRow)) has fully landed BEFORE we mutate the final state. Mutating while the
|
|
4028
|
+
// INSERT is still in flight lets the reload revert these values, and the chained UPDATE then persists
|
|
4029
|
+
// the stale 'Running' row (the same race fixed in the agent-run-step queue and action-execution-log).
|
|
4030
|
+
// The model call between create and update almost always covers this; a fast-failing prompt could not.
|
|
4031
|
+
await this._promptRunSaveChains.get(promptRun);
|
|
3918
4032
|
promptRun.CompletedAt = endTime;
|
|
3919
4033
|
promptRun.ExecutionTimeMS = executionTimeMS;
|
|
3920
4034
|
// Determine what to save as the result
|
|
@@ -4058,7 +4172,8 @@ export class AIPromptRunner {
|
|
|
4058
4172
|
type: e.Type,
|
|
4059
4173
|
value: e.Value
|
|
4060
4174
|
})) || [],
|
|
4061
|
-
validationDecision: this.getValidationDecisionDescription(parsedResult.validationResult?.Success || false, validationAttempts.length, promptRun.ValidationBehavior || 'Warn')
|
|
4175
|
+
validationDecision: this.getValidationDecisionDescription(parsedResult.validationResult?.Success || false, validationAttempts.length, promptRun.ValidationBehavior || 'Warn'),
|
|
4176
|
+
jsonRepairInfo: promptRun._jsonRepairInfo || null
|
|
4062
4177
|
});
|
|
4063
4178
|
}
|
|
4064
4179
|
else {
|
|
@@ -4068,6 +4183,12 @@ export class AIPromptRunner {
|
|
|
4068
4183
|
promptRun.FinalValidationPassed = parsedResult.validationResult?.Success !== false;
|
|
4069
4184
|
promptRun.LastAttemptAt = endTime;
|
|
4070
4185
|
promptRun.TotalRetryDurationMS = 0;
|
|
4186
|
+
// Even without validation, persist JSON repair info if a repair occurred
|
|
4187
|
+
if (promptRun._jsonRepairInfo) {
|
|
4188
|
+
promptRun.ValidationSummary = JSON.stringify({
|
|
4189
|
+
jsonRepairInfo: promptRun._jsonRepairInfo
|
|
4190
|
+
});
|
|
4191
|
+
}
|
|
4071
4192
|
}
|
|
4072
4193
|
// Set Success flag based on validation result
|
|
4073
4194
|
promptRun.Success = modelResult.success && (parsedResult.validationResult?.Success !== false);
|
|
@@ -4093,28 +4214,10 @@ export class AIPromptRunner {
|
|
|
4093
4214
|
if (promptRun.Cost !== undefined) {
|
|
4094
4215
|
promptRun.TotalCost = promptRun.Cost;
|
|
4095
4216
|
}
|
|
4096
|
-
|
|
4097
|
-
|
|
4098
|
-
|
|
4099
|
-
|
|
4100
|
-
try {
|
|
4101
|
-
if (promptRun.LatestResult?.CompleteMessage) {
|
|
4102
|
-
errorMsg = typeof promptRun.LatestResult.CompleteMessage === 'string'
|
|
4103
|
-
? promptRun.LatestResult.CompleteMessage
|
|
4104
|
-
: String(promptRun.LatestResult.CompleteMessage);
|
|
4105
|
-
}
|
|
4106
|
-
}
|
|
4107
|
-
catch (msgError) {
|
|
4108
|
-
errorMsg = 'Error accessing error message';
|
|
4109
|
-
}
|
|
4110
|
-
this.logError(`Failed to update AIPromptRun with results: ${errorMsg}`, {
|
|
4111
|
-
category: 'PromptRunUpdate',
|
|
4112
|
-
metadata: {
|
|
4113
|
-
promptRunId: promptRun.ID,
|
|
4114
|
-
updateError: errorMsg
|
|
4115
|
-
}
|
|
4116
|
-
});
|
|
4117
|
-
}
|
|
4217
|
+
// Finalize fire-and-forget. Chains after the initial INSERT (same entity instance) so the
|
|
4218
|
+
// 'Running' INSERT can never overwrite this finalized 'Completed'/'Failed' state. The
|
|
4219
|
+
// execution flow does NOT await this — see queuePromptRunSave / WaitForPendingPromptRunSaves.
|
|
4220
|
+
this.queuePromptRunSave(promptRun);
|
|
4118
4221
|
}
|
|
4119
4222
|
catch (error) {
|
|
4120
4223
|
this.logError(error, {
|