@memberjunction/ai-prompts 6.1.4 → 6.2.0-edge.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,16 +1,18 @@
1
- import { BaseLLM, ChatParams, ChatMessageRole, GetAIAPIKey, ErrorAnalyzer, ResolveFileInputStrategy, EncodeToolTurnsAsText } from '@memberjunction/ai';
1
+ import { BaseLLM, ChatParams, ChatMessageRole, ErrorAnalyzer, ResolveFileInputStrategy, EncodeToolTurnsAsText } from '@memberjunction/ai';
2
+ import { BaseModelRunner, } from './BaseModelRunner.js';
2
3
  import { GetToolCallingDecision, GetToolCallingMode, RecordToolCallingDecision, RecordToolCallingMode, ResolveNativeToolCalling } from './nativeToolCallingGate.js';
3
4
  import { AIModelRunner } from './AIModelRunner.js';
4
5
  import { AIModelSelectionInfo } from '@memberjunction/ai-core-plus';
5
- import { BaseEntitySaveQueue, LogErrorEx, LogStatus, LogStatusEx, IsVerboseLoggingEnabled, Metadata } from '@memberjunction/core';
6
+ import { LogStatus, IsVerboseLoggingEnabled, Metadata } from '@memberjunction/core';
6
7
  import { CleanJSON, RepairJSONEscaping, MJGlobal, JSONValidator, ValidationResult, ValidationErrorInfo, ValidationErrorType, UUIDsEqual, NormalizeUUID } from '@memberjunction/global';
7
- import { CredentialEngine } from '@memberjunction/credentials';
8
8
  import { TemplateEngineServer } from '@memberjunction/templates';
9
9
  import { ExecutionPlanner } from './ExecutionPlanner.js';
10
10
  import { AIPromptTimeoutError } from './AIPromptTimeoutError.js';
11
+ import { ParseManifestEntryMime, TrimSpacesAndTabs } from './linearTextScan.js';
11
12
  import { AIEngine } from '@memberjunction/aiengine';
12
13
  import { AIEngineBase } from '@memberjunction/ai-engine-base';
13
14
  import { SystemPlaceholderManager } from '@memberjunction/ai-core-plus';
15
+ import { ResolvePromptRunUserID } from '@memberjunction/ai-core-plus';
14
16
  // json5 is a CJS module: under this package's ESM output its import namespace has no
15
17
  // `parse` — only the default export does. `import * as JSON5` made JSON5.parse
16
18
  // undefined at runtime ("JSON5.parse is not a function"), silently disabling the
@@ -30,7 +32,17 @@ function mimeFromBlockType(type) {
30
32
  default: return 'application/octet-stream';
31
33
  }
32
34
  }
33
- export class AIPromptRunner {
35
+ export class AIPromptRunner extends BaseModelRunner {
36
+ /**
37
+ * The model type this runner requires. Returns 'LLM' for AIPromptRunner.
38
+ */
39
+ get RequiredModelType() {
40
+ return 'LLM';
41
+ }
42
+ /** Keeps this runner's uncategorized errors logged under `AIPromptRunner`, as before the base class existed. */
43
+ get DefaultLogCategory() {
44
+ return 'AIPromptRunner';
45
+ }
34
46
  /**
35
47
  * Process-wide cache of parsed `OutputExample` JSON, keyed by the raw example string.
36
48
  * A prompt's OutputExample is a static string reused across every run and every validation
@@ -46,29 +58,8 @@ export class AIPromptRunner {
46
58
  * already been selected. See the DECISION note in {@link selectModelWithAPIKeyTracked}.
47
59
  */
48
60
  static { this.NOT_EVALUATED_REASON = 'Not evaluated (a higher-priority candidate was already selected; set AIPromptParams.forceFullModelEvaluation to probe all)'; }
49
- /**
50
- * Optional metadata provider override. Callers should set
51
- * `instance.Provider = providerToUse` before invoking run methods
52
- * in multi-provider contexts. Falls back to the global default provider when unset.
53
- */
54
- get Provider() {
55
- return this._provider ?? this._metadata;
56
- }
57
- set Provider(value) {
58
- this._provider = value;
59
- }
60
61
  constructor() {
61
- this._provider = null;
62
- /**
63
- * Fire-and-forget AIPromptRun persistence. Prompt-run logging never blocks the execution path on a
64
- * DB round-trip; the shared {@link BaseEntitySaveQueue} sequences saves for the SAME entity (the
65
- * initial 'Running' INSERT always completes before the finalize UPDATE, and the finalize mutation
66
- * runs INSIDE the post-INSERT task so a slow INSERT can never clobber the finalized row). Failures
67
- * stay in this runner's structured log stream via the queue's `onError` hook.
68
- */
69
- this._promptRunQueue = new BaseEntitySaveQueue({
70
- onError: (message) => this.logError(message, { category: 'PromptRunSave' }),
71
- });
62
+ super();
72
63
  this._metadata = this._provider ?? new Metadata();
73
64
  this._templateEngine = TemplateEngineServer.Instance;
74
65
  this._executionPlanner = new ExecutionPlanner();
@@ -109,324 +100,6 @@ export class AIPromptRunner {
109
100
  get ModelRunner() {
110
101
  return this._modelRunner;
111
102
  }
112
- /**
113
- * Performs robust validation of an API key
114
- * @returns true if the API key is valid (not null, undefined, or empty/whitespace)
115
- */
116
- isValidAPIKey(apiKey) {
117
- if (apiKey === undefined || apiKey === null) {
118
- return false;
119
- }
120
- // Check if it's just whitespace
121
- const trimmed = apiKey.trim();
122
- return trimmed.length > 0;
123
- }
124
- /**
125
- * Internal logging helper that wraps LogStatusEx with verbose control
126
- * @param message The message to log
127
- * @param verboseOnly Whether this is a verbose-only message
128
- * @param params Optional prompt parameters for custom verbose check
129
- */
130
- logStatus(message, verboseOnly = false, params) {
131
- if (verboseOnly) {
132
- LogStatusEx({
133
- message,
134
- verboseOnly: true,
135
- isVerboseEnabled: () => params?.verbose === true || IsVerboseLoggingEnabled()
136
- });
137
- }
138
- else {
139
- LogStatus(message);
140
- }
141
- }
142
- /**
143
- * Helper method for enhanced error logging with metadata
144
- */
145
- logError(error, options) {
146
- let errorMessage = error instanceof Error ? error.message : error;
147
- const errorObj = error instanceof Error ? error : undefined;
148
- // Truncate extremely long error messages (like Groq's failed_generation JSON dumps)
149
- // Only truncate if maxErrorLength is explicitly set
150
- if (options?.maxErrorLength !== undefined && errorMessage.length > options.maxErrorLength) {
151
- errorMessage = errorMessage.substring(0, options.maxErrorLength) + '... [truncated]';
152
- }
153
- const metadata = {
154
- ...options?.metadata
155
- };
156
- // Add prompt information if available
157
- if (options?.prompt) {
158
- metadata.promptId = options.prompt.ID;
159
- metadata.promptName = options.prompt.Name;
160
- }
161
- // Add model information if available
162
- if (options?.model) {
163
- metadata.modelId = options.model.ID;
164
- metadata.modelName = options.model.Name;
165
- }
166
- LogErrorEx({
167
- message: errorMessage,
168
- error: errorObj,
169
- category: options?.category || 'AIPromptRunner',
170
- severity: options?.severity || 'error',
171
- metadata: Object.keys(metadata).length > 0 ? metadata : undefined
172
- });
173
- }
174
- /**
175
- * Checks if a model vendor is configured as an inference provider.
176
- * Delegates to the memoized {@link AIEngine.IsInferenceProvider} helper so the
177
- * "Inference Provider" vendor-type lookup happens once per engine load rather than on
178
- * every candidate in every selection pass.
179
- * @param modelVendor The model vendor to check
180
- * @returns true if the vendor is an inference provider
181
- */
182
- isInferenceProvider(modelVendor) {
183
- return AIEngine.Instance.IsInferenceProvider(modelVendor);
184
- }
185
- /**
186
- * Resolves credentials for AI model execution using a hierarchical resolution system.
187
- *
188
- * Resolution priority (highest to lowest):
189
- * 1. Per-request override: params.credentialId
190
- * 2. Prompt-Model specific: AIPromptModel.CredentialID
191
- * 3. Model-Vendor specific: AIModelVendor.CredentialID
192
- * 4. Vendor default: AIVendor.CredentialID
193
- * 5. Legacy: params.apiKeys[] array
194
- * 6. Legacy: AI_VENDOR_API_KEY__<DRIVER> environment variables
195
- *
196
- * IMPORTANT: When ANY credential ID is found (priorities 1-4), the system uses
197
- * the Credentials path and ignores legacy methods (priorities 5-6).
198
- *
199
- * @param driverClass - The driver class name (e.g., 'OpenAILLM')
200
- * @param promptId - The prompt ID for looking up AIPromptModel credentials
201
- * @param modelId - The model ID for looking up AIPromptModel and AIModelVendor credentials
202
- * @param vendorId - The vendor ID for looking up AIModelVendor and AIVendor credentials
203
- * @param params - The prompt execution parameters containing contextUser and optional credentialId
204
- * @returns The API key/configuration string to pass to the LLM constructor
205
- */
206
- async resolveCredentialForExecution(driverClass, promptId, modelId, vendorId, params) {
207
- const verbose = params.verbose === true || IsVerboseLoggingEnabled();
208
- // Priority 1: Per-request override - no failover, explicit choice
209
- if (params.credentialId) {
210
- return await this.resolveCredentialById(params.credentialId, 'per-request override', params, verbose);
211
- }
212
- // Ensure CredentialEngine is configured for binding lookups
213
- await CredentialEngine.Instance.Config(false, params.contextUser);
214
- // Priority 2: PromptModel bindings (most specific) - with failover
215
- if (promptId && modelId) {
216
- const promptModel = AIEngine.Instance.PromptModels.find(pm => UUIDsEqual(pm.PromptID, promptId) && UUIDsEqual(pm.ModelID, modelId));
217
- if (promptModel) {
218
- const bindings = AIEngine.Instance.GetCredentialBindingsForTarget('PromptModel', promptModel.ID);
219
- const result = await this.tryCredentialBindingsWithFailover(bindings, 'AICredentialBinding(PromptModel)', params, verbose);
220
- if (result)
221
- return result;
222
- }
223
- }
224
- // Priority 3: ModelVendor bindings - with failover
225
- if (modelId && vendorId) {
226
- const modelVendor = AIEngine.Instance.ModelVendorsByModelID.get(NormalizeUUID(modelId))
227
- ?.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
228
- if (modelVendor) {
229
- const bindings = AIEngine.Instance.GetCredentialBindingsForTarget('ModelVendor', modelVendor.ID);
230
- const result = await this.tryCredentialBindingsWithFailover(bindings, 'AICredentialBinding(ModelVendor)', params, verbose);
231
- if (result)
232
- return result;
233
- }
234
- }
235
- // Priority 4: Vendor bindings - with failover
236
- if (vendorId) {
237
- const bindings = AIEngine.Instance.GetCredentialBindingsForTarget('Vendor', vendorId);
238
- const result = await this.tryCredentialBindingsWithFailover(bindings, 'AICredentialBinding(Vendor)', params, verbose);
239
- if (result)
240
- return result;
241
- }
242
- // Priority 5: Type-based default credential
243
- // If the vendor declares a CredentialTypeID, try to find a default credential of that type
244
- if (vendorId) {
245
- const vendor = AIEngine.Instance.VendorsByID.get(NormalizeUUID(vendorId));
246
- if (vendor?.CredentialTypeID) {
247
- const defaultCredential = this.findDefaultCredentialByType(vendor.CredentialTypeID);
248
- if (defaultCredential) {
249
- const result = await this.tryResolveCredential(defaultCredential, 'type-based default', params, verbose);
250
- if (result)
251
- return result;
252
- }
253
- }
254
- }
255
- // No credential bindings found - fall back to legacy methods
256
- if (verbose) {
257
- this.logStatus(` Using legacy API key resolution for driver ${driverClass}`, true, params);
258
- }
259
- // Priority 6 & 7: Legacy apiKeys array and environment variables
260
- return GetAIAPIKey(driverClass, params.apiKeys, verbose);
261
- }
262
- /**
263
- * Attempts to resolve credentials from bindings with priority-based failover.
264
- * Tries each binding in priority order until one succeeds.
265
- */
266
- async tryCredentialBindingsWithFailover(bindings, source, params, verbose) {
267
- if (bindings.length === 0)
268
- return null;
269
- for (let i = 0; i < bindings.length; i++) {
270
- const binding = bindings[i];
271
- const credential = CredentialEngine.Instance.getCredentialById(binding.CredentialID);
272
- if (!credential) {
273
- if (verbose) {
274
- this.logStatus(` ⚠️ Credential ${binding.CredentialID} not found (priority ${binding.Priority}), trying next...`, true, params);
275
- }
276
- continue;
277
- }
278
- const result = await this.tryResolveCredential(credential, `${source} priority ${binding.Priority}`, params, verbose, i < bindings.length - 1 // hasMoreBindings
279
- );
280
- if (result)
281
- return result;
282
- }
283
- return null;
284
- }
285
- /**
286
- * Attempts to resolve a single credential, returning null on failure for failover support.
287
- */
288
- async tryResolveCredential(credential, source, params, verbose, hasMoreBindings = false) {
289
- try {
290
- // Check if credential is active and not expired
291
- if (!credential.IsActive) {
292
- if (verbose) {
293
- this.logStatus(` ⚠️ Credential "${credential.Name}" is inactive, trying next...`, true, params);
294
- }
295
- return null;
296
- }
297
- if (credential.ExpiresAt && new Date(credential.ExpiresAt) < new Date()) {
298
- if (verbose) {
299
- this.logStatus(` ⚠️ Credential "${credential.Name}" has expired, trying next...`, true, params);
300
- }
301
- return null;
302
- }
303
- // Resolve the credential values
304
- const resolved = await CredentialEngine.Instance.getCredential(credential.Name, {
305
- credentialId: credential.ID,
306
- contextUser: params.contextUser,
307
- subsystem: 'AIPromptRunner'
308
- });
309
- if (verbose) {
310
- this.logStatus(` 🔐 Using credential from ${source}: "${credential.Name}"`, true, params);
311
- }
312
- return JSON.stringify(resolved.values);
313
- }
314
- catch (error) {
315
- if (hasMoreBindings) {
316
- // More bindings to try - log warning and continue
317
- if (verbose) {
318
- this.logStatus(` ⚠️ Failed to resolve credential "${credential.Name}" from ${source}: ${error instanceof Error ? error.message : String(error)}, trying next...`, true, params);
319
- }
320
- return null;
321
- }
322
- else {
323
- // No more bindings - log error but still return null for legacy fallback
324
- this.logError(error instanceof Error ? error : new Error(String(error)), {
325
- category: 'CredentialResolution',
326
- severity: 'warning',
327
- metadata: {
328
- credentialId: credential.ID,
329
- credentialName: credential.Name,
330
- source
331
- },
332
- maxErrorLength: params.maxErrorLength
333
- });
334
- return null;
335
- }
336
- }
337
- }
338
- /**
339
- * Resolves a credential by its explicit ID (used for per-request override).
340
- * This does not support failover since it's an explicit choice.
341
- */
342
- async resolveCredentialById(credentialId, source, params, verbose) {
343
- await CredentialEngine.Instance.Config(false, params.contextUser);
344
- const credential = CredentialEngine.Instance.getCredentialById(credentialId);
345
- if (!credential) {
346
- throw new Error(`Credential with ID ${credentialId} not found`);
347
- }
348
- const resolved = await CredentialEngine.Instance.getCredential(credential.Name, {
349
- credentialId,
350
- contextUser: params.contextUser,
351
- subsystem: 'AIPromptRunner'
352
- });
353
- if (verbose) {
354
- this.logStatus(` 🔐 Using credential from ${source}: "${credential.Name}"`, true, params);
355
- }
356
- return JSON.stringify(resolved.values);
357
- }
358
- /**
359
- * Finds a default credential matching a specific credential type.
360
- */
361
- findDefaultCredentialByType(credentialTypeId) {
362
- const credentials = CredentialEngine.Instance.Credentials;
363
- return credentials.find(c => UUIDsEqual(c.CredentialTypeID, credentialTypeId) &&
364
- c.IsDefault === true &&
365
- c.IsActive === true &&
366
- (!c.ExpiresAt || new Date(c.ExpiresAt) > new Date())) || null;
367
- }
368
- /**
369
- * Checks if credentials are available for a given model-vendor combination.
370
- * This is a pre-flight check used during model selection to determine which
371
- * candidates have valid authentication configured.
372
- *
373
- * Checks the credential hierarchy:
374
- * 1. Per-request override: params.credentialId
375
- * 2. PromptModel bindings: AICredentialBinding WHERE BindingType='PromptModel'
376
- * 3. ModelVendor bindings: AICredentialBinding WHERE BindingType='ModelVendor'
377
- * 4. Vendor bindings: AICredentialBinding WHERE BindingType='Vendor'
378
- * 5. Type-based default: Credential.IsDefault=1 matching AIVendor.CredentialTypeID
379
- * 6. Legacy: params.apiKeys[] array
380
- * 7. Legacy: AI_VENDOR_API_KEY__<DRIVER> environment variables
381
- *
382
- * @param driverClass - The driver class name (e.g., 'OpenAILLM')
383
- * @param promptId - The prompt ID for looking up AIPromptModel bindings
384
- * @param modelId - The model ID for looking up AIPromptModel and AIModelVendor bindings
385
- * @param vendorId - The vendor ID for looking up AIModelVendor and AIVendor bindings
386
- * @param params - The prompt execution parameters
387
- * @returns true if credentials are available, false otherwise
388
- */
389
- hasCredentialsAvailable(driverClass, promptId, modelId, vendorId, params) {
390
- // Priority 1: Per-request override
391
- if (params?.credentialId) {
392
- // Assume valid if credential ID is provided - will be validated at execution time
393
- return true;
394
- }
395
- // Priority 2: PromptModel bindings
396
- if (promptId && modelId) {
397
- const promptModel = AIEngine.Instance.PromptModels.find(pm => UUIDsEqual(pm.PromptID, promptId) && UUIDsEqual(pm.ModelID, modelId));
398
- if (promptModel && AIEngine.Instance.HasCredentialBindings('PromptModel', promptModel.ID)) {
399
- return true;
400
- }
401
- }
402
- // Priority 3: ModelVendor bindings
403
- if (modelId && vendorId) {
404
- const modelVendor = AIEngine.Instance.ModelVendorsByModelID.get(NormalizeUUID(modelId))
405
- ?.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
406
- if (modelVendor && AIEngine.Instance.HasCredentialBindings('ModelVendor', modelVendor.ID)) {
407
- return true;
408
- }
409
- }
410
- // Priority 4: Vendor bindings
411
- if (vendorId) {
412
- if (AIEngine.Instance.HasCredentialBindings('Vendor', vendorId)) {
413
- return true;
414
- }
415
- }
416
- // Priority 5: Type-based default credential
417
- if (vendorId) {
418
- const vendor = AIEngine.Instance.VendorsByID.get(NormalizeUUID(vendorId));
419
- if (vendor?.CredentialTypeID) {
420
- const defaultCredential = this.findDefaultCredentialByType(vendor.CredentialTypeID);
421
- if (defaultCredential) {
422
- return true;
423
- }
424
- }
425
- }
426
- // Priority 6 & 7: Legacy methods - check if API key is available
427
- const apiKey = GetAIAPIKey(driverClass, params?.apiKeys, params?.verbose);
428
- return this.isValidAPIKey(apiKey);
429
- }
430
103
  /**
431
104
  * Executes an AI prompt with full support for templates, model selection, and validation.
432
105
  *
@@ -518,7 +191,7 @@ export class AIPromptRunner {
518
191
  // selected candidate's AIPromptModel bag. Resolving without those skips the two layers the
519
192
  // capability is normally declared on and silently inverts the decision.
520
193
  if (params.tools?.length) {
521
- const nativeDecision = this.resolveNativeToolCallingDecision(prompt, params, selection.model, selection.selectionInfo?.vendorSelected?.ID ?? params.override?.vendorId ?? null, selection.promptModelConfiguration);
194
+ const nativeDecision = this.ResolveNativeToolCallingDecision(prompt, params, selection.model, selection.selectionInfo?.vendorSelected?.ID ?? params.override?.vendorId ?? null, selection.promptModelConfiguration);
522
195
  params.data = {
523
196
  ...(params.data ?? {}),
524
197
  _NATIVE_TOOL_CALLING: nativeDecision.useNativeTools,
@@ -533,8 +206,10 @@ export class AIPromptRunner {
533
206
  this.logStatus(` Using system prompt override for prompt "${prompt.Name}" (bypassing hierarchical template rendering)`, true, params);
534
207
  }
535
208
  else {
536
- // Render all child prompt templates recursively
537
- childTemplateRenderingResult = await this.renderChildPromptTemplates(params.childPrompts, params, params.cancellationToken);
209
+ // Render all child prompt templates recursively (or reuse pre-rendered templates)
210
+ childTemplateRenderingResult = params.PreRenderedChildTemplates
211
+ ? { renderedTemplates: params.PreRenderedChildTemplates }
212
+ : await this.renderChildPromptTemplates(params.childPrompts, params, params.cancellationToken);
538
213
  // Render the parent prompt with child templates embedded
539
214
  renderedPromptText = await this.renderPromptWithChildTemplates(prompt, params, childTemplateRenderingResult.renderedTemplates);
540
215
  }
@@ -678,7 +353,7 @@ export class AIPromptRunner {
678
353
  let promptModelConfiguration = existingSelection?.promptModelConfiguration;
679
354
  let allCandidates = existingSelection?.allCandidates ?? [];
680
355
  // Credential probes already done during selection — reused by failover so it doesn't
681
- // recompute hasCredentialsAvailable for the prefix it walks before the selected candidate.
356
+ // recompute HasCredentialsAvailable for the prefix it walks before the selected candidate.
682
357
  let credentialAvailability = existingSelection?.credentialAvailability;
683
358
  if (!selectedModel) {
684
359
  // Determine which prompt to use for model selection
@@ -706,7 +381,7 @@ export class AIPromptRunner {
706
381
  throw new Error('Prompt execution was cancelled after model selection');
707
382
  }
708
383
  // Use existing prompt run if provided (hierarchical case) or create new one
709
- const promptRun = existingPromptRun || await this.createPromptRun(prompt, selectedModel, params, renderedPromptText, startTime, params.override?.vendorId, modelSelectionInfo);
384
+ const promptRun = existingPromptRun || await this.createPromptRun(prompt, selectedModel, params, renderedPromptText, startTime, params.override?.vendorId, modelSelectionInfo, params.RunType);
710
385
  // Check for cancellation before model execution
711
386
  if (params.cancellationToken?.aborted) {
712
387
  throw new Error('Prompt execution was cancelled before model execution');
@@ -808,22 +483,43 @@ export class AIPromptRunner {
808
483
  modelSelectionPrompt = params.modelSelectionPrompt;
809
484
  this.logStatus(` Using prompt "${modelSelectionPrompt.Name}" for model selection in parallel execution`, true, params);
810
485
  }
486
+ // Apply this runner's model-type floor to everything the planner may choose from: a prompt
487
+ // typed differently fails here, and neither the model pool nor the prompt's bindings can admit
488
+ // a model of another type (the planner's own filters treat an empty AIModelTypeID as "any").
489
+ this.AssertPromptMatchesRequiredType(prompt);
490
+ if (modelSelectionPrompt !== prompt) {
491
+ this.AssertPromptMatchesRequiredType(modelSelectionPrompt);
492
+ }
493
+ const requiredTypeId = this.RequiredModelTypeID();
494
+ const typedModels = AIEngine.Instance.Models.filter(m => UUIDsEqual(m.AIModelTypeID, requiredTypeId));
811
495
  // Get prompt-specific model associations using the model selection prompt
812
496
  const promptModels = AIEngine.Instance.PromptModels.filter((pm) => UUIDsEqual(pm.PromptID, modelSelectionPrompt.ID) &&
813
497
  (pm.Status === 'Active' || pm.Status === 'Preview') &&
814
- (!params.configurationId || !pm.ConfigurationID || UUIDsEqual(pm.ConfigurationID, params.configurationId)));
498
+ (!params.configurationId || !pm.ConfigurationID || UUIDsEqual(pm.ConfigurationID, params.configurationId)) &&
499
+ UUIDsEqual(AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID))?.AIModelTypeID, requiredTypeId));
815
500
  // Create execution plan using the modelSelectionPrompt for model configurations
816
- executionTasks = this._executionPlanner.createExecutionPlan(modelSelectionPrompt, promptModels, AIEngine.Instance.Models, renderedPromptText, params.contextUser, params.configurationId, params.conversationMessages, params.templateMessageRole || 'system');
501
+ executionTasks = this._executionPlanner.createExecutionPlan(modelSelectionPrompt, promptModels, typedModels, renderedPromptText, params.contextUser, params.configurationId, params.conversationMessages, params.templateMessageRole || 'system');
817
502
  }
818
503
  if (executionTasks.length === 0) {
819
504
  throw new Error(`No execution tasks created for parallel execution of prompt ${prompt.Name}`);
820
505
  }
506
+ const parallelUserId = ResolvePromptRunUserID({ UserID: params.UserID, ContextUser: params.contextUser }) ?? undefined;
507
+ for (const task of executionTasks) {
508
+ task.AgentID = params.agentId;
509
+ task.UserID = parallelUserId;
510
+ }
821
511
  // Check for cancellation before executing tasks
822
512
  if (params.cancellationToken?.aborted) {
823
513
  throw new Error('Parallel execution was cancelled before task execution');
824
514
  }
825
- // Execute tasks in parallel
826
- const parallelResult = await this.ParallelCoordinator.executeTasksInParallel(params, executionTasks, undefined, undefined, params.cancellationToken);
515
+ // 1. Create (or reuse existingPromptRun) the consolidated parent BEFORE parallel tasks run (§5.1).
516
+ // Use the first task's model for the initial ModelID; after selection, set ModelID/VendorID to the selected arm's.
517
+ const initialModel = existingSelection?.model || executionTasks[0].model;
518
+ const consolidatedPromptRun = existingPromptRun || await this.createPromptRun(prompt, initialModel, params, renderedPromptText, startTime, params.override?.vendorId, existingSelection?.selectionInfo, 'ParallelParent');
519
+ consolidatedPromptRun.RunType = 'ParallelParent';
520
+ consolidatedPromptRun.WasSelectedResult = false;
521
+ // Execute tasks in parallel - pass consolidatedPromptRun.ID as parentPromptRunId
522
+ const parallelResult = await this.ParallelCoordinator.executeTasksInParallel(params, executionTasks, undefined, consolidatedPromptRun.ID, params.cancellationToken);
827
523
  if (!parallelResult.success) {
828
524
  throw new Error(`Parallel execution failed: ${parallelResult.errors.join(', ')}`);
829
525
  }
@@ -833,17 +529,30 @@ export class AIPromptRunner {
833
529
  throw new Error(`No successful results from parallel execution`);
834
530
  }
835
531
  let selectedResult = successfulResults[0]; // Default to first
836
- // Use result selector if configured
532
+ // Use result selector if configured - pass consolidatedPromptRun.ID and contextUser
837
533
  if (successfulResults.length > 1 && prompt.ResultSelectorPromptID) {
838
534
  const selectionConfig = {
839
535
  method: 'PromptSelector',
840
536
  selectorPromptId: prompt.ResultSelectorPromptID,
841
537
  };
842
- const aiSelectedResult = await this.ParallelCoordinator.selectBestResult(successfulResults, selectionConfig, undefined, params.cancellationToken);
538
+ const aiSelectedResult = await this.ParallelCoordinator.selectBestResult(successfulResults, selectionConfig, consolidatedPromptRun.ID, params.cancellationToken, params.contextUser);
843
539
  if (aiSelectedResult) {
844
540
  selectedResult = aiSelectedResult;
845
541
  }
846
542
  }
543
+ // Update parent with selected arm's model and vendor
544
+ consolidatedPromptRun.ModelID = selectedResult.task.model.ID;
545
+ if (selectedResult.task.vendorId) {
546
+ consolidatedPromptRun.VendorID = selectedResult.task.vendorId;
547
+ }
548
+ else if (selectedResult.task.promptModel?.VendorID) {
549
+ consolidatedPromptRun.VendorID = selectedResult.task.promptModel.VendorID;
550
+ }
551
+ // Ensure selectedResult child is marked WasSelectedResult = true
552
+ if (selectedResult.promptRun) {
553
+ selectedResult.promptRun.WasSelectedResult = true;
554
+ await selectedResult.promptRun.Save();
555
+ }
847
556
  // Calculate total tokens and costs from all parallel executions
848
557
  let totalPromptTokens = 0;
849
558
  let totalCompletionTokens = 0;
@@ -866,9 +575,6 @@ export class AIPromptRunner {
866
575
  }
867
576
  }
868
577
  }
869
- // Use existing prompt run if provided (hierarchical case) or create new one
870
- // Use the model selection info if provided (from hierarchical execution)
871
- const consolidatedPromptRun = existingPromptRun || await this.createPromptRun(prompt, selectedResult.task.model, params, renderedPromptText, startTime, params.override?.vendorId, existingSelection?.selectionInfo);
872
578
  // Update with parallel execution metadata
873
579
  const endTime = new Date();
874
580
  consolidatedPromptRun.CompletedAt = endTime;
@@ -890,9 +596,8 @@ export class AIPromptRunner {
890
596
  // prices the full input including cached tokens rather than dropping them.
891
597
  consolidatedPromptRun.TokensCacheRead = selectedResultUsage.cacheReadTokens ?? 0;
892
598
  consolidatedPromptRun.TokensCacheWrite = selectedResultUsage.cacheWriteTokens ?? 0;
893
- if (selectedResultUsage.cost !== undefined) {
894
- consolidatedPromptRun.Cost = selectedResultUsage.cost;
895
- }
599
+ // NOTE (§5.1): On the parent, do NOT assign Cost from the selected arm. Leave Cost = null.
600
+ // TotalCost is maintained by server-side TriggerParentCostRollup.
896
601
  if (selectedResultUsage.costCurrency !== undefined) {
897
602
  consolidatedPromptRun.CostCurrency = selectedResultUsage.costCurrency;
898
603
  }
@@ -914,18 +619,19 @@ export class AIPromptRunner {
914
619
  messages: params.conversationMessages || [],
915
620
  });
916
621
  }
917
- // For parallel execution, set rollup fields to match totals (no child execution to roll up)
622
+ // For parallel execution, set rollup fields to match totals
918
623
  consolidatedPromptRun.TokensPromptRollup = totalPromptTokens;
919
624
  consolidatedPromptRun.TokensCompletionRollup = totalCompletionTokens;
920
625
  consolidatedPromptRun.TokensUsedRollup = totalPromptTokens + totalCompletionTokens;
921
626
  consolidatedPromptRun.TokensCacheReadRollup = totalCacheReadTokens;
922
627
  consolidatedPromptRun.TokensCacheWriteRollup = totalCacheWriteTokens;
923
628
  if (hasCost) {
629
+ consolidatedPromptRun.DescendantCost = totalCost;
924
630
  consolidatedPromptRun.TotalCost = totalCost;
925
631
  }
926
632
  // Set Status and WasSelectedResult for parallel execution
927
633
  consolidatedPromptRun.Status = parallelResult.successCount > 0 ? 'Completed' : 'Failed';
928
- consolidatedPromptRun.WasSelectedResult = true; // This is the consolidated result chosen by judge
634
+ consolidatedPromptRun.WasSelectedResult = false; // WasSelectedResult stays on the selected child, not the parent
929
635
  // Persist the consolidated run fire-and-forget; the finalize UPDATE chains after its INSERT via
930
636
  // the save queue. These fields are set after all the awaited parallel work, so the INSERT has long
931
637
  // landed — a plain Update (no post-INSERT callback) is race-safe here.
@@ -1060,6 +766,24 @@ export class AIPromptRunner {
1060
766
  * @param cancellationToken - Cancellation token for aborting rendering
1061
767
  * @returns Promise with rendered templates map
1062
768
  */
769
+ /**
770
+ * Render a set of child prompt templates WITHOUT executing anything, returning the rendered text
771
+ * keyed by each child's parent placeholder — exactly what the hierarchical execution path embeds
772
+ * into the parent template.
773
+ *
774
+ * Exposed for callers that need a child's rendered text before the run: the loop agent uses it to
775
+ * relocate a volatile specialization into the trailing runtime-state message (see
776
+ * `ResolveSpecializationPlacement` in `@memberjunction/ai-agents`) while the system prompt renders a
777
+ * stub in its place. Rendering is deterministic for the same inputs, so a subsequent execution of
778
+ * the same params reproduces the same text.
779
+ *
780
+ * @param childPrompts The child prompt params, as they would be passed in `AIPromptParams.childPrompts`.
781
+ * @param params The parent params (context user, data, template data) the children render against.
782
+ * @param cancellationToken Optional abort signal.
783
+ */
784
+ async RenderChildPromptTemplates(childPrompts, params, cancellationToken) {
785
+ return this.renderChildPromptTemplates(childPrompts, params, cancellationToken);
786
+ }
1063
787
  async renderChildPromptTemplates(childPrompts, params, cancellationToken) {
1064
788
  if (!childPrompts || childPrompts.length === 0) {
1065
789
  return {
@@ -1268,7 +992,7 @@ export class AIPromptRunner {
1268
992
  }
1269
993
  /**
1270
994
  * Selects the appropriate AI model based on prompt configuration and parameters.
1271
- * Uses the unified buildModelVendorCandidates method to create an ordered list of candidates,
995
+ * Uses the unified BuildModelVendorCandidates method to create an ordered list of candidates,
1272
996
  * then selects the first one with an available API key.
1273
997
  */
1274
998
  async selectModel(prompt, explicitModelId, contextUser, configurationId, vendorId, params) {
@@ -1295,7 +1019,7 @@ export class AIPromptRunner {
1295
1019
  configurationName = configuration?.Name;
1296
1020
  }
1297
1021
  // Build unified list of model-vendor candidates
1298
- const candidates = this.buildModelVendorCandidates(prompt, explicitModelId, configurationId, vendorId, params.verbose);
1022
+ const candidates = this.BuildModelVendorCandidates(prompt, explicitModelId, configurationId, vendorId, params?.verbose);
1299
1023
  // Track all models considered for selection info
1300
1024
  const modelsConsidered = [];
1301
1025
  if (candidates.length === 0) {
@@ -1422,525 +1146,6 @@ export class AIPromptRunner {
1422
1146
  };
1423
1147
  }
1424
1148
  }
1425
- /**
1426
- * Builds a unified, ordered list of model-vendor candidates based on all selection criteria.
1427
- * Uses a 3-phase approach to properly handle SelectionStrategy='Specific' with AIPromptModel priorities.
1428
- *
1429
- * Phase 1: Handle explicit model ID (highest priority)
1430
- * Phase 2: Check if SelectionStrategy='Specific' with AIPromptModel entries - use ONLY those with AIPromptModel priorities
1431
- * Phase 3: Use general selection strategy (fallback) - blended priorities from legacy behavior
1432
- *
1433
- * @param prompt - The AI prompt with selection criteria
1434
- * @param explicitModelId - Explicitly specified model ID (highest priority)
1435
- * @param configurationId - Configuration ID for filtering
1436
- * @param preferredVendorId - Preferred vendor ID
1437
- * @returns Ordered array of model-vendor candidates (highest priority first)
1438
- */
1439
- buildModelVendorCandidates(prompt, explicitModelId, configurationId, preferredVendorId, verbose) {
1440
- // PHASE 1: Handle explicit model ID (highest priority)
1441
- if (explicitModelId) {
1442
- return this.buildCandidatesForExplicitModel(explicitModelId, prompt, preferredVendorId);
1443
- }
1444
- // PHASE 2: SelectionStrategy='Specific' - Use explicit AIPromptModel configuration
1445
- if (prompt.SelectionStrategy === 'Specific') {
1446
- return this.buildCandidatesForSpecificStrategy(prompt, configurationId, verbose);
1447
- }
1448
- // PHASE 3: Build candidates with configuration-aware fallback hierarchy
1449
- // (SelectionStrategy='Default' or 'ByPower')
1450
- return this.buildCandidatesForGeneralSelection(prompt, configurationId, preferredVendorId, verbose);
1451
- }
1452
- /**
1453
- * PHASE 1: Build candidates for explicitly specified model ID.
1454
- * Returns candidates for the single model if it's active and compatible.
1455
- */
1456
- buildCandidatesForExplicitModel(explicitModelId, prompt, preferredVendorId) {
1457
- const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(explicitModelId));
1458
- if (!model || !model.IsActive) {
1459
- return [];
1460
- }
1461
- // Check model type compatibility
1462
- if (prompt.AIModelTypeID && !UUIDsEqual(model.AIModelTypeID, prompt.AIModelTypeID)) {
1463
- return [];
1464
- }
1465
- const candidates = this.createCandidatesForModel(model, 20000, 'explicit', preferredVendorId);
1466
- candidates.sort((a, b) => b.priority - a.priority);
1467
- return candidates;
1468
- }
1469
- /**
1470
- * PHASE 2: Build candidates for 'Specific' selection strategy.
1471
- * Uses AIPromptModel configuration with clean ranking:
1472
- * 1. Config-matching models first (by priority DESC)
1473
- * 2. Then universal (null config) models (by priority DESC)
1474
- */
1475
- buildCandidatesForSpecificStrategy(prompt, configurationId, verbose) {
1476
- // Get all active AIPromptModel records for this prompt
1477
- const allPromptModels = AIEngine.Instance.PromptModels.filter(pm => UUIDsEqual(pm.PromptID, prompt.ID) && (pm.Status === 'Active' || pm.Status === 'Preview'));
1478
- // Filter by configuration matching rules
1479
- const promptModels = this.filterPromptModelsByConfiguration(allPromptModels, configurationId);
1480
- // Sort: config-specific before universal, then by priority DESC within each group
1481
- const sortedPromptModels = this.sortPromptModelsForSpecificStrategy(promptModels, configurationId);
1482
- // Build candidates maintaining order
1483
- const candidates = this.buildCandidatesFromPromptModels(sortedPromptModels);
1484
- // If RequireSpecificModels is true (or no candidates at all), enforce strict behavior
1485
- if (candidates.length === 0 && prompt.RequireSpecificModels) {
1486
- const configInfo = configurationId ? ` with configuration "${configurationId}"` : '';
1487
- throw new Error(`SelectionStrategy is 'Specific' but no valid AIPromptModel candidates found for prompt "${prompt.Name}"${configInfo}. ` +
1488
- `Please configure AIPromptModel records for this prompt.`);
1489
- }
1490
- // When RequireSpecificModels is false, append power-matched fallback candidates
1491
- // so that if none of the specific models have valid credentials, the system
1492
- // gracefully falls back to other available models at a similar power level.
1493
- if (!prompt.RequireSpecificModels) {
1494
- this.appendPowerMatchedFallbackCandidates(candidates, prompt, sortedPromptModels, verbose);
1495
- }
1496
- if (candidates.length === 0) {
1497
- const configInfo = configurationId ? ` with configuration "${configurationId}"` : '';
1498
- throw new Error(`SelectionStrategy is 'Specific' but no valid AIPromptModel candidates found for prompt "${prompt.Name}"${configInfo}. ` +
1499
- `Please configure AIPromptModel records for this prompt.`);
1500
- }
1501
- if (verbose) {
1502
- LogStatus(`Using SelectionStrategy='Specific' with ${sortedPromptModels.length} AIPromptModel entries, generated ${candidates.length} candidates`);
1503
- }
1504
- return candidates;
1505
- }
1506
- /**
1507
- * Appends fallback candidates from the global model pool, sorted by proximity to the
1508
- * average power rank of the originally configured models. This ensures that when
1509
- * specific models lack credentials, the fallback uses models of similar capability
1510
- * rather than defaulting to the most or least powerful available model.
1511
- *
1512
- * Fallback candidates are given lower priority than any specific candidate so
1513
- * configured models are always preferred when their credentials are available.
1514
- */
1515
- appendPowerMatchedFallbackCandidates(candidates, prompt, configuredPromptModels, verbose) {
1516
- // Compute target power rank from the configured models
1517
- const targetPowerRank = this.computeTargetPowerRank(configuredPromptModels);
1518
- // Get all active models matching the prompt's model type, excluding already-present models
1519
- const existingModelIds = new Set(candidates.map(c => c.model.ID));
1520
- const fallbackPool = AIEngine.Instance.Models.filter(m => m.IsActive &&
1521
- !existingModelIds.has(m.ID) &&
1522
- (!prompt.AIModelTypeID || UUIDsEqual(m.AIModelTypeID, prompt.AIModelTypeID)));
1523
- if (fallbackPool.length === 0)
1524
- return;
1525
- // Sort by proximity to the target power rank
1526
- const sorted = this.sortByPowerProximity(fallbackPool, targetPowerRank);
1527
- // Assign priorities below the lowest specific candidate
1528
- const lowestSpecificPriority = candidates.length > 0
1529
- ? Math.min(...candidates.map(c => c.priority))
1530
- : 1000;
1531
- const fallbackBasePriority = lowestSpecificPriority - 100;
1532
- sorted.forEach((model, index) => {
1533
- const modelCandidates = this.createCandidatesForModel(model, fallbackBasePriority - index * 10, 'power-match-fallback');
1534
- candidates.push(...modelCandidates);
1535
- });
1536
- if (verbose && sorted.length > 0) {
1537
- LogStatus(`Appended ${sorted.length} power-matched fallback models (target PowerRank: ${targetPowerRank}) ` +
1538
- `for prompt "${prompt.Name}" since RequireSpecificModels is false`);
1539
- }
1540
- }
1541
- /**
1542
- * Computes the target power rank from configured AIPromptModel records.
1543
- * Uses the weighted average (by priority) of the configured models' power ranks,
1544
- * so higher-priority models have more influence on the target.
1545
- * Falls back to simple average if priorities are all zero.
1546
- */
1547
- computeTargetPowerRank(promptModels) {
1548
- if (promptModels.length === 0)
1549
- return 0;
1550
- const modelsWithPower = promptModels
1551
- .map(pm => {
1552
- const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1553
- return { powerRank: model?.PowerRank ?? 0, priority: pm.Priority || 1 };
1554
- });
1555
- const totalWeight = modelsWithPower.reduce((sum, m) => sum + m.priority, 0);
1556
- if (totalWeight === 0) {
1557
- // All priorities are 0, use simple average
1558
- return Math.round(modelsWithPower.reduce((sum, m) => sum + m.powerRank, 0) / modelsWithPower.length);
1559
- }
1560
- const weightedSum = modelsWithPower.reduce((sum, m) => sum + m.powerRank * m.priority, 0);
1561
- return Math.round(weightedSum / totalWeight);
1562
- }
1563
- /**
1564
- * Sorts models by proximity to a target power rank (closest first).
1565
- * When two models are equidistant, the higher-powered one is preferred.
1566
- */
1567
- sortByPowerProximity(models, targetPowerRank) {
1568
- return [...models].sort((a, b) => {
1569
- const distA = Math.abs((a.PowerRank ?? 0) - targetPowerRank);
1570
- const distB = Math.abs((b.PowerRank ?? 0) - targetPowerRank);
1571
- if (distA !== distB)
1572
- return distA - distB; // Closer to target first
1573
- return (b.PowerRank ?? 0) - (a.PowerRank ?? 0); // Tie-break: higher power first
1574
- });
1575
- }
1576
- /**
1577
- * PHASE 3: Build candidates for general selection strategies ('Default' or 'ByPower').
1578
- * Uses configuration-aware fallback hierarchy with legacy blended priority calculation.
1579
- */
1580
- buildCandidatesForGeneralSelection(prompt, configurationId, preferredVendorId, verbose) {
1581
- const preferredVendorName = preferredVendorId ?
1582
- AIEngine.Instance.VendorsByID.get(NormalizeUUID(preferredVendorId))?.Name : undefined;
1583
- // Get prompt models for configuration
1584
- const promptModels = this.getPromptModelsForConfiguration(prompt, configurationId);
1585
- const candidates = [];
1586
- if (promptModels.length > 0) {
1587
- // Use prompt-specific models with blended priorities
1588
- this.addPromptSpecificCandidates(candidates, promptModels, preferredVendorId);
1589
- // Add configuration fallback candidates if needed
1590
- if (configurationId) {
1591
- this.addConfigurationFallbackCandidates(candidates, prompt, configurationId, preferredVendorId, verbose);
1592
- }
1593
- }
1594
- else if (this.hasAnyPromptModelBindings(prompt)) {
1595
- // Bindings exist for this prompt but none are Active/Preview (e.g. deliberately deactivated) —
1596
- // do NOT silently fall back to the global model pool, which would mask an intentional
1597
- // "no model available for this prompt" state. Leave candidates empty so the caller surfaces
1598
- // a "no suitable model found" failure instead of succeeding against an unrelated model.
1599
- }
1600
- else {
1601
- // No prompt-specific bindings were ever configured, use the general selection strategy
1602
- this.addStrategyBasedCandidates(candidates, prompt, preferredVendorName);
1603
- }
1604
- // Sort all candidates by priority (highest first)
1605
- candidates.sort((a, b) => b.priority - a.priority);
1606
- return candidates;
1607
- }
1608
- /**
1609
- * Helper: Filter prompt models by configuration matching rules.
1610
- * Supports configuration inheritance - includes models from the entire inheritance chain.
1611
- */
1612
- filterPromptModelsByConfiguration(allPromptModels, configurationId) {
1613
- if (configurationId) {
1614
- // Get the configuration inheritance chain
1615
- const chain = AIEngine.Instance.GetConfigurationChain(configurationId);
1616
- const chainIds = new Set(chain.map(c => NormalizeUUID(c.ID)));
1617
- // Include models matching any config in the chain, plus null-config (universal fallback)
1618
- return allPromptModels.filter(pm => (pm.ConfigurationID && chainIds.has(NormalizeUUID(pm.ConfigurationID))) ||
1619
- pm.ConfigurationID === null);
1620
- }
1621
- else {
1622
- // No config specified - only include null-config models
1623
- return allPromptModels.filter(pm => pm.ConfigurationID === null);
1624
- }
1625
- }
1626
- /**
1627
- * Helper: Sort prompt models for 'Specific' strategy.
1628
- * Respects configuration inheritance chain - child configs first, then parents, then null-config.
1629
- * Within each config level, sorts by priority DESC.
1630
- */
1631
- sortPromptModelsForSpecificStrategy(promptModels, configurationId) {
1632
- if (!configurationId) {
1633
- // No config specified - just sort by priority
1634
- return promptModels.sort((a, b) => (b.Priority || 0) - (a.Priority || 0));
1635
- }
1636
- // Get the configuration inheritance chain and create position map
1637
- const chain = AIEngine.Instance.GetConfigurationChain(configurationId);
1638
- const chainOrder = new Map(chain.map((c, index) => [c.ID, index]));
1639
- return promptModels.sort((a, b) => {
1640
- // Primary: Chain position (lower index = higher priority, null config = last)
1641
- const aChainPos = a.ConfigurationID ? (chainOrder.get(a.ConfigurationID) ?? 999) : 1000;
1642
- const bChainPos = b.ConfigurationID ? (chainOrder.get(b.ConfigurationID) ?? 999) : 1000;
1643
- if (aChainPos !== bChainPos) {
1644
- return aChainPos - bChainPos; // Lower chain position first (child before parent)
1645
- }
1646
- // Secondary: Higher priority first within same config level
1647
- return (b.Priority || 0) - (a.Priority || 0);
1648
- });
1649
- }
1650
- /**
1651
- * Helper: Build candidates from sorted AIPromptModel records.
1652
- * Expands VendorID=null to all vendors for that model.
1653
- */
1654
- buildCandidatesFromPromptModels(promptModels) {
1655
- const candidates = [];
1656
- for (let i = 0; i < promptModels.length; i++) {
1657
- const pm = promptModels[i];
1658
- // Compute priority as inverse of array position so highest-priority (first) gets the largest number
1659
- const computedPriority = promptModels.length - i;
1660
- const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1661
- if (!model || !model.IsActive)
1662
- continue;
1663
- if (pm.VendorID) {
1664
- // Specific vendor specified - create single candidate
1665
- const candidate = this.createCandidateForSpecificVendor(model, pm, computedPriority);
1666
- if (candidate) {
1667
- candidates.push(candidate);
1668
- }
1669
- }
1670
- else {
1671
- // No vendor specified - create candidates for all vendors
1672
- const vendorCandidates = this.createCandidatesForAllVendors(model, computedPriority);
1673
- candidates.push(...vendorCandidates);
1674
- }
1675
- }
1676
- return candidates;
1677
- }
1678
- /**
1679
- * Helper: Create candidate for specific vendor from AIPromptModel.
1680
- */
1681
- createCandidateForSpecificVendor(model, promptModel, computedPriority = 0) {
1682
- // Use the model's precomputed ModelVendors (grouped at engine load) instead of scanning
1683
- // the global ModelVendors array — model.ID === promptModel.ModelID here.
1684
- const modelVendor = model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, promptModel.VendorID) &&
1685
- mv.Status === 'Active' &&
1686
- this.isInferenceProvider(mv));
1687
- if (!modelVendor)
1688
- return null;
1689
- return {
1690
- model,
1691
- vendorId: modelVendor.VendorID,
1692
- vendorName: modelVendor.Vendor,
1693
- driverClass: modelVendor.DriverClass || model.DriverClass,
1694
- apiName: modelVendor.APIName || model.APIName,
1695
- supportsEffortLevel: modelVendor.SupportsEffortLevel ?? model.SupportsEffortLevel ?? false,
1696
- effortLevel: promptModel.EffortLevel ?? undefined, // Model-specific effort level override
1697
- promptModelConfiguration: promptModel.PromptConfigurationObject,
1698
- isPreferredVendor: false,
1699
- priority: computedPriority,
1700
- source: 'prompt-model'
1701
- };
1702
- }
1703
- /**
1704
- * Helper: Create candidates for all vendors of a model, sorted by vendor priority.
1705
- */
1706
- createCandidatesForAllVendors(model, computedPriority = 0) {
1707
- const vendors = model.ModelVendors
1708
- .filter(mv => mv.Status === 'Active' &&
1709
- this.isInferenceProvider(mv))
1710
- .sort((a, b) => (b.Priority || 0) - (a.Priority || 0));
1711
- const candidates = [];
1712
- for (const vendor of vendors) {
1713
- candidates.push({
1714
- model,
1715
- vendorId: vendor.VendorID,
1716
- vendorName: vendor.Vendor,
1717
- driverClass: vendor.DriverClass || model.DriverClass,
1718
- apiName: vendor.APIName || model.APIName,
1719
- supportsEffortLevel: vendor.SupportsEffortLevel ?? model.SupportsEffortLevel ?? false,
1720
- isPreferredVendor: false,
1721
- priority: computedPriority,
1722
- source: 'prompt-model'
1723
- });
1724
- }
1725
- // If no vendors found, use model defaults
1726
- if (candidates.length === 0 && model.DriverClass) {
1727
- candidates.push({
1728
- model,
1729
- driverClass: model.DriverClass,
1730
- apiName: model.APIName,
1731
- supportsEffortLevel: model.SupportsEffortLevel ?? false,
1732
- isPreferredVendor: false,
1733
- priority: computedPriority,
1734
- source: 'prompt-model'
1735
- });
1736
- }
1737
- return candidates;
1738
- }
1739
- /**
1740
- * Helper: true if this prompt has any AIPromptModel bindings at all, regardless of Status or
1741
- * ConfigurationID. Distinguishes "no bindings were ever configured" (general selection strategy
1742
- * should apply) from "bindings exist but are all Inactive" (no model should be selected).
1743
- */
1744
- hasAnyPromptModelBindings(prompt) {
1745
- return AIEngine.Instance.PromptModels.some(pm => UUIDsEqual(pm.PromptID, prompt.ID));
1746
- }
1747
- /**
1748
- * Helper: Get prompt models for configuration with inheritance chain fallback.
1749
- * Walks the configuration inheritance chain looking for prompt models.
1750
- * Returns models from the first config in the chain that has any, or falls back to null-config.
1751
- */
1752
- getPromptModelsForConfiguration(prompt, configurationId) {
1753
- if (configurationId) {
1754
- // Get the configuration inheritance chain (child -> parent -> grandparent -> ...)
1755
- const chain = AIEngine.Instance.GetConfigurationChain(configurationId);
1756
- // Walk the chain looking for prompt models
1757
- for (const config of chain) {
1758
- const promptModels = AIEngine.Instance.PromptModels.filter(pm => UUIDsEqual(pm.PromptID, prompt.ID) &&
1759
- (pm.Status === 'Active' || pm.Status === 'Preview') &&
1760
- UUIDsEqual(pm.ConfigurationID, config.ID));
1761
- if (promptModels.length > 0) {
1762
- return promptModels;
1763
- }
1764
- }
1765
- // No match in chain, fall back to NULL config models
1766
- LogStatus(`No models found in configuration chain for "${configurationId}", falling back to default models`);
1767
- }
1768
- // Return null-config (universal) models
1769
- return AIEngine.Instance.PromptModels.filter(pm => UUIDsEqual(pm.PromptID, prompt.ID) &&
1770
- (pm.Status === 'Active' || pm.Status === 'Preview') &&
1771
- !pm.ConfigurationID);
1772
- }
1773
- /**
1774
- * Helper: Add prompt-specific candidates with blended priorities (legacy behavior).
1775
- */
1776
- addPromptSpecificCandidates(candidates, promptModels, preferredVendorId) {
1777
- for (const pm of promptModels) {
1778
- const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1779
- if (model && model.IsActive) {
1780
- const modelCandidates = this.createCandidatesForModel(model, 5000, 'prompt-model', preferredVendorId, pm.Priority);
1781
- candidates.push(...modelCandidates);
1782
- }
1783
- }
1784
- }
1785
- /**
1786
- * Helper: Add configuration fallback candidates from the inheritance chain.
1787
- * Adds models from parent configs (with decreasing priority) and null-config models as final fallback.
1788
- */
1789
- addConfigurationFallbackCandidates(candidates, prompt, configurationId, preferredVendorId, verbose) {
1790
- const chain = AIEngine.Instance.GetConfigurationChain(configurationId);
1791
- // Add models from parent configs (skip index 0 which is the direct config, already handled)
1792
- for (let i = 1; i < chain.length; i++) {
1793
- const parentConfig = chain[i];
1794
- const parentModels = AIEngine.Instance.PromptModels.filter(pm => UUIDsEqual(pm.PromptID, prompt.ID) &&
1795
- (pm.Status === 'Active' || pm.Status === 'Preview') &&
1796
- UUIDsEqual(pm.ConfigurationID, parentConfig.ID));
1797
- if (parentModels.length > 0 && verbose) {
1798
- LogStatus(`Adding ${parentModels.length} models from parent config "${parentConfig.Name}" as fallback`);
1799
- }
1800
- for (const pm of parentModels) {
1801
- const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1802
- if (model && model.IsActive) {
1803
- // Decrease base priority for each level up the chain (3000, 2500, 2000, etc.)
1804
- const basePriority = 3000 - (i * 500);
1805
- const modelCandidates = this.createCandidatesForModel(model, basePriority, 'prompt-model', preferredVendorId, pm.Priority);
1806
- candidates.push(...modelCandidates);
1807
- }
1808
- }
1809
- }
1810
- // Finally add NULL config models (universal fallback) with lowest priority
1811
- const nullConfigModels = AIEngine.Instance.PromptModels.filter(pm => UUIDsEqual(pm.PromptID, prompt.ID) &&
1812
- (pm.Status === 'Active' || pm.Status === 'Preview') &&
1813
- !pm.ConfigurationID);
1814
- if (nullConfigModels.length > 0 && verbose) {
1815
- LogStatus(`Adding ${nullConfigModels.length} NULL configuration models as universal fallback`);
1816
- }
1817
- for (const pm of nullConfigModels) {
1818
- const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1819
- if (model && model.IsActive) {
1820
- const modelCandidates = this.createCandidatesForModel(model, 1000, // Lowest priority tier
1821
- 'prompt-model', preferredVendorId, pm.Priority);
1822
- candidates.push(...modelCandidates);
1823
- }
1824
- }
1825
- }
1826
- /**
1827
- * Helper: Add strategy-based candidates when no prompt models exist.
1828
- */
1829
- addStrategyBasedCandidates(candidates, prompt, preferredVendorName) {
1830
- let modelPool = this.getModelPoolForStrategy(prompt, preferredVendorName);
1831
- modelPool = this.sortModelPoolByStrategy(modelPool, prompt);
1832
- // Create candidates for each model in the pool
1833
- modelPool.forEach((model, index) => {
1834
- const basePriority = 1000 - index * 10; // Decrease priority by position
1835
- const source = prompt.SelectionStrategy === 'ByPower' ? 'power-rank' : 'model-type';
1836
- candidates.push(...this.createCandidatesForModel(model, basePriority, source));
1837
- });
1838
- }
1839
- /**
1840
- * Helper: Get model pool filtered for strategy.
1841
- */
1842
- getModelPoolForStrategy(prompt, preferredVendorName) {
1843
- return AIEngine.Instance.Models.filter(m => m.IsActive &&
1844
- (!prompt.AIModelTypeID || UUIDsEqual(m.AIModelTypeID, prompt.AIModelTypeID)) &&
1845
- (!preferredVendorName ||
1846
- m.ModelVendors.some(mv => mv.Status === 'Active' &&
1847
- mv.Vendor === preferredVendorName &&
1848
- this.isInferenceProvider(mv))));
1849
- }
1850
- /**
1851
- * Helper: Sort model pool by selection strategy.
1852
- */
1853
- sortModelPoolByStrategy(modelPool, prompt) {
1854
- if (prompt.SelectionStrategy === 'ByPower') {
1855
- return this.sortByPowerPreference(modelPool, prompt.PowerPreference);
1856
- }
1857
- else {
1858
- // Default strategy
1859
- const minPowerRank = prompt.MinPowerRank || 0;
1860
- return modelPool
1861
- .filter(m => m.PowerRank >= minPowerRank)
1862
- .sort((a, b) => b.PowerRank - a.PowerRank);
1863
- }
1864
- }
1865
- /**
1866
- * Helper: Sort models by power preference.
1867
- */
1868
- sortByPowerPreference(modelPool, powerPreference) {
1869
- const pool = [...modelPool];
1870
- switch (powerPreference) {
1871
- case 'Highest':
1872
- return pool.sort((a, b) => b.PowerRank - a.PowerRank);
1873
- case 'Lowest':
1874
- return pool.sort((a, b) => a.PowerRank - b.PowerRank);
1875
- case 'Balanced':
1876
- const avgPower = pool.reduce((sum, m) => sum + m.PowerRank, 0) / pool.length;
1877
- return pool.sort((a, b) => Math.abs(a.PowerRank - avgPower) - Math.abs(b.PowerRank - avgPower));
1878
- default:
1879
- return pool.sort((a, b) => b.PowerRank - a.PowerRank);
1880
- }
1881
- }
1882
- /**
1883
- * Helper: Create candidates for a model with AIModelVendor priorities (legacy behavior).
1884
- */
1885
- createCandidatesForModel(model, basePriority, source, preferredVendorId, promptModelPriority) {
1886
- const modelCandidates = [];
1887
- // Get all vendors for this model - filter for inference providers only.
1888
- // Uses the model's precomputed ModelVendors (grouped at engine load) rather than scanning
1889
- // the global ModelVendors array.
1890
- const modelVendors = model.ModelVendors
1891
- .filter(mv => mv.Status === 'Active' && this.isInferenceProvider(mv))
1892
- .sort((a, b) => b.Priority - a.Priority);
1893
- // First, add preferred vendor if it exists
1894
- if (preferredVendorId) {
1895
- const preferredVendor = modelVendors.find(mv => UUIDsEqual(mv.VendorID, preferredVendorId));
1896
- if (preferredVendor) {
1897
- modelCandidates.push({
1898
- model,
1899
- vendorId: preferredVendor.VendorID,
1900
- vendorName: preferredVendor.Vendor,
1901
- driverClass: preferredVendor.DriverClass || model.DriverClass,
1902
- apiName: preferredVendor.APIName || model.APIName,
1903
- supportsEffortLevel: preferredVendor.SupportsEffortLevel ?? model.SupportsEffortLevel ?? false,
1904
- isPreferredVendor: true,
1905
- priority: basePriority + 1000, // Boost priority for preferred vendor
1906
- source
1907
- });
1908
- }
1909
- }
1910
- // Then add other vendors in priority order
1911
- for (const vendor of modelVendors) {
1912
- if (!UUIDsEqual(vendor.VendorID, preferredVendorId)) {
1913
- modelCandidates.push({
1914
- model,
1915
- vendorId: vendor.VendorID,
1916
- vendorName: vendor.Vendor,
1917
- driverClass: vendor.DriverClass || model.DriverClass,
1918
- apiName: vendor.APIName || model.APIName,
1919
- supportsEffortLevel: vendor.SupportsEffortLevel ?? model.SupportsEffortLevel ?? false,
1920
- isPreferredVendor: false,
1921
- priority: basePriority + (vendor.Priority || 0),
1922
- source
1923
- });
1924
- }
1925
- }
1926
- // If no vendors found, add model with its default driver
1927
- if (modelCandidates.length === 0 && model.DriverClass) {
1928
- modelCandidates.push({
1929
- model,
1930
- driverClass: model.DriverClass,
1931
- apiName: model.APIName,
1932
- supportsEffortLevel: model.SupportsEffortLevel ?? false,
1933
- isPreferredVendor: false,
1934
- priority: basePriority,
1935
- source
1936
- });
1937
- }
1938
- // Apply prompt model priority if provided (legacy blended approach)
1939
- if (promptModelPriority !== undefined) {
1940
- modelCandidates.forEach(c => c.priority += promptModelPriority * 10);
1941
- }
1942
- return modelCandidates;
1943
- }
1944
1149
  /**
1945
1150
  * Creates a properly typed AIModelSelectionInfo instance.
1946
1151
  * TypeScript requires instantiating the class to get the getValidCandidates() method.
@@ -1968,7 +1173,7 @@ export class AIPromptRunner {
1968
1173
  // DECISION (performance): candidates are ordered by priority, and we only need the
1969
1174
  // highest-priority candidate that has working credentials. So once we find that first
1970
1175
  // hit, we STOP credential-probing the remaining candidates and record them as
1971
- // "not-evaluated" rather than running a `hasCredentialsAvailable` check (which does
1176
+ // "not-evaluated" rather than running a `HasCredentialsAvailable` check (which does
1972
1177
  // env-var lookups + binding scans) for every configured model on every prompt run.
1973
1178
  // The remaining candidates are still kept in `consideredModels` (and in the returned
1974
1179
  // `allCandidates` from selectModel, which is the FULL ordered list) so failover and the
@@ -2001,7 +1206,7 @@ export class AIPromptRunner {
2001
1206
  }
2002
1207
  else {
2003
1208
  // Check for credentials using hierarchical resolution
2004
- hasCredentials = this.hasCredentialsAvailable(candidate.driverClass, promptId, candidate.model.ID, candidate.vendorId, params);
1209
+ hasCredentials = this.HasCredentialsAvailable(candidate.driverClass, promptId, candidate.model.ID, candidate.vendorId, params);
2005
1210
  credentialCache.set(cacheKey, hasCredentials);
2006
1211
  }
2007
1212
  // Track this model as considered with availability status
@@ -2050,6 +1255,11 @@ export class AIPromptRunner {
2050
1255
  */
2051
1256
  buildNoModelFoundMessage(promptName, selectionInfo) {
2052
1257
  const base = `No suitable model found for prompt ${promptName}`;
1258
+ // A selection step that threw (for example the model-type floor) records its error here; show it
1259
+ // rather than the generic "no candidates" text. Every other reason keeps its detailed message below.
1260
+ if (selectionInfo?.selectionReason?.startsWith('Error during model selection:')) {
1261
+ return `${base}. ${selectionInfo.selectionReason}`;
1262
+ }
2053
1263
  if (!selectionInfo?.modelsConsidered || selectionInfo.modelsConsidered.length === 0) {
2054
1264
  return `${base}. No model-vendor candidates were available. Please ensure AI models are configured for this prompt.`;
2055
1265
  }
@@ -2067,244 +1277,6 @@ export class AIPromptRunner {
2067
1277
  }
2068
1278
  return `${base}. ${selectionInfo.selectionReason || 'Unknown reason'}`;
2069
1279
  }
2070
- /**
2071
- * Creates an AIPromptRun entity for execution tracking
2072
- */
2073
- /**
2074
- * Resolves the scalar inference parameters for a run: each value is the per-request override
2075
- * from `additionalParameters` when supplied, otherwise the prompt's configured default. This
2076
- * is the single source of truth for parameter precedence so {@link executeModel} (ChatParams)
2077
- * and {@link createPromptRun} (the persisted record) stay in lockstep. Stop sequences and
2078
- * assistant prefill are intentionally excluded — their representations differ per target.
2079
- */
2080
- resolveScalarInferenceParams(prompt, additionalParameters) {
2081
- const pick = (override, promptDefault) => override !== undefined ? override : (promptDefault != null ? promptDefault : undefined);
2082
- const ap = additionalParameters;
2083
- return {
2084
- temperature: pick(ap?.temperature, prompt.Temperature),
2085
- topP: pick(ap?.topP, prompt.TopP),
2086
- topK: pick(ap?.topK, prompt.TopK),
2087
- minP: pick(ap?.minP, prompt.MinP),
2088
- frequencyPenalty: pick(ap?.frequencyPenalty, prompt.FrequencyPenalty),
2089
- presencePenalty: pick(ap?.presencePenalty, prompt.PresencePenalty),
2090
- seed: pick(ap?.seed, prompt.Seed),
2091
- includeLogProbs: pick(ap?.includeLogProbs, prompt.IncludeLogProbs),
2092
- topLogProbs: pick(ap?.topLogProbs, prompt.TopLogProbs),
2093
- };
2094
- }
2095
- /**
2096
- * Awaits all in-flight prompt-run saves queued by this runner instance. The normal execution path
2097
- * does NOT call this — prompt-run persistence is intentionally fire-and-forget. Exposed for tests
2098
- * and for callers that need the AIPromptRun rows durably written before proceeding.
2099
- */
2100
- async WaitForPendingPromptRunSaves() {
2101
- await this._promptRunQueue.Flush();
2102
- }
2103
- async createPromptRun(prompt, model, params, systemPromptText, startTime, vendorId, modelSelectionInfo) {
2104
- const provider = params.provider ?? Metadata.Provider;
2105
- const promptRun = await provider.GetEntityObject('MJ: AI Prompt Runs', params.contextUser);
2106
- try {
2107
- promptRun.NewRecord();
2108
- promptRun.PromptID = prompt.ID;
2109
- promptRun.ModelID = model.ID;
2110
- // Attribute the run to the agent that caused it, when there is one. PromptID alone cannot do
2111
- // this: agents share agent-type-level prompts, so a parent and its sub-agent produce runs of
2112
- // the SAME prompt. See AIPromptParams.agentId for why this was previously always null.
2113
- if (params.agentId) {
2114
- promptRun.AgentID = params.agentId;
2115
- }
2116
- // Set ChildPromptID if this is a hierarchical execution with child prompts
2117
- if (params.childPrompts && params.childPrompts.length > 0) {
2118
- promptRun.ChildPromptID = params.childPrompts[0].childPrompt.prompt.ID;
2119
- }
2120
- // Set initial status and tracking fields
2121
- promptRun.Status = 'Running';
2122
- promptRun.Cancelled = false;
2123
- promptRun.CacheHit = false;
2124
- promptRun.StreamingEnabled = !!params.onStreaming;
2125
- promptRun.WasSelectedResult = false;
2126
- // Set model selection tracking fields
2127
- if (modelSelectionInfo) {
2128
- // Convert the rich entity objects to simple IDs/names for database storage
2129
- const dbSelectionInfo = {
2130
- configurationId: modelSelectionInfo.aiConfiguration?.ID,
2131
- configurationName: modelSelectionInfo.aiConfiguration?.Name,
2132
- modelsConsidered: modelSelectionInfo.modelsConsidered.map(mc => ({
2133
- modelId: mc.model.ID,
2134
- modelName: mc.model.Name,
2135
- vendorId: mc.vendor?.ID,
2136
- vendorName: mc.vendor?.Name || 'default',
2137
- priority: mc.priority,
2138
- available: mc.available,
2139
- unavailableReason: mc.unavailableReason
2140
- })),
2141
- modelSelected: modelSelectionInfo.modelSelected?.ID,
2142
- vendorSelected: modelSelectionInfo.vendorSelected?.ID,
2143
- selectionReason: modelSelectionInfo.selectionReason,
2144
- fallbackUsed: modelSelectionInfo.fallbackUsed,
2145
- selectionStrategy: modelSelectionInfo.selectionStrategy
2146
- };
2147
- promptRun.ModelSelection = JSON.stringify(dbSelectionInfo);
2148
- promptRun.SelectionStrategy = modelSelectionInfo.selectionStrategy || 'Default';
2149
- // Set ModelPowerRank if available
2150
- if (model.PowerRank != null) {
2151
- promptRun.ModelPowerRank = model.PowerRank;
2152
- }
2153
- }
2154
- // Set original model tracking for failover
2155
- promptRun.OriginalModelID = model.ID;
2156
- promptRun.OriginalRequestStartTime = startTime;
2157
- // Initialize failover tracking fields
2158
- promptRun.FailoverAttempts = 0;
2159
- promptRun.FailoverErrors = null;
2160
- promptRun.FailoverDurations = null;
2161
- promptRun.TotalFailoverDuration = 0;
2162
- // Check if model has pre-selected vendor info from selectModel
2163
- const modelWithVendor = model;
2164
- if (modelSelectionInfo) {
2165
- promptRun.VendorID = modelSelectionInfo.vendorSelected?.ID || vendorId || modelWithVendor._selectedVendorId;
2166
- }
2167
- else if (vendorId) {
2168
- // Explicit vendor ID provided
2169
- promptRun.VendorID = vendorId;
2170
- }
2171
- else if (modelWithVendor._selectedVendorId) {
2172
- // Use vendor selected during model selection (with API key verification)
2173
- promptRun.VendorID = modelWithVendor._selectedVendorId;
2174
- }
2175
- else {
2176
- // Fallback: grab the highest priority AI Model Vendor record for this model (inference providers only)
2177
- const modelVendors = model.ModelVendors
2178
- .filter((mv) => mv.Status === 'Active' && this.isInferenceProvider(mv))
2179
- .sort((a, b) => b.Priority - a.Priority);
2180
- if (modelVendors.length > 0) {
2181
- promptRun.VendorID = modelVendors[0].VendorID;
2182
- }
2183
- }
2184
- promptRun.ConfigurationID = params.configurationId;
2185
- promptRun.RunAt = startTime;
2186
- // Resolve and save the effort level used (same precedence as ChatParams resolution).
2187
- // EffortLevel is a numeric column with a CHECK (1-100), so a provider-named level such as
2188
- // 'xhigh' is deliberately not persisted here — it still reaches the driver via ChatParams.
2189
- if (typeof params.effortLevel === 'number') {
2190
- promptRun.EffortLevel = params.effortLevel;
2191
- }
2192
- else if (prompt.EffortLevel !== undefined && prompt.EffortLevel !== null) {
2193
- promptRun.EffortLevel = prompt.EffortLevel;
2194
- }
2195
- // If neither is set, EffortLevel remains null (provider default was used)
2196
- // Set ParentID for hierarchical prompt execution tracking
2197
- if (params.parentPromptRunId) {
2198
- promptRun.ParentID = params.parentPromptRunId;
2199
- }
2200
- // Set RerunFromPromptRunID if this is a rerun
2201
- if (params.rerunFromPromptRunID) {
2202
- promptRun.RerunFromPromptRunID = params.rerunFromPromptRunID;
2203
- }
2204
- // Always save the response format from the prompt if it exists
2205
- if (prompt.ResponseFormat && prompt.ResponseFormat !== 'Any') {
2206
- promptRun.ResponseFormat = prompt.ResponseFormat;
2207
- }
2208
- // Save the actual values that will be used (prompt defaults overridden by additionalParameters).
2209
- // Uses the shared resolver so the persisted record matches what executeModel sends to the model.
2210
- const resolvedParams = this.resolveScalarInferenceParams(prompt, params.additionalParameters);
2211
- if (resolvedParams.temperature !== undefined)
2212
- promptRun.Temperature = resolvedParams.temperature;
2213
- if (resolvedParams.topP !== undefined)
2214
- promptRun.TopP = resolvedParams.topP;
2215
- if (resolvedParams.topK !== undefined)
2216
- promptRun.TopK = resolvedParams.topK;
2217
- if (resolvedParams.minP !== undefined)
2218
- promptRun.MinP = resolvedParams.minP;
2219
- if (resolvedParams.frequencyPenalty !== undefined)
2220
- promptRun.FrequencyPenalty = resolvedParams.frequencyPenalty;
2221
- if (resolvedParams.presencePenalty !== undefined)
2222
- promptRun.PresencePenalty = resolvedParams.presencePenalty;
2223
- if (resolvedParams.seed !== undefined)
2224
- promptRun.Seed = resolvedParams.seed;
2225
- if (resolvedParams.includeLogProbs !== undefined)
2226
- promptRun.LogProbs = resolvedParams.includeLogProbs;
2227
- if (resolvedParams.topLogProbs !== undefined)
2228
- promptRun.TopLogProbs = resolvedParams.topLogProbs;
2229
- // Stop sequences + assistant prefill: stored from the prompt, with the additionalParameters
2230
- // array (JSON-encoded) taking precedence when supplied.
2231
- if (prompt.StopSequences)
2232
- promptRun.StopSequences = prompt.StopSequences;
2233
- if (prompt.AssistantPrefill)
2234
- promptRun.AssistantPrefill = prompt.AssistantPrefill;
2235
- if (params.additionalParameters?.stopSequences !== undefined && params.additionalParameters.stopSequences.length > 0) {
2236
- promptRun.StopSequences = JSON.stringify(params.additionalParameters.stopSequences);
2237
- }
2238
- // Store the input data/context as JSON in Messages field.
2239
- // Also capture callers that supply conversationMessages directly (e.g. templateMessageRole='none',
2240
- // no rendered system prompt) — otherwise their assembled prompt would never be persisted.
2241
- if (params.data || params.templateData || systemPromptText || (params.conversationMessages?.length ?? 0) > 0) {
2242
- const messages = [];
2243
- if (systemPromptText) {
2244
- // Build the system prompt content, including prefill fallback if applicable
2245
- let systemContent = systemPromptText;
2246
- if (prompt.AssistantPrefill && prompt.PrefillFallbackMode === 'SystemInstruction') {
2247
- const fallbackTemplate = this.resolvePrefillFallbackText(model, vendorId);
2248
- // Function replacement: prefill text is authored content that routinely
2249
- // contains `$` (LaTeX `$$`, currency, JSON fragments), and a string
2250
- // replacement would expand it. See issue #3171.
2251
- const prefill = prompt.AssistantPrefill;
2252
- const fallbackInstruction = fallbackTemplate.replace(/\{\{prefill\}\}/g, () => prefill);
2253
- systemContent += '\n\n' + fallbackInstruction;
2254
- }
2255
- messages.push({
2256
- role: 'system',
2257
- content: systemContent
2258
- });
2259
- }
2260
- // Always include any caller-supplied conversation messages (previously only recorded when a
2261
- // template system prompt was present, which dropped them for the pure-conversationMessages path).
2262
- messages.push(...(params.conversationMessages || []));
2263
- promptRun.Messages = JSON.stringify({
2264
- data: params.data,
2265
- templateData: params.templateData,
2266
- messages: messages || [],
2267
- });
2268
- }
2269
- // Populate new retry tracking columns with initial values
2270
- promptRun.ValidationBehavior = prompt.ValidationBehavior || 'Warn';
2271
- promptRun.RetryStrategy = prompt.RetryStrategy || 'Fixed';
2272
- promptRun.MaxRetriesConfigured = prompt.MaxRetries || 0;
2273
- promptRun.FirstAttemptAt = startTime;
2274
- promptRun.ValidationAttemptCount = 0; // Will be updated during execution
2275
- promptRun.SuccessfulValidationCount = 0;
2276
- promptRun.FinalValidationPassed = false; // Will be updated after execution
2277
- // Persist the initial 'Running' record fire-and-forget. The ID was already assigned by
2278
- // NewRecord() above, so callers (and the onPromptRunCreated callback) have it immediately —
2279
- // we don't block the model call on the INSERT. The finalize UPDATE chains after this INSERT
2280
- // via the instance-keyed save queue, so ordering is guaranteed.
2281
- this._promptRunQueue.Insert(promptRun);
2282
- // Invoke callback if provided. The ID is available without awaiting the save (client-generated
2283
- // by NewRecord()), so agent-run/step linking that depends on it works immediately.
2284
- if (params.onPromptRunCreated) {
2285
- try {
2286
- await params.onPromptRunCreated(promptRun.ID);
2287
- }
2288
- catch (callbackError) {
2289
- LogStatus(`Error in onPromptRunCreated callback: ${callbackError.message}`);
2290
- // Don't fail the execution if callback fails
2291
- }
2292
- }
2293
- return promptRun;
2294
- }
2295
- catch (error) {
2296
- const msg = `Error creating prompt run record: ${error.message} - ${promptRun?.LatestResult?.CompleteMessage} - ${promptRun?.LatestResult?.Errors[0]?.Message}`;
2297
- this.logError(msg, {
2298
- category: 'PromptRunSave',
2299
- metadata: {
2300
- promptRunId: promptRun.ID,
2301
- saveError: promptRun.LatestResult?.CompleteMessage
2302
- },
2303
- maxErrorLength: params.maxErrorLength
2304
- });
2305
- throw new Error(msg);
2306
- }
2307
- }
2308
1280
  /**
2309
1281
  * Renders the prompt template with provided data
2310
1282
  */
@@ -2363,12 +1335,12 @@ export class AIPromptRunner {
2363
1335
  * capabilities. It will attempt to execute with different models/vendors according
2364
1336
  * to the configured failover strategy when errors occur.
2365
1337
  *
2366
- * The method calls several smaller, focused helper methods:
2367
- * - buildFailoverCandidates: Creates candidate models based on type restrictions
2368
- * - createCandidatesFromModels: Converts models to vendor-specific candidates
2369
- * - updatePromptRunWithFailoverSuccess: Records successful failover metadata
2370
- * - updatePromptRunWithFailoverFailure: Records failed failover metadata
2371
- * - createFailoverErrorResult: Creates standardized error response
1338
+ * Candidates come from model selection (`allCandidates`), already filtered to the prompt's
1339
+ * model type by ID. When failover applies, the loop itself is
1340
+ * {@link BaseModelRunner.ExecuteWithFailover}: this method supplies the chat call on each candidate
1341
+ * (`executeModel` with that candidate's model, vendor, driver, effort level and prompt-model
1342
+ * configuration) and the final error result (`createFailoverErrorResult`). The base records
1343
+ * failover success or failure on the prompt run.
2372
1344
  */
2373
1345
  async executeModelWithFailover(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, allCandidates, promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability, promptModelConfiguration) {
2374
1346
  // Get failover configuration (used for errorScope filtering)
@@ -2377,224 +1349,7 @@ export class AIPromptRunner {
2377
1349
  if (!allCandidates || allCandidates.length === 0 || failoverConfig.strategy === 'None') {
2378
1350
  return this.executeModel(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole, cancellationToken, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, promptModelConfiguration);
2379
1351
  }
2380
- // Track failover attempts
2381
- const failoverAttempts = [];
2382
- let lastError = null;
2383
- // Cache credential availability per driver:model:vendor for the duration of this failover
2384
- // scan so we don't repeat env-var / binding lookups while walking the candidate list.
2385
- //
2386
- // PERF: seed it with the probes model SELECTION already performed (same key format). Selection
2387
- // walks the priority list until it finds the first credentialed candidate, so this map holds
2388
- // the prefix it rejected (known false) PLUS the selected candidate (known true) — which is
2389
- // exactly the segment failover re-walks on the happy path. Reusing those results means the
2390
- // common case (and any caller looping failover) does ZERO redundant hasCredentialsAvailable
2391
- // calls. The not-evaluated tail is intentionally absent, so failover still lazily probes it
2392
- // only if a real failure forces it to walk down there.
2393
- const failoverCredentialCache = credentialAvailability
2394
- ? new Map(credentialAvailability)
2395
- : new Map();
2396
- const candidateHasCredentials = (c) => {
2397
- const key = `${c.driverClass}:${c.model.ID}:${c.vendorId || 'default'}`;
2398
- let has = failoverCredentialCache.get(key);
2399
- if (has === undefined) {
2400
- has = this.hasCredentialsAvailable(c.driverClass, prompt.ID, c.model.ID, c.vendorId, params);
2401
- failoverCredentialCache.set(key, has);
2402
- }
2403
- return has;
2404
- };
2405
- let skippedForCredentials = 0;
2406
- // Iterate through all candidates in priority order with instant failover
2407
- for (let i = 0; i < allCandidates.length; i++) {
2408
- const candidate = allCandidates[i];
2409
- const attemptStartTime = Date.now();
2410
- // Skip candidates with no credentials configured. `allCandidates` is intentionally the
2411
- // FULL priority-ordered list (see the DECISION note in selectModelWithAPIKeyTracked),
2412
- // so it can include vendors that have no API key in this environment. Firing a live
2413
- // request at one of those produces a misleading "401 invalid API key" — and because an
2414
- // Authentication error is treated as fatal, it would halt failover before any
2415
- // credentialed candidate is ever reached. Skipping here makes failover land on the
2416
- // first candidate that can actually authenticate (mirroring model selection's own
2417
- // highest-priority-with-credentials rule).
2418
- if (!candidateHasCredentials(candidate)) {
2419
- skippedForCredentials++;
2420
- continue;
2421
- }
2422
- try {
2423
- // Log the attempt if not the first one
2424
- if (i > 0) {
2425
- const vendorName = candidate.vendorName || 'default';
2426
- LogStatusEx({
2427
- message: `🔄 Trying candidate ${i + 1}/${allCandidates.length}: ${candidate.model.Name} via ${vendorName}`,
2428
- category: 'AI',
2429
- additionalArgs: [{
2430
- promptId: prompt.ID,
2431
- modelId: candidate.model.ID,
2432
- model: candidate.model.Name,
2433
- vendorId: candidate.vendorId,
2434
- vendor: candidate.vendorName,
2435
- attemptNumber: i + 1
2436
- }]
2437
- });
2438
- }
2439
- // Execute the model with this candidate
2440
- const result = await this.executeModel(candidate.model, renderedPrompt, prompt, params, candidate.vendorId || null, conversationMessages, templateMessageRole, cancellationToken, candidate.driverClass, candidate.apiName, candidate.supportsEffortLevel, candidate.effortLevel, candidate.promptModelConfiguration);
2441
- // CRITICAL FIX: Check if result failed but is retriable (network errors, rate limits, etc.)
2442
- // Provider drivers (GeminiLLM, OpenAILLM, etc.) catch errors internally and return ChatResult{success: false}
2443
- // instead of throwing, so we must check result.success here.
2444
- if (!result.success && result.errorInfo?.canFailover) {
2445
- lastError = result.exception || new Error(result.errorMessage || 'Model execution failed');
2446
- // Use shared failover error handling logic
2447
- const decision = await this.processFailoverError(lastError, result.errorInfo, candidate, attemptStartTime, i, allCandidates, failoverAttempts, prompt, failoverConfig);
2448
- // Update candidates list (may have been filtered)
2449
- allCandidates = decision.updatedCandidates;
2450
- if (decision.shouldRetry) {
2451
- i--; // Retry same model/vendor
2452
- continue;
2453
- }
2454
- if (decision.shouldContinue) {
2455
- continue; // Try next candidate
2456
- }
2457
- // Otherwise break (fatal error or last candidate)
2458
- break;
2459
- }
2460
- // If we reach here, the result was successful
2461
- // Update promptRun with failover information if we had prior failures
2462
- if (failoverAttempts.length > 0 && promptRun) {
2463
- this.updatePromptRunWithFailoverSuccess(promptRun, failoverAttempts, candidate.model, candidate.vendorId || null);
2464
- }
2465
- return result;
2466
- }
2467
- catch (error) {
2468
- lastError = error;
2469
- // Analyze error to get error info
2470
- const errorInfo = ErrorAnalyzer.analyzeError(lastError);
2471
- // Use shared failover error handling logic
2472
- const decision = await this.processFailoverError(lastError, errorInfo, candidate, attemptStartTime, i, allCandidates, failoverAttempts, prompt, failoverConfig);
2473
- // Update candidates list (may have been filtered)
2474
- allCandidates = decision.updatedCandidates;
2475
- if (decision.shouldRetry) {
2476
- i--; // Retry same model/vendor
2477
- continue;
2478
- }
2479
- if (decision.shouldContinue) {
2480
- continue; // Try next candidate
2481
- }
2482
- // Otherwise break (fatal error or last candidate)
2483
- break;
2484
- }
2485
- }
2486
- // All candidates failed
2487
- if (promptRun && failoverAttempts.length > 0) {
2488
- this.updatePromptRunWithFailoverFailure(promptRun, failoverAttempts);
2489
- }
2490
- // If every candidate was skipped for missing credentials we never attempted a call and
2491
- // have no underlying error to report — surface an actionable message instead of null.
2492
- if (!lastError && failoverAttempts.length === 0 && skippedForCredentials > 0) {
2493
- lastError = new Error(`No API credentials configured for any of the ${skippedForCredentials} candidate model-vendor combination(s) for prompt "${prompt.Name}".`);
2494
- }
2495
- return this.createFailoverErrorResult(lastError, failoverAttempts);
2496
- }
2497
- /**
2498
- * Builds failover candidates for a prompt based on available models and type restrictions
2499
- */
2500
- async buildFailoverCandidates(prompt) {
2501
- const aiEngine = AIEngine.Instance;
2502
- // Get all models, filtered by type if specified
2503
- let allModels;
2504
- if (prompt.AIModelTypeID) {
2505
- // Find the model type from the prompt
2506
- const modelType = aiEngine.ModelTypes.find(mt => UUIDsEqual(mt.ID, prompt.AIModelTypeID));
2507
- if (!modelType) {
2508
- throw new Error(`Model type ${prompt.AIModelTypeID} not found`);
2509
- }
2510
- // Get all models of this specific type
2511
- const targetTypeName = modelType.Name.trim().toLowerCase();
2512
- allModels = aiEngine.Models.filter(m => {
2513
- // Guard against AIModelType being non-string (defensive coding for data issues)
2514
- const mType = typeof m.AIModelType === 'string' ? m.AIModelType.trim().toLowerCase() : '';
2515
- return mType === targetTypeName;
2516
- });
2517
- }
2518
- else {
2519
- // No type restriction - get all models
2520
- allModels = aiEngine.Models;
2521
- }
2522
- return this.createCandidatesFromModels(allModels);
2523
- }
2524
- /**
2525
- * Creates model-vendor candidates from a list of models
2526
- */
2527
- createCandidatesFromModels(models) {
2528
- const candidates = [];
2529
- for (const model of models) {
2530
- const vendors = model.ModelVendors || [];
2531
- if (vendors.length === 0) {
2532
- // Model without specific vendors
2533
- candidates.push({
2534
- model: model,
2535
- vendorId: undefined,
2536
- vendorName: undefined,
2537
- driverClass: model.DriverClass,
2538
- apiName: model.APIName,
2539
- supportsEffortLevel: model.SupportsEffortLevel ?? false,
2540
- isPreferredVendor: false,
2541
- priority: model.PowerRank || 0,
2542
- source: 'power-rank'
2543
- });
2544
- }
2545
- else {
2546
- // Add each vendor as a separate candidate
2547
- for (const vendor of vendors) {
2548
- candidates.push({
2549
- model: model,
2550
- vendorId: vendor.VendorID,
2551
- vendorName: vendor.Vendor,
2552
- driverClass: vendor.DriverClass || model.DriverClass,
2553
- apiName: vendor.APIName || model.APIName,
2554
- supportsEffortLevel: vendor.SupportsEffortLevel ?? model.SupportsEffortLevel ?? false,
2555
- isPreferredVendor: vendor.Priority > 0,
2556
- priority: (model.PowerRank || 0) + (vendor.Priority || 0),
2557
- source: 'power-rank'
2558
- });
2559
- }
2560
- }
2561
- }
2562
- return candidates;
2563
- }
2564
- /**
2565
- * Updates prompt run with successful failover tracking data
2566
- */
2567
- updatePromptRunWithFailoverSuccess(promptRun, failoverAttempts, currentModel, currentVendorId) {
2568
- promptRun.FailoverAttempts = failoverAttempts.length;
2569
- promptRun.FailoverErrors = JSON.stringify(failoverAttempts.map(a => ({
2570
- model: a.modelId,
2571
- vendor: a.vendorId,
2572
- error: a.error.message,
2573
- errorType: a.errorType
2574
- })));
2575
- promptRun.FailoverDurations = JSON.stringify(failoverAttempts.map(a => a.duration));
2576
- promptRun.TotalFailoverDuration = failoverAttempts.reduce((sum, a) => sum + a.duration, 0);
2577
- // Update ModelID if we ended up using a different model
2578
- if (!UUIDsEqual(currentModel.ID, promptRun.OriginalModelID)) {
2579
- promptRun.ModelID = currentModel.ID;
2580
- }
2581
- if (currentVendorId && !UUIDsEqual(currentVendorId, promptRun.VendorID)) {
2582
- promptRun.VendorID = currentVendorId;
2583
- }
2584
- }
2585
- /**
2586
- * Updates prompt run with failover failure tracking data
2587
- */
2588
- updatePromptRunWithFailoverFailure(promptRun, failoverAttempts) {
2589
- promptRun.FailoverAttempts = failoverAttempts.length;
2590
- promptRun.FailoverErrors = JSON.stringify(failoverAttempts.map(a => ({
2591
- model: a.modelId,
2592
- vendor: a.vendorId,
2593
- error: a.error.message,
2594
- errorType: a.errorType
2595
- })));
2596
- promptRun.FailoverDurations = JSON.stringify(failoverAttempts.map(a => a.duration));
2597
- promptRun.TotalFailoverDuration = failoverAttempts.reduce((sum, a) => sum + a.duration, 0);
1352
+ return this.ExecuteWithFailover(prompt, params, allCandidates, failoverConfig, (candidate) => this.executeModel(candidate.model, renderedPrompt, prompt, params, candidate.vendorId || null, conversationMessages, templateMessageRole, cancellationToken, candidate.driverClass, candidate.apiName, candidate.supportsEffortLevel, candidate.effortLevel, candidate.promptModelConfiguration), (lastError, failoverAttempts) => this.createFailoverErrorResult(lastError, failoverAttempts), promptRun, credentialAvailability);
2598
1353
  }
2599
1354
  /**
2600
1355
  * Creates an error result for failed failover attempts
@@ -2658,7 +1413,7 @@ export class AIPromptRunner {
2658
1413
  * Never throws: the gate is an opt-in enhancement and must not be able to fail a run that would
2659
1414
  * otherwise succeed, so any configuration problem resolves to the path that has always worked.
2660
1415
  */
2661
- resolveNativeToolCallingDecision(prompt, params, model, vendorId, promptModelConfiguration) {
1416
+ ResolveNativeToolCallingDecision(prompt, params, model, vendorId, promptModelConfiguration) {
2662
1417
  try {
2663
1418
  return ResolveNativeToolCalling({
2664
1419
  catalogConfiguration: AIEngine.Instance.GetEffectiveModelConfiguration(model.ID, vendorId
@@ -2667,7 +1422,7 @@ export class AIPromptRunner {
2667
1422
  // order. Picking the developer row merges an empty config layer and silently drops any
2668
1423
  // per-serving-path LLM.* knob (notably the SupportsNativeToolCalling kill switch).
2669
1424
  ? model.ModelVendors?.find(mv => UUIDsEqual(mv.VendorID, vendorId)
2670
- && mv.Status === 'Active' && this.isInferenceProvider(mv))?.ID
1425
+ && mv.Status === 'Active' && this.IsInferenceProvider(mv))?.ID
2671
1426
  : undefined),
2672
1427
  promptConfiguration: prompt.PromptConfigurationObject,
2673
1428
  promptModelConfiguration,
@@ -2683,8 +1438,12 @@ export class AIPromptRunner {
2683
1438
  return { useNativeTools: false, mode: 'Envelope', controlFlow: 'envelope', toolResults: false };
2684
1439
  }
2685
1440
  }
1441
+ /** @deprecated Use {@link ResolveNativeToolCallingDecision}. */
1442
+ resolveNativeToolCallingDecision(prompt, params, model, vendorId, promptModelConfiguration) {
1443
+ return this.ResolveNativeToolCallingDecision(prompt, params, model, vendorId, promptModelConfiguration);
1444
+ }
2686
1445
  applyNativeToolCalling(chatParams, prompt, params, model, vendorId, promptModelConfiguration) {
2687
- const decision = this.resolveNativeToolCallingDecision(prompt, params, model, vendorId, promptModelConfiguration);
1446
+ const decision = this.ResolveNativeToolCallingDecision(prompt, params, model, vendorId, promptModelConfiguration);
2688
1447
  if (decision.warning) {
2689
1448
  console.warn(`AIPromptRunner: ${decision.warning} (prompt "${prompt.Name}", model "${model.Name}"` +
2690
1449
  `${vendorId ? `, vendor ${vendorId}` : ''})`);
@@ -2790,7 +1549,7 @@ export class AIPromptRunner {
2790
1549
  supportsEffortLevel = model.SupportsEffortLevel ?? false;
2791
1550
  if (vendorId) {
2792
1551
  // Find the AIModelVendor record for this specific vendor - must be an inference provider
2793
- const modelVendor = model.ModelVendors.find((mv) => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active' && this.isInferenceProvider(mv));
1552
+ const modelVendor = model.ModelVendors.find((mv) => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active' && this.IsInferenceProvider(mv));
2794
1553
  if (modelVendor) {
2795
1554
  driverClass = modelVendor.DriverClass || driverClass;
2796
1555
  apiName = modelVendor.APIName || apiName;
@@ -2804,7 +1563,7 @@ export class AIPromptRunner {
2804
1563
  }
2805
1564
  }
2806
1565
  // Resolve credentials using hierarchical resolution (Credentials system with legacy fallback)
2807
- const apiKey = await this.resolveCredentialForExecution(driverClass, prompt.ID, model.ID, vendorId ?? undefined, params);
1566
+ const apiKey = await this.ResolveCredentialForExecution(driverClass, prompt.ID, model.ID, vendorId ?? undefined, params);
2808
1567
  // Create LLM instance with vendor-specific driver class
2809
1568
  llm = MJGlobal.Instance.ClassFactory.CreateInstance(BaseLLM, driverClass, apiKey);
2810
1569
  // Prepare chat parameters
@@ -2841,7 +1600,7 @@ export class AIPromptRunner {
2841
1600
  // Stop sequences are handled separately: the prompt value is comma-delimited and gated by
2842
1601
  // driver support; additionalParameters supplies a ready-made array that overrides it.
2843
1602
  if (prompt.StopSequences && this.shouldApplyStopSequences(prompt, model, vendorId, llm)) {
2844
- chatParams.stopSequences = prompt.StopSequences.split(',').map((s) => s.replace(AIPromptRunner.STOP_SEQUENCE_TRIM_REGEX, '')).filter((s) => s.length > 0);
1603
+ chatParams.stopSequences = prompt.StopSequences.split(',').map((s) => TrimSpacesAndTabs(s)).filter((s) => s.length > 0);
2845
1604
  }
2846
1605
  if (params.additionalParameters?.stopSequences !== undefined) {
2847
1606
  chatParams.stopSequences = params.additionalParameters.stopSequences;
@@ -2986,85 +1745,6 @@ export class AIPromptRunner {
2986
1745
  executionBound.Dispose();
2987
1746
  }
2988
1747
  }
2989
- /**
2990
- * Engine-level default model-call timeout, in milliseconds, applied when the caller supplies no
2991
- * `AIPromptParams.timeoutMS`. `undefined` (the default) means NO implicit bound — a prompt run
2992
- * with neither a timeout nor a cancellation token stays unbounded, exactly as before, so this
2993
- * change is behavior-preserving for existing callers.
2994
- *
2995
- * Subclasses (or a host application's runner subclass) can override this to impose a global
2996
- * safety ceiling on every prompt call.
2997
- */
2998
- get DefaultPromptTimeoutMS() {
2999
- return undefined;
3000
- }
3001
- /**
3002
- * Resolves the per-model-call timeout: the caller's `AIPromptParams.timeoutMS`, else the runner's
3003
- * {@link DefaultPromptTimeoutMS}. Non-positive / non-numeric values mean "no timeout".
3004
- *
3005
- * The bound is applied PER MODEL CALL (not per prompt execution), which mirrors the parallel
3006
- * path's existing `taskTimeoutMS` semantics: each failover candidate / validation retry gets a
3007
- * fresh budget rather than sharing one wall-clock window.
3008
- *
3009
- * NOTE (issue #3064): there is deliberately NO prompt-entity source here yet — the `AIPrompt`
3010
- * table has no `TimeoutMS` column today. Once a migration adds one and CodeGen regenerates the
3011
- * entity, this becomes `prompt.TimeoutMS ?? params.timeoutMS ?? this.DefaultPromptTimeoutMS`
3012
- * and every bound below starts honoring the per-prompt configuration with no other change.
3013
- */
3014
- getEffectiveTimeoutMS(params) {
3015
- const timeoutMS = params.timeoutMS ?? this.DefaultPromptTimeoutMS;
3016
- return typeof timeoutMS === 'number' && timeoutMS > 0 ? timeoutMS : undefined;
3017
- }
3018
- /**
3019
- * Composes the caller-supplied cancellation token with the resolved model-call timeout into a
3020
- * single {@link AbortSignal} that bounds one model call. NEITHER bound is discarded:
3021
- *
3022
- * - caller token only → the caller's signal is used directly (behavior unchanged)
3023
- * - timeout only → an internal controller aborts after the timeout elapses
3024
- * - both → an internal controller relays the caller's abort AND fires on timeout;
3025
- * whichever happens first wins
3026
- * - neither → `Signal` is undefined and the call runs unbounded (legacy behavior)
3027
- *
3028
- * Implemented with an AbortController + relay listener rather than `AbortSignal.any()` so it works
3029
- * on Node 18 (where `AbortSignal.any` does not exist — it landed in Node 20.3).
3030
- */
3031
- createExecutionBound(prompt, params, cancellationToken) {
3032
- const timeoutMS = this.getEffectiveTimeoutMS(params);
3033
- if (timeoutMS === undefined) {
3034
- // No prompt timeout: use the caller's token as-is (or nothing at all).
3035
- return { Signal: cancellationToken, TimeoutMS: undefined, TimedOut: () => false, Dispose: () => { } };
3036
- }
3037
- const controller = new AbortController();
3038
- let timedOut = false;
3039
- const relayCallerAbort = () => {
3040
- if (!controller.signal.aborted) {
3041
- controller.abort(cancellationToken?.reason ?? 'Chat completion was cancelled');
3042
- }
3043
- };
3044
- if (cancellationToken) {
3045
- if (cancellationToken.aborted) {
3046
- relayCallerAbort();
3047
- }
3048
- else {
3049
- cancellationToken.addEventListener('abort', relayCallerAbort, { once: true });
3050
- }
3051
- }
3052
- const timer = setTimeout(() => {
3053
- if (!controller.signal.aborted) {
3054
- timedOut = true;
3055
- controller.abort(new AIPromptTimeoutError(prompt.Name, timeoutMS));
3056
- }
3057
- }, timeoutMS);
3058
- return {
3059
- Signal: controller.signal,
3060
- TimeoutMS: timeoutMS,
3061
- TimedOut: () => timedOut,
3062
- Dispose: () => {
3063
- clearTimeout(timer);
3064
- cancellationToken?.removeEventListener('abort', relayCallerAbort);
3065
- },
3066
- };
3067
- }
3068
1748
  /**
3069
1749
  * Runs the model call, racing it against the composed execution bound so a hung provider surfaces
3070
1750
  * as a rejected promise the caller's failover/retry logic can act on.
@@ -3183,7 +1863,6 @@ export class AIPromptRunner {
3183
1863
  const result = [];
3184
1864
  let inManifest = false;
3185
1865
  let mutated = false;
3186
- const entryRegex = /^\*\*[A-Z]+\*\* — .+? \[(?<mime>[^\]]+)\]/;
3187
1866
  for (const line of lines) {
3188
1867
  if (line.startsWith('## Available Artifacts')) {
3189
1868
  inManifest = true;
@@ -3196,10 +1875,11 @@ export class AIPromptRunner {
3196
1875
  result.push(line);
3197
1876
  if (!inManifest)
3198
1877
  continue;
3199
- const match = entryRegex.exec(line);
3200
- if (!match)
1878
+ // `**A** — name [mime]`; parsed linearly (CodeQL js/polynomial-redos flagged the regex form).
1879
+ const entryMime = ParseManifestEntryMime(line);
1880
+ if (entryMime === null)
3201
1881
  continue;
3202
- const mime = (match.groups?.mime ?? '').toLowerCase();
1882
+ const mime = entryMime.toLowerCase();
3203
1883
  const modality = mime.split('/')[0];
3204
1884
  if (modality !== 'image' && modality !== 'audio' && modality !== 'video')
3205
1885
  continue;
@@ -3315,23 +1995,6 @@ export class AIPromptRunner {
3315
1995
  }
3316
1996
  return messages;
3317
1997
  }
3318
- /**
3319
- * Default fallback instruction text used when no PrefillFallbackText is configured
3320
- * at any level of the AIModelType → AIModel → AIModelVendor cascade.
3321
- */
3322
- static { this.DEFAULT_PREFILL_FALLBACK = '# **CRITICAL**\nYour response must start with exactly: {{prefill}}\nDo not add quotes, markdown formatting, or any other characters before it.'; }
3323
- /**
3324
- * Regex used to trim only horizontal whitespace (spaces and tabs) from the start and end
3325
- * of each stop sequence token after comma-splitting.
3326
- *
3327
- * We intentionally do NOT use String.trim() here because stop sequences can legitimately
3328
- * begin or end with newline characters. For example, the sequence "\n```" is designed to
3329
- * match only a closing code fence (preceded by a newline), distinguishing it from an
3330
- * opening "```json" fence that does not start with a newline. Using trim() would strip
3331
- * that leading "\n", turning "\n```" into "```" and causing the stop to fire on the
3332
- * opening fence instead — producing an empty response for non-native prefill providers.
3333
- */
3334
- static { this.STOP_SEQUENCE_TRIM_REGEX = /^[ \t]+|[ \t]+$/g; }
3335
1998
  /**
3336
1999
  * Substrings that mark a provider failure as TOOLS-specific, so the native call is worth one
3337
2000
  * envelope retry (see {@link AIPromptRunner.isToolSpecificFailure}). Drawn from how the
@@ -3429,28 +2092,6 @@ export class AIPromptRunner {
3429
2092
  }
3430
2093
  return supportsPrefill;
3431
2094
  }
3432
- /**
3433
- * Resolves the prefill fallback instruction text using the cascade:
3434
- * AIModelType → AIModel → AIModelVendor (most specific non-null wins).
3435
- * Falls back to DEFAULT_PREFILL_FALLBACK if none are configured.
3436
- */
3437
- resolvePrefillFallbackText(model, vendorId) {
3438
- // Start with model type default
3439
- const modelType = AIEngine.Instance.ModelTypesByID.get(NormalizeUUID(model.AIModelTypeID));
3440
- let fallbackText = modelType?.PrefillFallbackText ?? null;
3441
- // Model-level override
3442
- if (model.PrefillFallbackText != null) {
3443
- fallbackText = model.PrefillFallbackText;
3444
- }
3445
- // Vendor-level override
3446
- if (vendorId) {
3447
- const modelVendor = model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
3448
- if (modelVendor?.PrefillFallbackText != null) {
3449
- fallbackText = modelVendor.PrefillFallbackText;
3450
- }
3451
- }
3452
- return fallbackText ?? AIPromptRunner.DEFAULT_PREFILL_FALLBACK;
3453
- }
3454
2095
  /**
3455
2096
  * Applies assistant prefill to ChatParams based on prompt configuration and provider support.
3456
2097
  * Handles the full prefill resolution logic including fallback to system instructions.
@@ -3490,6 +2131,33 @@ export class AIPromptRunner {
3490
2131
  }
3491
2132
  // 'Ignore' and 'None' — silently skip, no action needed
3492
2133
  }
2134
+ /**
2135
+ * Default fallback instruction text used when no PrefillFallbackText is configured
2136
+ * at any level of the AIModelType → AIModel → AIModelVendor cascade.
2137
+ */
2138
+ static { this.DEFAULT_PREFILL_FALLBACK = '# **CRITICAL**\nYour response must start with exactly: {{prefill}}\nDo not add quotes, markdown formatting, or any other characters before it.'; }
2139
+ /**
2140
+ * Resolves the prefill fallback instruction text using the cascade:
2141
+ * AIModelType → AIModel → AIModelVendor (most specific non-null wins).
2142
+ * Falls back to DEFAULT_PREFILL_FALLBACK if none are configured.
2143
+ */
2144
+ resolvePrefillFallbackText(model, vendorId) {
2145
+ // Start with model type default
2146
+ const modelType = AIEngine.Instance.ModelTypesByID.get(NormalizeUUID(model.AIModelTypeID));
2147
+ let fallbackText = modelType?.PrefillFallbackText ?? null;
2148
+ // Model-level override
2149
+ if (model.PrefillFallbackText != null) {
2150
+ fallbackText = model.PrefillFallbackText;
2151
+ }
2152
+ // Vendor-level override
2153
+ if (vendorId) {
2154
+ const modelVendor = model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
2155
+ if (modelVendor?.PrefillFallbackText != null) {
2156
+ fallbackText = modelVendor.PrefillFallbackText;
2157
+ }
2158
+ }
2159
+ return fallbackText ?? AIPromptRunner.DEFAULT_PREFILL_FALLBACK;
2160
+ }
3493
2161
  /**
3494
2162
  * Executes the model with retry logic for validation failures
3495
2163
  */
@@ -3509,7 +2177,7 @@ export class AIPromptRunner {
3509
2177
  }
3510
2178
  if (attempt > 0) {
3511
2179
  LogStatus(` 🔄 Retrying execution due to validation failure, attempt ${attempt + 1}/${maxRetries + 1}`);
3512
- await this.applyRetryDelay(prompt, attempt);
2180
+ await this.ApplyRetryDelay(prompt, attempt);
3513
2181
  }
3514
2182
  // Execute the AI model with failover support
3515
2183
  const modelResult = await this.executeModelWithFailover(selectedModel, renderedPromptText, prompt, params, promptRun.VendorID, params.conversationMessages, params.templateMessageRole || 'system', params.cancellationToken, allCandidates, // Pass the candidates from initial selection
@@ -3576,15 +2244,16 @@ export class AIPromptRunner {
3576
2244
  }
3577
2245
  // Validation failed, check if we should retry
3578
2246
  // BUG FIX: Only retry in Strict mode, not in Warn or None modes
3579
- if (prompt.ValidationBehavior === 'Strict' && attempt < maxRetries) {
2247
+ const effectiveValidationBehavior = params?.validationBehavior || prompt.ValidationBehavior;
2248
+ if (effectiveValidationBehavior === 'Strict' && attempt < maxRetries) {
3580
2249
  lastError = new Error(`Validation failed: ${validationErrors?.map(e => e.Message).join('; ')}`);
3581
2250
  LogStatus(` ⚠️ Validation failed on attempt ${attempt + 1}, will retry (Strict mode)`);
3582
2251
  continue; // Retry
3583
2252
  }
3584
2253
  else {
3585
2254
  // Either not strict mode or no more retries, return what we have
3586
- const reason = prompt.ValidationBehavior !== 'Strict'
3587
- ? `${prompt.ValidationBehavior || 'None'} mode - continuing with invalid output (no retry)`
2255
+ const reason = effectiveValidationBehavior !== 'Strict'
2256
+ ? `${effectiveValidationBehavior || 'None'} mode - continuing with invalid output (no retry)`
3588
2257
  : 'max retries exceeded';
3589
2258
  LogStatus(` ⚠️ Validation failed on attempt ${attempt + 1}, stopping retries (${reason})`);
3590
2259
  return {
@@ -3628,166 +2297,6 @@ export class AIPromptRunner {
3628
2297
  // Should not reach here, but just in case
3629
2298
  throw lastError || new Error('Execution failed after all retry attempts');
3630
2299
  }
3631
- /**
3632
- * Applies retry delay based on the prompt's retry strategy
3633
- */
3634
- /**
3635
- * Calculates retry delay for rate limit and other retriable errors.
3636
- * Uses the prompt's RetryStrategy and can respect suggested delays from provider.
3637
- */
3638
- calculateRetryDelay(prompt, attemptNumber, suggestedDelaySeconds) {
3639
- // Use provider's suggested delay if available
3640
- if (suggestedDelaySeconds && suggestedDelaySeconds > 0) {
3641
- return suggestedDelaySeconds * 1000; // Convert to milliseconds
3642
- }
3643
- const baseDelay = prompt.RetryDelayMS || 1000; // Default 1 second
3644
- let delay = baseDelay;
3645
- switch (prompt.RetryStrategy) {
3646
- case 'Fixed':
3647
- delay = baseDelay;
3648
- break;
3649
- case 'Linear':
3650
- delay = baseDelay * attemptNumber;
3651
- break;
3652
- case 'Exponential':
3653
- delay = baseDelay * Math.pow(2, attemptNumber - 1);
3654
- break;
3655
- default:
3656
- delay = baseDelay;
3657
- }
3658
- return delay;
3659
- }
3660
- async applyRetryDelay(prompt, attemptNumber, suggestedDelaySeconds) {
3661
- const delay = this.calculateRetryDelay(prompt, attemptNumber, suggestedDelaySeconds);
3662
- const delaySeconds = (delay / 1000).toFixed(1);
3663
- LogStatus(` Waiting ${delaySeconds}s before retry (strategy: ${prompt.RetryStrategy || 'Fixed'})...`);
3664
- await new Promise(resolve => setTimeout(resolve, delay));
3665
- }
3666
- /**
3667
- * Filters out all candidates from a vendor when a vendor-level error occurs.
3668
- * Vendor-level errors affect all models from that vendor:
3669
- * - Authentication: Invalid API key
3670
- * - VendorValidationError: API schema/validation requirements
3671
- */
3672
- filterVendorCandidates(errorType, currentVendorId, allCandidates) {
3673
- if (errorType !== 'Authentication' && errorType !== 'VendorValidationError') {
3674
- return allCandidates; // No filtering needed for non-vendor-level errors
3675
- }
3676
- const failedVendorId = currentVendorId || 'default';
3677
- const beforeCount = allCandidates.length;
3678
- // Filter out ALL candidates from this vendor
3679
- const filteredCandidates = allCandidates.filter(c => (c.vendorId || 'default') !== failedVendorId);
3680
- const removedCount = beforeCount - filteredCandidates.length;
3681
- if (removedCount > 0) {
3682
- const vendorName = AIEngine.Instance.VendorsByID.get(NormalizeUUID(failedVendorId))?.Name || failedVendorId;
3683
- const remainingCount = filteredCandidates.length;
3684
- // Log appropriate message based on error type
3685
- let reason;
3686
- let icon;
3687
- if (errorType === 'Authentication') {
3688
- reason = 'Invalid API key';
3689
- icon = '🔒';
3690
- }
3691
- else if (errorType === 'VendorValidationError') {
3692
- reason = 'API schema incompatibility';
3693
- icon = '⚠️';
3694
- }
3695
- else {
3696
- reason = 'Vendor-level error';
3697
- icon = '❌';
3698
- }
3699
- this.logStatus(` ${icon} ${reason} for ${vendorName} - excluding ${removedCount} model${removedCount === 1 ? '' : 's'} from this vendor (${remainingCount} remaining)`, true);
3700
- }
3701
- return filteredCandidates;
3702
- }
3703
- /**
3704
- * Handles rate limit errors by retrying the same model/vendor with backoff.
3705
- * Returns true if the caller should continue (retry), false if should proceed to failover.
3706
- */
3707
- async handleRateLimitRetry(errorAnalysis, currentModel, currentVendorId, failoverAttempts, prompt, attemptNumber, maxAttempts, failoverAttempt) {
3708
- const isRateLimit = errorAnalysis.errorType === 'RateLimit';
3709
- if (!isRateLimit) {
3710
- return false; // Not a rate limit error
3711
- }
3712
- // Count how many times we've retried this specific model/vendor for rate limits
3713
- const rateLimitRetryCount = failoverAttempts.filter(a => UUIDsEqual(a.modelId, currentModel.ID) &&
3714
- UUIDsEqual(a.vendorId, currentVendorId) &&
3715
- a.errorType === 'RateLimit').length;
3716
- // Use MaxRetries from prompt configuration, default to 3 if not set
3717
- const maxRetries = prompt.MaxRetries ?? 3;
3718
- // Retry up to MaxRetries times before giving up and failing over
3719
- const shouldRetry = rateLimitRetryCount <= maxRetries;
3720
- if (shouldRetry) {
3721
- const modelName = currentModel.Name;
3722
- const vendorName = currentVendorId
3723
- ? AIEngine.Instance.VendorsByID.get(NormalizeUUID(currentVendorId))?.Name || 'default'
3724
- : 'default';
3725
- this.logStatus(` ⏳ Rate limit hit - retrying ${modelName} (${vendorName}) with backoff (attempt ${rateLimitRetryCount}/${maxRetries})`, true);
3726
- this.logFailoverAttempt(prompt.ID, failoverAttempt, true);
3727
- // Apply backoff delay before retry
3728
- if (attemptNumber < maxAttempts) {
3729
- await this.applyRetryDelay(prompt, rateLimitRetryCount, errorAnalysis.suggestedRetryDelaySeconds);
3730
- }
3731
- return true; // Signal to continue with same model/vendor
3732
- }
3733
- return false; // Too many retries, proceed to failover
3734
- }
3735
- /**
3736
- * Processes a failover error (either from catch block or from failed ChatResult).
3737
- * Handles vendor filtering, rate limit retries, fatal error detection, and failover logic.
3738
- *
3739
- * @returns Decision object indicating whether to retry same model, continue to next candidate, or stop
3740
- */
3741
- async processFailoverError(error, errorInfo, candidate, attemptStartTime, attemptIndex, allCandidates, failoverAttempts, prompt, failoverConfig) {
3742
- const attemptDuration = Date.now() - attemptStartTime;
3743
- // Create failover attempt record
3744
- const failoverAttempt = {
3745
- attemptNumber: attemptIndex + 1,
3746
- modelId: candidate.model.ID,
3747
- vendorId: candidate.vendorId,
3748
- error: error,
3749
- errorType: errorInfo.errorType,
3750
- duration: attemptDuration,
3751
- timestamp: new Date()
3752
- };
3753
- failoverAttempts.push(failoverAttempt);
3754
- // Vendor-level errors: filter out all candidates from this vendor
3755
- let updatedCandidates = allCandidates;
3756
- if (errorInfo.errorType === 'Authentication' || errorInfo.errorType === 'VendorValidationError') {
3757
- updatedCandidates = this.filterVendorCandidates(errorInfo.errorType, candidate.vendorId, allCandidates);
3758
- }
3759
- const isLastCandidate = attemptIndex === updatedCandidates.length - 1;
3760
- // Fatal errors: stop immediately
3761
- if (errorInfo.severity === 'Fatal') {
3762
- const errorMessage = error?.message || 'Unknown error';
3763
- LogErrorEx(`Stopping failover: Fatal error (${errorInfo.errorType}): ${errorMessage}`);
3764
- this.logFailoverAttempt(prompt.ID, failoverAttempt, false);
3765
- return { shouldRetry: false, shouldContinue: false, updatedCandidates };
3766
- }
3767
- // Check errorScope filter if configured
3768
- if (failoverConfig.errorScope && failoverConfig.errorScope !== 'All') {
3769
- const matchesScope = this.errorMatchesScope(errorInfo.errorType, failoverConfig.errorScope);
3770
- if (!matchesScope) {
3771
- this.logFailoverAttempt(prompt.ID, failoverAttempt, false);
3772
- return { shouldRetry: false, shouldContinue: false, updatedCandidates };
3773
- }
3774
- }
3775
- // Rate limit errors: check if we should retry the same model before failing over
3776
- if (errorInfo.errorType === 'RateLimit') {
3777
- const shouldRetry = await this.handleRateLimitRetry(errorInfo, candidate.model, candidate.vendorId, failoverAttempts, prompt, attemptIndex, updatedCandidates.length, failoverAttempt);
3778
- if (shouldRetry) {
3779
- return { shouldRetry: true, shouldContinue: false, updatedCandidates };
3780
- }
3781
- }
3782
- // If this is the last candidate, we're done
3783
- if (isLastCandidate) {
3784
- this.logFailoverAttempt(prompt.ID, failoverAttempt, false);
3785
- return { shouldRetry: false, shouldContinue: false, updatedCandidates };
3786
- }
3787
- // Log and signal to continue to next candidate
3788
- this.logFailoverAttempt(prompt.ID, failoverAttempt, true);
3789
- return { shouldRetry: false, shouldContinue: true, updatedCandidates };
3790
- }
3791
2300
  /**
3792
2301
  * Transitions to the next failover candidate.
3793
2302
  * Returns the next candidate info or null if no candidates are available.
@@ -3816,28 +2325,6 @@ export class AIPromptRunner {
3816
2325
  supportsEffortLevel: nextCandidate.supportsEffortLevel || false
3817
2326
  };
3818
2327
  }
3819
- /**
3820
- * Provides a human-readable description of the validation decision
3821
- */
3822
- getValidationDecisionDescription(finalSuccess, totalAttempts, validationBehavior) {
3823
- if (finalSuccess) {
3824
- return totalAttempts === 1
3825
- ? 'Validation passed on first attempt'
3826
- : `Validation passed after ${totalAttempts} attempts`;
3827
- }
3828
- else {
3829
- switch (validationBehavior) {
3830
- case 'Strict':
3831
- return `Validation failed after ${totalAttempts} attempts - execution marked as failed (Strict mode)`;
3832
- case 'Warn':
3833
- return `Validation failed after ${totalAttempts} attempts - warning logged, execution continued (Warn mode)`;
3834
- case 'None':
3835
- return `Validation skipped or ignored (None mode)`;
3836
- default:
3837
- return `Validation failed after ${totalAttempts} attempts - behavior: ${validationBehavior}`;
3838
- }
3839
- }
3840
- }
3841
2328
  /**
3842
2329
  * Generates a JSON schema from an example object for validation
3843
2330
  */
@@ -4064,7 +2551,8 @@ export class AIPromptRunner {
4064
2551
  validationResult.Errors = validationErrors.length > 0 ? validationErrors : [
4065
2552
  new ValidationErrorInfo('general', error.message, undefined, ValidationErrorType.Failure)
4066
2553
  ];
4067
- switch (prompt.ValidationBehavior) {
2554
+ const effectiveValidationBehavior = params?.validationBehavior || prompt.ValidationBehavior;
2555
+ switch (effectiveValidationBehavior) {
4068
2556
  case 'Strict':
4069
2557
  return { result: undefined, validationResult, validationErrors: validationResult.Errors };
4070
2558
  case 'Warn':
@@ -4319,7 +2807,12 @@ export class AIPromptRunner {
4319
2807
  ERROR_MESSAGE: trueError,
4320
2808
  MALFORMED_JSON: rawOutput
4321
2809
  },
4322
- skipValidation: true // don't want to validate as this would cause recursive infinity scenario if the JSON is invalid. Just one shot, fix or no fix
2810
+ skipValidation: true, // one shot, fix or no fix: no validation retries on the repair itself.
2811
+ // attemptJSONRepair is deliberately NOT set, so a repair can never start a repair of its own.
2812
+ // That keeps every run within two levels of an agent step's target, which is as far as
2813
+ // vwAIUsageFacts looks for the agent run (pinned by the nesting-depth tests).
2814
+ agentId: params.agentId,
2815
+ UserID: ResolvePromptRunUserID({ UserID: params.UserID, ContextUser: params.contextUser }) ?? undefined,
4323
2816
  });
4324
2817
  if (!repairResult.success || !repairResult.result) {
4325
2818
  throw new Error('AI-based JSON repair failed' + (repairResult.errorMessage ? `: ${repairResult.errorMessage}` : ''));
@@ -4468,496 +2961,343 @@ export class AIPromptRunner {
4468
2961
  }
4469
2962
  return validationErrors;
4470
2963
  }
2964
+ // ==================== PROMPT RUN LIFECYCLE ====================
4471
2965
  /**
4472
- * Updates the AIPromptRun entity with execution results
2966
+ * Creates an AIPromptRun entity for execution tracking
4473
2967
  */
4474
- async updatePromptRun(promptRun, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens) {
4475
- // Fire-and-forget finalize UPDATE. The field mutations run INSIDE the post-INSERT task (after the
4476
- // 'Running' INSERT + its finalizeSave reload land), so the reload can never revert them and the
4477
- // chained UPDATE persists the finalized state — the "stuck at Running" race is structurally
4478
- // impossible. The execution flow does NOT await the save.
4479
- this._promptRunQueue.Update(promptRun, () => this.applyFinalizedPromptRunFields(promptRun, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens));
2968
+ async createPromptRun(prompt, model, params, systemPromptText, startTime, vendorId, modelSelectionInfo, runType) {
2969
+ return this.CreateRunRecord(prompt, model, params, startTime, vendorId, modelSelectionInfo, (promptRun) => {
2970
+ this.applyChatRequestFields(promptRun, prompt, model, params, systemPromptText, startTime, vendorId);
2971
+ // An explicit run type (the consolidated parallel parent) wins over params.RunType.
2972
+ if (runType) {
2973
+ promptRun.RunType = runType;
2974
+ }
2975
+ });
4480
2976
  }
4481
2977
  /**
4482
- * Populates a prompt-run's finalized fields (result, tokens, cost, timing, rollups) from the model
4483
- * result. Runs INSIDE the post-INSERT save task — see {@link updatePromptRun}. Errors here are
4484
- * logged (non-fatal): the AIPromptRun is observability, not part of the prompt's success contract.
4485
- */
4486
- applyFinalizedPromptRunFields(promptRun, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens) {
4487
- try {
4488
- promptRun.CompletedAt = endTime;
4489
- promptRun.ExecutionTimeMS = executionTimeMS;
4490
- // Determine what to save as the result
4491
- let resultToSave;
4492
- const rawResult = modelResult.data?.choices?.[0]?.message?.content || '';
4493
- if (parsedResult.result === undefined ||
4494
- parsedResult.result === null ||
4495
- (typeof parsedResult.result === 'string' && parsedResult.result.trim().length === 0)) {
4496
- // Use raw result as fallback when parsed result is undefined, null, or empty string
4497
- resultToSave = rawResult;
4498
- // Also set error message when we have to fall back to raw result
4499
- if (!promptRun.ErrorMessage) {
4500
- const validationErrors = parsedResult.validationResult?.Errors;
4501
- if (validationErrors && validationErrors.length > 0) {
4502
- promptRun.ErrorMessage = `JSON parsing/validation failed: ${validationErrors.map(e => e.Message).join('; ')}`;
4503
- }
4504
- else {
4505
- promptRun.ErrorMessage = 'Failed to parse result into expected format; raw output saved instead';
4506
- }
4507
- }
4508
- }
4509
- else if (typeof parsedResult.result === 'string') {
4510
- resultToSave = parsedResult.result;
4511
- }
4512
- else {
4513
- resultToSave = JSON.stringify(parsedResult.result);
4514
- }
4515
- promptRun.Result = resultToSave;
4516
- // Extract token usage and cost - use cumulative if retries occurred
4517
- if (cumulativeTokens && validationAttempts && validationAttempts.length > 1) {
4518
- // Multiple attempts occurred, use cumulative totals. cumulativeTokens.promptTokens is the
4519
- // UNCACHED ("net-new") input summed across attempts; cache reads/writes are NOT summed (the
4520
- // re-sent prefix would over-count) and are persisted from the final model result below.
4521
- // TokensUsed must equal TokensPrompt + TokensCompletion (AIPromptRun invariant), so it does
4522
- // NOT include the cache buckets — those live in TokensCacheRead/TokensCacheWrite.
4523
- promptRun.TokensPrompt = cumulativeTokens.promptTokens;
4524
- promptRun.TokensCompletion = cumulativeTokens.completionTokens;
4525
- promptRun.TokensUsed = cumulativeTokens.promptTokens + cumulativeTokens.completionTokens;
4526
- promptRun.Cost = cumulativeTokens.totalCost;
4527
- // Cost currency from the last model result
4528
- if (modelResult.data?.usage?.costCurrency !== undefined) {
4529
- promptRun.CostCurrency = modelResult.data.usage.costCurrency;
2978
+ * Sets a prompt-run's chat-specific request fields: messages, prefill, sampling parameters,
2979
+ * response format, streaming, effort level, child prompt and the validation/retry columns. Called by
2980
+ * {@link BaseModelRunner.CreateRunRecord} just before the INSERT is queued.
2981
+ */
2982
+ applyChatRequestFields(promptRun, prompt, model, params, systemPromptText, startTime, vendorId) {
2983
+ // Set ChildPromptID if this is a hierarchical execution with child prompts
2984
+ if (params.childPrompts && params.childPrompts.length > 0) {
2985
+ promptRun.ChildPromptID = params.childPrompts[0].childPrompt.prompt.ID;
2986
+ }
2987
+ promptRun.StreamingEnabled = !!params.onStreaming;
2988
+ // Resolve and save the effort level used (same precedence as ChatParams resolution).
2989
+ // EffortLevel is a numeric column with a CHECK (1-100), so a provider-named level such as
2990
+ // 'xhigh' is deliberately not persisted here — it still reaches the driver via ChatParams.
2991
+ if (typeof params.effortLevel === 'number') {
2992
+ promptRun.EffortLevel = params.effortLevel;
2993
+ }
2994
+ else if (prompt.EffortLevel !== undefined && prompt.EffortLevel !== null) {
2995
+ promptRun.EffortLevel = prompt.EffortLevel;
2996
+ }
2997
+ // If neither is set, EffortLevel remains null (provider default was used)
2998
+ // Always save the response format from the prompt if it exists
2999
+ if (prompt.ResponseFormat && prompt.ResponseFormat !== 'Any') {
3000
+ promptRun.ResponseFormat = prompt.ResponseFormat;
3001
+ }
3002
+ // Save the actual values that will be used (prompt defaults overridden by additionalParameters).
3003
+ // Uses the shared resolver so the persisted record matches what executeModel sends to the model.
3004
+ const resolvedParams = this.resolveScalarInferenceParams(prompt, params.additionalParameters);
3005
+ if (resolvedParams.temperature !== undefined)
3006
+ promptRun.Temperature = resolvedParams.temperature;
3007
+ if (resolvedParams.topP !== undefined)
3008
+ promptRun.TopP = resolvedParams.topP;
3009
+ if (resolvedParams.topK !== undefined)
3010
+ promptRun.TopK = resolvedParams.topK;
3011
+ if (resolvedParams.minP !== undefined)
3012
+ promptRun.MinP = resolvedParams.minP;
3013
+ if (resolvedParams.frequencyPenalty !== undefined)
3014
+ promptRun.FrequencyPenalty = resolvedParams.frequencyPenalty;
3015
+ if (resolvedParams.presencePenalty !== undefined)
3016
+ promptRun.PresencePenalty = resolvedParams.presencePenalty;
3017
+ if (resolvedParams.seed !== undefined)
3018
+ promptRun.Seed = resolvedParams.seed;
3019
+ if (resolvedParams.includeLogProbs !== undefined)
3020
+ promptRun.LogProbs = resolvedParams.includeLogProbs;
3021
+ if (resolvedParams.topLogProbs !== undefined)
3022
+ promptRun.TopLogProbs = resolvedParams.topLogProbs;
3023
+ // Stop sequences + assistant prefill: stored from the prompt, with the additionalParameters
3024
+ // array (JSON-encoded) taking precedence when supplied.
3025
+ if (prompt.StopSequences)
3026
+ promptRun.StopSequences = prompt.StopSequences;
3027
+ if (prompt.AssistantPrefill)
3028
+ promptRun.AssistantPrefill = prompt.AssistantPrefill;
3029
+ if (params.additionalParameters?.stopSequences !== undefined && params.additionalParameters.stopSequences.length > 0) {
3030
+ promptRun.StopSequences = JSON.stringify(params.additionalParameters.stopSequences);
3031
+ }
3032
+ // Store the input data/context as JSON in Messages field.
3033
+ // Also capture callers that supply conversationMessages directly (e.g. templateMessageRole='none',
3034
+ // no rendered system prompt) — otherwise their assembled prompt would never be persisted.
3035
+ if (params.data || params.templateData || systemPromptText || (params.conversationMessages?.length ?? 0) > 0) {
3036
+ const messages = [];
3037
+ if (systemPromptText) {
3038
+ // Build the system prompt content, including prefill fallback if applicable
3039
+ let systemContent = systemPromptText;
3040
+ if (prompt.AssistantPrefill && prompt.PrefillFallbackMode === 'SystemInstruction') {
3041
+ const fallbackTemplate = this.resolvePrefillFallbackText(model, vendorId);
3042
+ // Function replacement: prefill text is authored content that routinely
3043
+ // contains `$` (LaTeX `$$`, currency, JSON fragments), and a string
3044
+ // replacement would expand it. See issue #3171.
3045
+ const prefill = prompt.AssistantPrefill;
3046
+ const fallbackInstruction = fallbackTemplate.replace(/\{\{prefill\}\}/g, () => prefill);
3047
+ systemContent += '\n\n' + fallbackInstruction;
4530
3048
  }
3049
+ messages.push({
3050
+ role: 'system',
3051
+ content: systemContent
3052
+ });
4531
3053
  }
4532
- else if (modelResult.data?.usage) {
4533
- // Single attempt, use standard token tracking
4534
- promptRun.TokensUsed = modelResult.data.usage.totalTokens;
4535
- promptRun.TokensPrompt = modelResult.data.usage.promptTokens;
4536
- promptRun.TokensCompletion = modelResult.data.usage.completionTokens;
4537
- // Save cost information if available
4538
- if (modelResult.data.usage.cost !== undefined) {
4539
- promptRun.Cost = modelResult.data.usage.cost;
4540
- }
4541
- if (modelResult.data.usage.costCurrency !== undefined) {
4542
- promptRun.CostCurrency = modelResult.data.usage.costCurrency;
4543
- }
4544
- // Save timing information if available
4545
- if (modelResult.data.usage.queueTime !== undefined) {
4546
- promptRun.QueueTime = modelResult.data.usage.queueTime;
4547
- }
4548
- if (modelResult.data.usage.promptTime !== undefined) {
4549
- promptRun.PromptTime = modelResult.data.usage.promptTime;
3054
+ // Always include any caller-supplied conversation messages (previously only recorded when a
3055
+ // template system prompt was present, which dropped them for the pure-conversationMessages path).
3056
+ messages.push(...(params.conversationMessages || []));
3057
+ promptRun.Messages = JSON.stringify({
3058
+ data: params.data,
3059
+ templateData: params.templateData,
3060
+ messages: messages || [],
3061
+ });
3062
+ }
3063
+ // Populate new retry tracking columns with initial values
3064
+ promptRun.ValidationBehavior = params.validationBehavior || prompt.ValidationBehavior || 'Warn';
3065
+ promptRun.RetryStrategy = prompt.RetryStrategy || 'Fixed';
3066
+ promptRun.MaxRetriesConfigured = prompt.MaxRetries || 0;
3067
+ promptRun.FirstAttemptAt = startTime;
3068
+ promptRun.ValidationAttemptCount = 0; // Will be updated during execution
3069
+ promptRun.SuccessfulValidationCount = 0;
3070
+ promptRun.FinalValidationPassed = false; // Will be updated after execution
3071
+ }
3072
+ /**
3073
+ * Updates the AIPromptRun entity with execution results
3074
+ */
3075
+ async updatePromptRun(promptRun, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens) {
3076
+ // A chat run succeeds only if the model call succeeded AND its output did not fail validation.
3077
+ const success = modelResult.success && (parsedResult.validationResult?.Success !== false);
3078
+ return this.FinalizeRunRecord(promptRun, success, endTime, executionTimeMS, (run) => this.applyChatResultFields(run, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens));
3079
+ }
3080
+ /**
3081
+ * Populates a prompt-run's chat-specific finalized fields (result, tokens, cost, timing, validation)
3082
+ * from the model result. Runs INSIDE the post-INSERT save task — see {@link BaseModelRunner.FinalizeRunRecord},
3083
+ * which sets the completion timing, `Success` and `Status` before this runs and the rollups after it,
3084
+ * and logs (non-fatal) any error thrown here: the AIPromptRun is observability, not part of the
3085
+ * prompt's success contract.
3086
+ */
3087
+ applyChatResultFields(promptRun, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens) {
3088
+ // Determine what to save as the result
3089
+ let resultToSave;
3090
+ const rawResult = modelResult.data?.choices?.[0]?.message?.content || '';
3091
+ if (parsedResult.result === undefined ||
3092
+ parsedResult.result === null ||
3093
+ (typeof parsedResult.result === 'string' && parsedResult.result.trim().length === 0)) {
3094
+ // Use raw result as fallback when parsed result is undefined, null, or empty string
3095
+ resultToSave = rawResult;
3096
+ // Also set error message when we have to fall back to raw result
3097
+ if (!promptRun.ErrorMessage) {
3098
+ const validationErrors = parsedResult.validationResult?.Errors;
3099
+ if (validationErrors && validationErrors.length > 0) {
3100
+ promptRun.ErrorMessage = `JSON parsing/validation failed: ${validationErrors.map(e => e.Message).join('; ')}`;
4550
3101
  }
4551
- if (modelResult.data.usage.completionTime !== undefined) {
4552
- promptRun.CompletionTime = modelResult.data.usage.completionTime;
3102
+ else {
3103
+ promptRun.ErrorMessage = 'Failed to parse result into expected format; raw output saved instead';
4553
3104
  }
4554
3105
  }
4555
- // Provider prompt-cache token counts (informational; no cost is derived here). Taken from the
4556
- // final model result in both the single-attempt and retry paths — cache reads are best
4557
- // represented by the final call rather than summed across retries (which would over-count the
4558
- // re-sent prefix). 0 means "no cache activity reported", consistent with ModelUsage defaults.
4559
- if (modelResult.data?.usage) {
4560
- promptRun.TokensCacheRead = modelResult.data.usage.cacheReadTokens ?? 0;
4561
- promptRun.TokensCacheWrite = modelResult.data.usage.cacheWriteTokens ?? 0;
3106
+ }
3107
+ else if (typeof parsedResult.result === 'string') {
3108
+ resultToSave = parsedResult.result;
3109
+ }
3110
+ else {
3111
+ resultToSave = JSON.stringify(parsedResult.result);
3112
+ }
3113
+ promptRun.Result = resultToSave;
3114
+ // Extract token usage and cost - use cumulative if retries occurred
3115
+ if (cumulativeTokens && validationAttempts && validationAttempts.length > 1) {
3116
+ // Multiple attempts occurred, use cumulative totals. cumulativeTokens.promptTokens is the
3117
+ // UNCACHED ("net-new") input summed across attempts; cache reads/writes are NOT summed (the
3118
+ // re-sent prefix would over-count) and are persisted from the final model result below.
3119
+ // TokensUsed must equal TokensPrompt + TokensCompletion (AIPromptRun invariant), so it does
3120
+ // NOT include the cache buckets — those live in TokensCacheRead/TokensCacheWrite.
3121
+ promptRun.TokensPrompt = cumulativeTokens.promptTokens;
3122
+ promptRun.TokensCompletion = cumulativeTokens.completionTokens;
3123
+ promptRun.TokensUsed = cumulativeTokens.promptTokens + cumulativeTokens.completionTokens;
3124
+ promptRun.Cost = cumulativeTokens.totalCost;
3125
+ // Cost currency from the last model result
3126
+ if (modelResult.data?.usage?.costCurrency !== undefined) {
3127
+ promptRun.CostCurrency = modelResult.data.usage.costCurrency;
3128
+ }
3129
+ }
3130
+ else if (modelResult.data?.usage) {
3131
+ // Single attempt, use standard token tracking
3132
+ promptRun.TokensUsed = modelResult.data.usage.totalTokens;
3133
+ promptRun.TokensPrompt = modelResult.data.usage.promptTokens;
3134
+ promptRun.TokensCompletion = modelResult.data.usage.completionTokens;
3135
+ // Save cost information if available
3136
+ if (modelResult.data.usage.cost !== undefined) {
3137
+ promptRun.Cost = modelResult.data.usage.cost;
3138
+ }
3139
+ if (modelResult.data.usage.costCurrency !== undefined) {
3140
+ promptRun.CostCurrency = modelResult.data.usage.costCurrency;
3141
+ }
3142
+ // Save timing information if available
3143
+ if (modelResult.data.usage.queueTime !== undefined) {
3144
+ promptRun.QueueTime = modelResult.data.usage.queueTime;
3145
+ }
3146
+ if (modelResult.data.usage.promptTime !== undefined) {
3147
+ promptRun.PromptTime = modelResult.data.usage.promptTime;
3148
+ }
3149
+ if (modelResult.data.usage.completionTime !== undefined) {
3150
+ promptRun.CompletionTime = modelResult.data.usage.completionTime;
3151
+ }
3152
+ }
3153
+ // Provider prompt-cache token counts (informational; no cost is derived here). Taken from the
3154
+ // final model result in both the single-attempt and retry paths — cache reads are best
3155
+ // represented by the final call rather than summed across retries (which would over-count the
3156
+ // re-sent prefix). 0 means "no cache activity reported", consistent with ModelUsage defaults.
3157
+ if (modelResult.data?.usage) {
3158
+ promptRun.TokensCacheRead = modelResult.data.usage.cacheReadTokens ?? 0;
3159
+ promptRun.TokensCacheWrite = modelResult.data.usage.cacheWriteTokens ?? 0;
3160
+ }
3161
+ // Save model-specific response details if available
3162
+ if (modelResult.modelSpecificResponseDetails) {
3163
+ promptRun.ModelSpecificResponseDetails = JSON.stringify(modelResult.modelSpecificResponseDetails);
3164
+ }
3165
+ // Populate retry tracking columns
3166
+ if (validationAttempts && validationAttempts.length > 0) {
3167
+ // Update retry tracking columns
3168
+ promptRun.ValidationAttemptCount = validationAttempts.length;
3169
+ promptRun.SuccessfulValidationCount = validationAttempts.filter(a => a.success).length;
3170
+ promptRun.FinalValidationPassed = parsedResult.validationResult?.Success === true;
3171
+ promptRun.LastAttemptAt = endTime;
3172
+ // Calculate total retry duration (excluding first attempt)
3173
+ if (validationAttempts.length > 1) {
3174
+ const firstAttemptTime = validationAttempts[0].timestamp;
3175
+ const lastAttemptTime = validationAttempts[validationAttempts.length - 1].timestamp;
3176
+ promptRun.TotalRetryDurationMS = lastAttemptTime.getTime() - firstAttemptTime.getTime();
4562
3177
  }
4563
- // Save model-specific response details if available
4564
- if (modelResult.modelSpecificResponseDetails) {
4565
- promptRun.ModelSpecificResponseDetails = JSON.stringify(modelResult.modelSpecificResponseDetails);
3178
+ else {
3179
+ promptRun.TotalRetryDurationMS = 0;
4566
3180
  }
4567
- // Populate retry tracking columns
4568
- if (validationAttempts && validationAttempts.length > 0) {
4569
- // Update retry tracking columns
4570
- promptRun.ValidationAttemptCount = validationAttempts.length;
4571
- promptRun.SuccessfulValidationCount = validationAttempts.filter(a => a.success).length;
4572
- promptRun.FinalValidationPassed = parsedResult.validationResult?.Success === true;
4573
- promptRun.LastAttemptAt = endTime;
4574
- // Calculate total retry duration (excluding first attempt)
4575
- if (validationAttempts.length > 1) {
4576
- const firstAttemptTime = validationAttempts[0].timestamp;
4577
- const lastAttemptTime = validationAttempts[validationAttempts.length - 1].timestamp;
4578
- promptRun.TotalRetryDurationMS = lastAttemptTime.getTime() - firstAttemptTime.getTime();
4579
- }
4580
- else {
4581
- promptRun.TotalRetryDurationMS = 0;
4582
- }
4583
- // Get final validation error if any
4584
- const finalAttempt = validationAttempts[validationAttempts.length - 1];
4585
- if (!finalAttempt.success && finalAttempt.errorMessage) {
4586
- promptRun.FinalValidationError = finalAttempt.errorMessage.substring(0, 500); // Truncate to fit column
4587
- promptRun.ValidationErrorCount = finalAttempt.validationErrors?.length || 0;
4588
- }
4589
- // Find most common validation error
4590
- if (validationAttempts.some(a => !a.success)) {
4591
- const errorCounts = new Map();
4592
- validationAttempts.forEach(attempt => {
4593
- if (!attempt.success && attempt.errorMessage) {
4594
- const count = errorCounts.get(attempt.errorMessage) || 0;
4595
- errorCounts.set(attempt.errorMessage, count + 1);
4596
- }
4597
- });
4598
- if (errorCounts.size > 0) {
4599
- const [commonError] = [...errorCounts.entries()].sort((a, b) => b[1] - a[1])[0];
4600
- promptRun.CommonValidationError = commonError.substring(0, 255); // Truncate to fit column
3181
+ // Get final validation error if any
3182
+ const finalAttempt = validationAttempts[validationAttempts.length - 1];
3183
+ if (!finalAttempt.success && finalAttempt.errorMessage) {
3184
+ promptRun.FinalValidationError = finalAttempt.errorMessage.substring(0, 500); // Truncate to fit column
3185
+ promptRun.ValidationErrorCount = finalAttempt.validationErrors?.length || 0;
3186
+ }
3187
+ // Find most common validation error
3188
+ if (validationAttempts.some(a => !a.success)) {
3189
+ const errorCounts = new Map();
3190
+ validationAttempts.forEach(attempt => {
3191
+ if (!attempt.success && attempt.errorMessage) {
3192
+ const count = errorCounts.get(attempt.errorMessage) || 0;
3193
+ errorCounts.set(attempt.errorMessage, count + 1);
4601
3194
  }
4602
- }
4603
- // Store detailed attempts in JSON columns
4604
- promptRun.ValidationAttempts = JSON.stringify(validationAttempts.map(a => ({
4605
- attemptNumber: a.attemptNumber,
4606
- success: a.success,
4607
- errorMessage: a.errorMessage,
4608
- validationErrorCount: a.validationErrors?.length || 0,
4609
- timestamp: a.timestamp.toISOString(),
4610
- outputLength: a.rawOutput?.length || 0
4611
- })));
3195
+ });
3196
+ if (errorCounts.size > 0) {
3197
+ const [commonError] = [...errorCounts.entries()].sort((a, b) => b[1] - a[1])[0];
3198
+ promptRun.CommonValidationError = commonError.substring(0, 255); // Truncate to fit column
3199
+ }
3200
+ }
3201
+ // Store detailed attempts in JSON columns
3202
+ promptRun.ValidationAttempts = JSON.stringify(validationAttempts.map(a => ({
3203
+ attemptNumber: a.attemptNumber,
3204
+ success: a.success,
3205
+ errorMessage: a.errorMessage,
3206
+ validationErrorCount: a.validationErrors?.length || 0,
3207
+ timestamp: a.timestamp.toISOString(),
3208
+ outputLength: a.rawOutput?.length || 0
3209
+ })));
3210
+ promptRun.ValidationSummary = JSON.stringify({
3211
+ totalAttempts: validationAttempts.length,
3212
+ successfulAttempts: validationAttempts.filter(a => a.success).length,
3213
+ finalSuccess: parsedResult.validationResult?.Success || false,
3214
+ validationBehavior: promptRun.ValidationBehavior,
3215
+ retryStrategy: promptRun.RetryStrategy,
3216
+ maxRetriesConfigured: promptRun.MaxRetriesConfigured,
3217
+ actualRetriesUsed: validationAttempts.length - 1,
3218
+ totalDurationMS: executionTimeMS,
3219
+ retryDurationMS: promptRun.TotalRetryDurationMS || 0,
3220
+ outputType: prompt.OutputType || 'unknown',
3221
+ hasOutputExample: !!(prompt.OutputExample),
3222
+ schemaValidationUsed: !!(prompt.OutputExample && prompt.OutputType === 'object'),
3223
+ finalValidationErrors: parsedResult.validationResult?.Errors?.map(e => ({
3224
+ source: e.Source,
3225
+ message: e.Message,
3226
+ type: e.Type,
3227
+ value: e.Value
3228
+ })) || [],
3229
+ validationDecision: this.getValidationDecisionDescription(parsedResult.validationResult?.Success || false, validationAttempts.length, promptRun.ValidationBehavior || 'Warn'),
3230
+ jsonRepairInfo: promptRun._jsonRepairInfo || null
3231
+ });
3232
+ }
3233
+ else {
3234
+ // No validation attempts (possibly skipped validation)
3235
+ promptRun.ValidationAttemptCount = 1; // At least one attempt was made
3236
+ promptRun.SuccessfulValidationCount = parsedResult.validationResult?.Success !== false ? 1 : 0;
3237
+ promptRun.FinalValidationPassed = parsedResult.validationResult?.Success !== false;
3238
+ promptRun.LastAttemptAt = endTime;
3239
+ promptRun.TotalRetryDurationMS = 0;
3240
+ // Even without validation, persist JSON repair info if a repair occurred
3241
+ if (promptRun._jsonRepairInfo) {
4612
3242
  promptRun.ValidationSummary = JSON.stringify({
4613
- totalAttempts: validationAttempts.length,
4614
- successfulAttempts: validationAttempts.filter(a => a.success).length,
4615
- finalSuccess: parsedResult.validationResult?.Success || false,
4616
- validationBehavior: promptRun.ValidationBehavior,
4617
- retryStrategy: promptRun.RetryStrategy,
4618
- maxRetriesConfigured: promptRun.MaxRetriesConfigured,
4619
- actualRetriesUsed: validationAttempts.length - 1,
4620
- totalDurationMS: executionTimeMS,
4621
- retryDurationMS: promptRun.TotalRetryDurationMS || 0,
4622
- outputType: prompt.OutputType || 'unknown',
4623
- hasOutputExample: !!(prompt.OutputExample),
4624
- schemaValidationUsed: !!(prompt.OutputExample && prompt.OutputType === 'object'),
4625
- finalValidationErrors: parsedResult.validationResult?.Errors?.map(e => ({
4626
- source: e.Source,
4627
- message: e.Message,
4628
- type: e.Type,
4629
- value: e.Value
4630
- })) || [],
4631
- validationDecision: this.getValidationDecisionDescription(parsedResult.validationResult?.Success || false, validationAttempts.length, promptRun.ValidationBehavior || 'Warn'),
4632
- jsonRepairInfo: promptRun._jsonRepairInfo || null
3243
+ jsonRepairInfo: promptRun._jsonRepairInfo
4633
3244
  });
4634
3245
  }
4635
- else {
4636
- // No validation attempts (possibly skipped validation)
4637
- promptRun.ValidationAttemptCount = 1; // At least one attempt was made
4638
- promptRun.SuccessfulValidationCount = parsedResult.validationResult?.Success !== false ? 1 : 0;
4639
- promptRun.FinalValidationPassed = parsedResult.validationResult?.Success !== false;
4640
- promptRun.LastAttemptAt = endTime;
4641
- promptRun.TotalRetryDurationMS = 0;
4642
- // Even without validation, persist JSON repair info if a repair occurred
4643
- if (promptRun._jsonRepairInfo) {
4644
- promptRun.ValidationSummary = JSON.stringify({
4645
- jsonRepairInfo: promptRun._jsonRepairInfo
4646
- });
4647
- }
4648
- }
4649
- // Set Success flag based on validation result
4650
- promptRun.Success = modelResult.success && (parsedResult.validationResult?.Success !== false);
4651
- // Set final Status based on success
4652
- promptRun.Status = promptRun.Success ? 'Completed' : 'Failed';
4653
- // Set ErrorDetails if failed
4654
- if (!promptRun.Success) {
4655
- if (!modelResult.success && modelResult.errorMessage) {
4656
- promptRun.ErrorDetails = modelResult.errorMessage;
4657
- }
4658
- else if (parsedResult.validationResult?.Success === false) {
4659
- promptRun.ErrorDetails = `Validation failed: ${parsedResult.validationResult.Errors?.map(e => e.Message).join(', ')}`;
4660
- }
3246
+ }
3247
+ // Success and Status were set by FinalizeRunRecord from the outcome updatePromptRun passed it.
3248
+ // Set ErrorDetails if failed
3249
+ if (!promptRun.Success) {
3250
+ if (!modelResult.success && modelResult.errorMessage) {
3251
+ promptRun.ErrorDetails = modelResult.errorMessage;
4661
3252
  }
4662
- // Note: Failover tracking fields are now updated directly in executeModelWithFailover
4663
- // The promptRun entity already has the failover information set
4664
- // With template composition, we only execute once so rollup equals regular fields
4665
- promptRun.TokensPromptRollup = promptRun.TokensPrompt;
4666
- promptRun.TokensCompletionRollup = promptRun.TokensCompletion;
4667
- promptRun.TokensUsedRollup = promptRun.TokensUsed;
4668
- promptRun.TokensCacheReadRollup = promptRun.TokensCacheRead;
4669
- promptRun.TokensCacheWriteRollup = promptRun.TokensCacheWrite;
4670
- if (promptRun.Cost !== undefined) {
4671
- promptRun.TotalCost = promptRun.Cost;
3253
+ else if (parsedResult.validationResult?.Success === false) {
3254
+ promptRun.ErrorDetails = `Validation failed: ${parsedResult.validationResult.Errors?.map(e => e.Message).join(', ')}`;
4672
3255
  }
4673
3256
  }
4674
- catch (error) {
4675
- this.logError(error, {
4676
- category: 'PromptRunUpdate',
4677
- metadata: {
4678
- promptRunId: promptRun.ID
4679
- }
4680
- });
4681
- }
4682
- }
4683
- // ==================== CONTEXT LENGTH METHODS ====================
4684
- /**
4685
- * Estimates the number of tokens in a rendered prompt and conversation messages.
4686
- * This is a rough estimation based on character count and typical token ratios.
4687
- *
4688
- * @param renderedPrompt - The rendered prompt text
4689
- * @param conversationMessages - Optional conversation messages
4690
- * @returns Estimated token count
4691
- */
4692
- // ==================== FAILOVER METHODS ====================
4693
- /**
4694
- * Retrieves failover configuration from the prompt entity.
4695
- *
4696
- * @param prompt - The AI prompt entity containing failover settings
4697
- * @returns FailoverConfiguration object with strategy and settings
4698
- *
4699
- * @remarks
4700
- * This method extracts failover configuration from the prompt entity and provides
4701
- * default values when configuration is not specified. Override this method to
4702
- * implement custom failover configuration logic.
4703
- */
4704
- getFailoverConfiguration(prompt) {
4705
- return {
4706
- strategy: prompt.FailoverStrategy || 'None',
4707
- maxAttempts: prompt.FailoverMaxAttempts || 3,
4708
- delaySeconds: prompt.FailoverDelaySeconds || 1,
4709
- modelStrategy: prompt.FailoverModelStrategy || 'PreferSameModel',
4710
- errorScope: prompt.FailoverErrorScope || 'All'
4711
- };
4712
- }
4713
- /**
4714
- * Determines whether a failover attempt should be made based on the error and configuration.
4715
- *
4716
- * @param error - The error that occurred during execution
4717
- * @param config - The failover configuration
4718
- * @param attemptNumber - The current attempt number (1-based)
4719
- * @returns True if failover should be attempted, false otherwise
4720
- *
4721
- * @remarks
4722
- * This method uses the ErrorAnalyzer to classify errors and determine if they are
4723
- * eligible for failover based on the configured error scope. Override this method
4724
- * to implement custom failover decision logic.
4725
- */
4726
- shouldAttemptFailover(error, config, attemptNumber) {
4727
- // Don't failover if strategy is None or we've exceeded max attempts
4728
- if (config.strategy === 'None' || attemptNumber > config.maxAttempts) {
4729
- return false;
4730
- }
4731
- // Analyze the error to determine if it's eligible for failover
4732
- const errorAnalysis = ErrorAnalyzer.analyzeError(error);
4733
- // Check if error analysis allows failover
4734
- if (!errorAnalysis.canFailover) {
4735
- return false;
4736
- }
4737
- // Check error scope configuration
4738
- switch (config.errorScope) {
4739
- case 'NetworkOnly':
4740
- return errorAnalysis.errorType === 'NetworkError';
4741
- case 'RateLimitOnly':
4742
- return errorAnalysis.errorType === 'RateLimit';
4743
- case 'ServiceErrorOnly':
4744
- return errorAnalysis.errorType === 'ServiceUnavailable' ||
4745
- errorAnalysis.errorType === 'InternalServerError';
4746
- case 'All':
4747
- default:
4748
- return true;
4749
- }
4750
- }
4751
- /**
4752
- * Checks if an error type matches the configured error scope
4753
- *
4754
- * @param errorType - The error type from ErrorAnalyzer
4755
- * @param scope - The configured error scope
4756
- * @returns True if the error matches the scope
4757
- */
4758
- errorMatchesScope(errorType, scope) {
4759
- switch (scope) {
4760
- case 'NetworkOnly':
4761
- return errorType === 'NetworkError';
4762
- case 'RateLimitOnly':
4763
- return errorType === 'RateLimit';
4764
- case 'ServiceErrorOnly':
4765
- return errorType === 'ServiceUnavailable' || errorType === 'InternalServerError';
4766
- case 'All':
4767
- default:
4768
- return true;
4769
- }
4770
3257
  }
4771
3258
  /**
4772
- * Calculates the delay before the next failover attempt.
4773
- *
4774
- * @param attemptNumber - The current attempt number (1-based)
4775
- * @param baseDelaySeconds - The base delay in seconds from configuration
4776
- * @param previousError - The error from the previous attempt
4777
- * @returns Delay in milliseconds before the next attempt
4778
- *
4779
- * @remarks
4780
- * Implements exponential backoff with jitter by default. The delay increases
4781
- * exponentially with each attempt and includes random jitter to prevent
4782
- * thundering herd problems. Override this method to implement custom delay logic.
4783
- */
4784
- calculateFailoverDelay(attemptNumber, baseDelaySeconds) {
4785
- // Exponential backoff: delay = base * 2^(attempt-1)
4786
- const exponentialDelay = baseDelaySeconds * Math.pow(2, attemptNumber - 1);
4787
- // Add jitter (0-25% of delay) to prevent thundering herd
4788
- const jitter = exponentialDelay * 0.25 * Math.random();
4789
- // Cap at 30 seconds to prevent excessive delays
4790
- const totalDelay = Math.min(exponentialDelay + jitter, 30);
4791
- return totalDelay * 1000; // Convert to milliseconds
4792
- }
4793
- /**
4794
- * Selects candidate models for failover based on the strategy and current failure.
4795
- *
4796
- * @param currentModel - The model that just failed
4797
- * @param currentVendorId - The vendor ID that just failed
4798
- * @param strategy - The failover strategy to use
4799
- * @param modelStrategy - The model selection preference
4800
- * @param allCandidates - All available model-vendor candidates
4801
- * @param attemptHistory - History of previous failover attempts
4802
- * @returns Array of candidates sorted by priority (highest first)
4803
- *
4804
- * @remarks
4805
- * This method implements different strategies for selecting failover candidates:
4806
- * - SameModelDifferentVendor: Try the same model with different vendors
4807
- * - NextBestModel: Try different models in order of preference
4808
- * - PowerRank: Use the global power ranking of models
4809
- *
4810
- * Override this method to implement custom candidate selection logic.
3259
+ * Provides a human-readable description of the validation decision
4811
3260
  */
4812
- selectFailoverCandidates(currentModel, currentVendorId, strategy, modelStrategy, allCandidates, attemptHistory) {
4813
- // Filter out candidates that have already failed
4814
- // Note: Authentication errors are already filtered from allCandidates upstream,
4815
- // so we only need to filter out specific model/vendor pairs that have failed
4816
- const failedPairs = new Set(attemptHistory.map(a => `${a.modelId}:${a.vendorId || 'default'}`));
4817
- const availableCandidates = allCandidates.filter(c => {
4818
- const key = `${c.model.ID}:${c.vendorId || 'default'}`;
4819
- return !failedPairs.has(key);
4820
- });
4821
- // Check if we have context length exceeded errors in the attempt history
4822
- const hasContextLengthError = attemptHistory.some(a => a.errorType === 'ContextLengthExceeded' ||
4823
- ErrorAnalyzer.analyzeError(a.error).errorType === 'ContextLengthExceeded');
4824
- // Apply strategy-specific filtering and sorting
4825
- let candidates;
4826
- switch (strategy) {
4827
- case 'SameModelDifferentVendor':
4828
- // Only consider same model with different vendors
4829
- candidates = availableCandidates.filter(c => UUIDsEqual(c.model.ID, currentModel.ID) && !UUIDsEqual(c.vendorId, currentVendorId));
4830
- break;
4831
- case 'NextBestModel':
4832
- // Consider all models, apply model strategy preference
4833
- candidates = availableCandidates;
4834
- if (modelStrategy === 'RequireSameModel') {
4835
- candidates = candidates.filter(c => UUIDsEqual(c.model.ID, currentModel.ID));
4836
- }
4837
- else if (modelStrategy === 'PreferSameModel') {
4838
- // Sort to put same model first
4839
- candidates.sort((a, b) => {
4840
- const aSameModel = UUIDsEqual(a.model.ID, currentModel.ID) ? 1 : 0;
4841
- const bSameModel = UUIDsEqual(b.model.ID, currentModel.ID) ? 1 : 0;
4842
- return bSameModel - aSameModel;
4843
- });
4844
- }
4845
- else if (modelStrategy === 'PreferDifferentModel') {
4846
- // Sort to put different models first
4847
- candidates.sort((a, b) => {
4848
- const aDiffModel = !UUIDsEqual(a.model.ID, currentModel.ID) ? 1 : 0;
4849
- const bDiffModel = !UUIDsEqual(b.model.ID, currentModel.ID) ? 1 : 0;
4850
- return bDiffModel - aDiffModel;
4851
- });
4852
- }
4853
- break;
4854
- case 'PowerRank':
4855
- // Use all candidates, they're already sorted by power rank
4856
- candidates = availableCandidates;
4857
- break;
4858
- default:
4859
- candidates = [];
4860
- }
4861
- // If we have context length errors, prioritize models with larger context windows
4862
- if (hasContextLengthError) {
4863
- const currentMaxTokens = currentModel.ModelVendors?.length > 0 ?
4864
- Math.max(...currentModel.ModelVendors.map(mv => mv.MaxInputTokens || 0)) : 0;
4865
- // Filter out models with same or smaller context windows
4866
- candidates = candidates.filter(c => {
4867
- const candidateMaxTokens = c.model.ModelVendors?.length > 0 ?
4868
- Math.max(...c.model.ModelVendors.map(mv => mv.MaxInputTokens || 0)) : 0;
4869
- return candidateMaxTokens > currentMaxTokens;
4870
- });
4871
- // If no larger models exist, this is a fatal error - return empty to stop retrying
4872
- if (candidates.length === 0) {
4873
- LogStatusEx({
4874
- message: `❌ Context length exceeded and no models with larger context windows available. Current model: ${currentModel.Name} (${currentMaxTokens} max tokens). This is a fatal error.`,
4875
- category: 'AI',
4876
- additionalArgs: [{
4877
- currentModel: currentModel.Name,
4878
- currentMaxTokens,
4879
- availableModels: allCandidates.map(c => c.model.Name).join(', '),
4880
- reason: 'No models with larger context windows available for failover'
4881
- }]
4882
- });
4883
- // Return empty array - caller will see no candidates and stop retrying
4884
- return [];
4885
- }
4886
- // Sort by priority first (existing algorithm), then by context window size as tiebreaker
4887
- candidates.sort((a, b) => {
4888
- // Primary sort: priority (higher is better) - maintains existing algorithm
4889
- if (a.priority !== b.priority) {
4890
- return b.priority - a.priority;
4891
- }
4892
- // Secondary sort: context window size (largest first) - only as tiebreaker
4893
- const aMaxTokens = a.model.ModelVendors?.length > 0 ?
4894
- Math.max(...a.model.ModelVendors.map((mv) => mv.MaxInputTokens || 0)) : 0;
4895
- const bMaxTokens = b.model.ModelVendors?.length > 0 ?
4896
- Math.max(...b.model.ModelVendors.map((mv) => mv.MaxInputTokens || 0)) : 0;
4897
- return bMaxTokens - aMaxTokens;
4898
- });
4899
- // Log context-aware failover selection
4900
- const bestCandidate = candidates[0];
4901
- const bestCandidateMaxTokens = bestCandidate.model.ModelVendors?.length > 0 ?
4902
- Math.max(...bestCandidate.model.ModelVendors.map((mv) => mv.MaxInputTokens || 0)) : 0;
4903
- LogStatusEx({
4904
- message: `🔄 Context-aware failover: Selected model ${bestCandidate.model.Name} with ${bestCandidateMaxTokens} max input tokens (vs ${currentMaxTokens} for failed model)`,
4905
- category: 'AI',
4906
- additionalArgs: [{
4907
- currentModel: currentModel.Name,
4908
- currentMaxTokens,
4909
- selectedModel: bestCandidate.model.Name,
4910
- selectedMaxTokens: bestCandidateMaxTokens,
4911
- candidateCount: candidates.length
4912
- }]
4913
- });
3261
+ getValidationDecisionDescription(finalSuccess, totalAttempts, validationBehavior) {
3262
+ if (finalSuccess) {
3263
+ return totalAttempts === 1
3264
+ ? 'Validation passed on first attempt'
3265
+ : `Validation passed after ${totalAttempts} attempts`;
4914
3266
  }
4915
3267
  else {
4916
- // Final sort by priority (higher is better) for non-context-length errors
4917
- candidates.sort((a, b) => b.priority - a.priority);
3268
+ switch (validationBehavior) {
3269
+ case 'Strict':
3270
+ return `Validation failed after ${totalAttempts} attempts - execution marked as failed (Strict mode)`;
3271
+ case 'Warn':
3272
+ return `Validation failed after ${totalAttempts} attempts - warning logged, execution continued (Warn mode)`;
3273
+ case 'None':
3274
+ return `Validation skipped or ignored (None mode)`;
3275
+ default:
3276
+ return `Validation failed after ${totalAttempts} attempts - behavior: ${validationBehavior}`;
3277
+ }
4918
3278
  }
4919
- return candidates;
4920
3279
  }
4921
3280
  /**
4922
- * Logs a failover attempt for tracking and debugging.
4923
- *
4924
- * @param promptId - The ID of the prompt being executed
4925
- * @param attempt - The failover attempt details
4926
- * @param willRetry - Whether another attempt will be made
4927
- *
4928
- * @remarks
4929
- * This method logs detailed information about each failover attempt to help with
4930
- * debugging and monitoring. Override this method to implement custom logging or
4931
- * integrate with external monitoring systems.
3281
+ * Resolves the scalar inference parameters for a run: each value is the per-request override
3282
+ * from `additionalParameters` when supplied, otherwise the prompt's configured default. This
3283
+ * is the single source of truth for parameter precedence so {@link executeModel} (ChatParams)
3284
+ * and {@link createPromptRun} (the persisted record) stay in lockstep. Stop sequences and
3285
+ * assistant prefill are intentionally excluded — their representations differ per target.
4932
3286
  */
4933
- logFailoverAttempt(promptId, attempt, willRetry) {
4934
- const message = `Failover attempt ${attempt.attemptNumber} for prompt ${promptId}`;
4935
- const metadata = {
4936
- promptId,
4937
- attemptNumber: attempt.attemptNumber,
4938
- modelId: attempt.modelId,
4939
- vendorId: attempt.vendorId,
4940
- errorType: attempt.errorType,
4941
- duration: attempt.duration,
4942
- willRetry,
4943
- error: attempt.error.message
3287
+ resolveScalarInferenceParams(prompt, additionalParameters) {
3288
+ const pick = (override, promptDefault) => override !== undefined ? override : (promptDefault != null ? promptDefault : undefined);
3289
+ const ap = additionalParameters;
3290
+ return {
3291
+ temperature: pick(ap?.temperature, prompt.Temperature),
3292
+ topP: pick(ap?.topP, prompt.TopP),
3293
+ topK: pick(ap?.topK, prompt.TopK),
3294
+ minP: pick(ap?.minP, prompt.MinP),
3295
+ frequencyPenalty: pick(ap?.frequencyPenalty, prompt.FrequencyPenalty),
3296
+ presencePenalty: pick(ap?.presencePenalty, prompt.PresencePenalty),
3297
+ seed: pick(ap?.seed, prompt.Seed),
3298
+ includeLogProbs: pick(ap?.includeLogProbs, prompt.IncludeLogProbs),
3299
+ topLogProbs: pick(ap?.topLogProbs, prompt.TopLogProbs),
4944
3300
  };
4945
- if (willRetry) {
4946
- LogStatusEx({
4947
- message: `⚡ ${message}`,
4948
- category: 'AI',
4949
- additionalArgs: [metadata]
4950
- });
4951
- }
4952
- else {
4953
- LogErrorEx({
4954
- message: message,
4955
- error: attempt.error,
4956
- category: 'AI',
4957
- severity: 'error',
4958
- metadata: metadata
4959
- });
4960
- }
4961
3301
  }
4962
3302
  }
4963
3303
  //# sourceMappingURL=AIPromptRunner.js.map