@memberjunction/ai-prompts 5.40.1 → 5.41.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -26,6 +26,21 @@ function mimeFromBlockType(type) {
26
26
  }
27
27
  }
28
28
  export class AIPromptRunner {
29
+ /**
30
+ * Process-wide cache of parsed `OutputExample` JSON, keyed by the raw example string.
31
+ * A prompt's OutputExample is a static string reused across every run and every validation
32
+ * retry, so re-parsing it each time is pure waste. Keyed by content (not prompt ID) so two
33
+ * prompts sharing an identical example share one parsed entry and an edited example never
34
+ * serves a stale parse. Stores `{ parsed }` on success or `{ error }` on failure so we cache
35
+ * the failure too rather than re-throwing-and-reparsing bad JSON every attempt.
36
+ */
37
+ static { this._outputExampleCache = new Map(); }
38
+ /**
39
+ * Marker used in `AIModelSelectionInfo.modelsConsidered[].unavailableReason` for candidates
40
+ * that were intentionally NOT credential-checked because a higher-priority candidate had
41
+ * already been selected. See the DECISION note in {@link selectModelWithAPIKeyTracked}.
42
+ */
43
+ static { this.NOT_EVALUATED_REASON = 'Not evaluated (a higher-priority candidate was already selected; set AIPromptParams.forceFullModelEvaluation to probe all)'; }
29
44
  /**
30
45
  * Optional metadata provider override. Callers should set
31
46
  * `instance.Provider = providerToUse` before invoking run methods
@@ -39,6 +54,16 @@ export class AIPromptRunner {
39
54
  }
40
55
  constructor() {
41
56
  this._provider = null;
57
+ /**
58
+ * Instance-keyed chain of in-flight AIPromptRun saves. Mirrors the BaseAgent step-save pattern:
59
+ * prompt-run persistence is fire-and-forget so the execution path never blocks on a DB
60
+ * round-trip, but saves for the SAME entity are sequenced — the initial 'Running' INSERT always
61
+ * completes before the finalize UPDATE, so a slow INSERT can never clobber the finalized row.
62
+ * Keyed by the entity INSTANCE (stable), not its ID. See {@link queuePromptRunSave}.
63
+ */
64
+ this._promptRunSaveChains = new Map();
65
+ /** All queued prompt-run save promises, for optional flushing via {@link WaitForPendingPromptRunSaves}. */
66
+ this._pendingPromptRunSaves = [];
42
67
  this._metadata = this._provider ?? new Metadata();
43
68
  this._templateEngine = TemplateEngineServer.Instance;
44
69
  this._executionPlanner = new ExecutionPlanner();
@@ -116,19 +141,15 @@ export class AIPromptRunner {
116
141
  });
117
142
  }
118
143
  /**
119
- * Checks if a model vendor is configured as an inference provider
144
+ * Checks if a model vendor is configured as an inference provider.
145
+ * Delegates to the memoized {@link AIEngine.IsInferenceProvider} helper so the
146
+ * "Inference Provider" vendor-type lookup happens once per engine load rather than on
147
+ * every candidate in every selection pass.
120
148
  * @param modelVendor The model vendor to check
121
149
  * @returns true if the vendor is an inference provider
122
150
  */
123
151
  isInferenceProvider(modelVendor) {
124
- // Find the inference provider type from cached vendor type definitions
125
- const inferenceProviderType = AIEngine.Instance.VendorTypeDefinitions.find(vt => vt.Name === 'Inference Provider');
126
- if (!inferenceProviderType) {
127
- // Fallback to checking if it's not a model developer (should rarely happen)
128
- const modelDeveloperType = AIEngine.Instance.VendorTypeDefinitions.find(vt => vt.Name === 'Model Developer');
129
- return !UUIDsEqual(modelVendor.TypeID, modelDeveloperType?.ID);
130
- }
131
- return UUIDsEqual(modelVendor.TypeID, inferenceProviderType.ID);
152
+ return AIEngine.Instance.IsInferenceProvider(modelVendor);
132
153
  }
133
154
  /**
134
155
  * Resolves credentials for AI model execution using a hierarchical resolution system.
@@ -171,7 +192,8 @@ export class AIPromptRunner {
171
192
  }
172
193
  // Priority 3: ModelVendor bindings - with failover
173
194
  if (modelId && vendorId) {
174
- const modelVendor = AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.ModelID, modelId) && UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
195
+ const modelVendor = AIEngine.Instance.ModelVendorsByModelID.get(NormalizeUUID(modelId))
196
+ ?.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
175
197
  if (modelVendor) {
176
198
  const bindings = AIEngine.Instance.GetCredentialBindingsForTarget('ModelVendor', modelVendor.ID);
177
199
  const result = await this.tryCredentialBindingsWithFailover(bindings, 'AICredentialBinding(ModelVendor)', params, verbose);
@@ -189,7 +211,7 @@ export class AIPromptRunner {
189
211
  // Priority 5: Type-based default credential
190
212
  // If the vendor declares a CredentialTypeID, try to find a default credential of that type
191
213
  if (vendorId) {
192
- const vendor = AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, vendorId));
214
+ const vendor = AIEngine.Instance.VendorsByID.get(NormalizeUUID(vendorId));
193
215
  if (vendor?.CredentialTypeID) {
194
216
  const defaultCredential = this.findDefaultCredentialByType(vendor.CredentialTypeID);
195
217
  if (defaultCredential) {
@@ -348,7 +370,8 @@ export class AIPromptRunner {
348
370
  }
349
371
  // Priority 3: ModelVendor bindings
350
372
  if (modelId && vendorId) {
351
- const modelVendor = AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.ModelID, modelId) && UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
373
+ const modelVendor = AIEngine.Instance.ModelVendorsByModelID.get(NormalizeUUID(modelId))
374
+ ?.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
352
375
  if (modelVendor && AIEngine.Instance.HasCredentialBindings('ModelVendor', modelVendor.ID)) {
353
376
  return true;
354
377
  }
@@ -361,7 +384,7 @@ export class AIPromptRunner {
361
384
  }
362
385
  // Priority 5: Type-based default credential
363
386
  if (vendorId) {
364
- const vendor = AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, vendorId));
387
+ const vendor = AIEngine.Instance.VendorsByID.get(NormalizeUUID(vendorId));
365
388
  if (vendor?.CredentialTypeID) {
366
389
  const defaultCredential = this.findDefaultCredentialByType(vendor.CredentialTypeID);
367
390
  if (defaultCredential) {
@@ -607,8 +630,8 @@ export class AIPromptRunner {
607
630
  // we received model selection info, need to lookup vendor driver class and api name from there
608
631
  const vendorID = modelSelectionInfo.vendorSelected?.ID;
609
632
  const modelID = modelSelectionInfo.modelSelected.ID;
610
- const modelVendor = AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, vendorID) &&
611
- UUIDsEqual(mv.ModelID, modelID));
633
+ const modelVendor = AIEngine.Instance.ModelVendorsByModelID.get(NormalizeUUID(modelID))
634
+ ?.find(mv => UUIDsEqual(mv.VendorID, vendorID));
612
635
  if (modelVendor) {
613
636
  vendorDriverClass = modelVendor.DriverClass;
614
637
  vendorApiName = modelVendor.APIName;
@@ -852,18 +875,8 @@ export class AIPromptRunner {
852
875
  // Set Status and WasSelectedResult for parallel execution
853
876
  consolidatedPromptRun.Status = parallelResult.successCount > 0 ? 'Completed' : 'Failed';
854
877
  consolidatedPromptRun.WasSelectedResult = true; // This is the consolidated result chosen by judge
855
- const saveResult = await consolidatedPromptRun.Save();
856
- if (!saveResult) {
857
- this.logError(`Failed to save consolidated AIPromptRun: ${consolidatedPromptRun.LatestResult?.CompleteMessage || 'Unknown error'}`, {
858
- category: 'ConsolidatedPromptRunSave',
859
- metadata: {
860
- promptRunId: consolidatedPromptRun.ID,
861
- executionTasks: executionTasks.length,
862
- successfulResults: successfulResults.length
863
- },
864
- maxErrorLength: params.maxErrorLength
865
- });
866
- }
878
+ // Persist the consolidated run fire-and-forget; chains after its INSERT via the save queue.
879
+ this.queuePromptRunSave(consolidatedPromptRun);
867
880
  // Create additional results from all other successful results (excluding the best one)
868
881
  const additionalResults = [];
869
882
  // Sort successful results by ranking (if available) or keep original order
@@ -1225,7 +1238,7 @@ export class AIPromptRunner {
1225
1238
  }
1226
1239
  // Get configuration info if provided
1227
1240
  if (configurationId) {
1228
- configuration = AIEngine.Instance.Configurations.find(c => UUIDsEqual(c.ID, configurationId));
1241
+ configuration = AIEngine.Instance.ConfigurationsByID.get(NormalizeUUID(configurationId));
1229
1242
  configurationName = configuration?.Name;
1230
1243
  }
1231
1244
  // Build unified list of model-vendor candidates
@@ -1309,7 +1322,7 @@ export class AIPromptRunner {
1309
1322
  // Get selected vendor entity
1310
1323
  let selectedVendor;
1311
1324
  if (selected.vendorId) {
1312
- selectedVendor = AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, selected.vendorId));
1325
+ selectedVendor = AIEngine.Instance.VendorsByID.get(NormalizeUUID(selected.vendorId));
1313
1326
  }
1314
1327
  return {
1315
1328
  model: selected.model,
@@ -1385,7 +1398,7 @@ export class AIPromptRunner {
1385
1398
  * Returns candidates for the single model if it's active and compatible.
1386
1399
  */
1387
1400
  buildCandidatesForExplicitModel(explicitModelId, prompt, preferredVendorId) {
1388
- const model = AIEngine.Instance.Models.find(m => UUIDsEqual(m.ID, explicitModelId));
1401
+ const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(explicitModelId));
1389
1402
  if (!model || !model.IsActive) {
1390
1403
  return [];
1391
1404
  }
@@ -1480,7 +1493,7 @@ export class AIPromptRunner {
1480
1493
  return 0;
1481
1494
  const modelsWithPower = promptModels
1482
1495
  .map(pm => {
1483
- const model = AIEngine.Instance.Models.find(m => UUIDsEqual(m.ID, pm.ModelID));
1496
+ const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1484
1497
  return { powerRank: model?.PowerRank ?? 0, priority: pm.Priority || 1 };
1485
1498
  });
1486
1499
  const totalWeight = modelsWithPower.reduce((sum, m) => sum + m.priority, 0);
@@ -1510,7 +1523,7 @@ export class AIPromptRunner {
1510
1523
  */
1511
1524
  buildCandidatesForGeneralSelection(prompt, configurationId, preferredVendorId, verbose) {
1512
1525
  const preferredVendorName = preferredVendorId ?
1513
- AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, preferredVendorId))?.Name : undefined;
1526
+ AIEngine.Instance.VendorsByID.get(NormalizeUUID(preferredVendorId))?.Name : undefined;
1514
1527
  // Get prompt models for configuration
1515
1528
  const promptModels = this.getPromptModelsForConfiguration(prompt, configurationId);
1516
1529
  const candidates = [];
@@ -1582,7 +1595,7 @@ export class AIPromptRunner {
1582
1595
  const pm = promptModels[i];
1583
1596
  // Compute priority as inverse of array position so highest-priority (first) gets the largest number
1584
1597
  const computedPriority = promptModels.length - i;
1585
- const model = AIEngine.Instance.Models.find(m => UUIDsEqual(m.ID, pm.ModelID));
1598
+ const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1586
1599
  if (!model || !model.IsActive)
1587
1600
  continue;
1588
1601
  if (pm.VendorID) {
@@ -1604,8 +1617,9 @@ export class AIPromptRunner {
1604
1617
  * Helper: Create candidate for specific vendor from AIPromptModel.
1605
1618
  */
1606
1619
  createCandidateForSpecificVendor(model, promptModel, computedPriority = 0) {
1607
- const modelVendor = AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.ModelID, promptModel.ModelID) &&
1608
- UUIDsEqual(mv.VendorID, promptModel.VendorID) &&
1620
+ // Use the model's precomputed ModelVendors (grouped at engine load) instead of scanning
1621
+ // the global ModelVendors array — model.ID === promptModel.ModelID here.
1622
+ const modelVendor = model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, promptModel.VendorID) &&
1609
1623
  mv.Status === 'Active' &&
1610
1624
  this.isInferenceProvider(mv));
1611
1625
  if (!modelVendor)
@@ -1627,9 +1641,8 @@ export class AIPromptRunner {
1627
1641
  * Helper: Create candidates for all vendors of a model, sorted by vendor priority.
1628
1642
  */
1629
1643
  createCandidatesForAllVendors(model, computedPriority = 0) {
1630
- const vendors = AIEngine.Instance.ModelVendors
1631
- .filter(mv => UUIDsEqual(mv.ModelID, model.ID) &&
1632
- mv.Status === 'Active' &&
1644
+ const vendors = model.ModelVendors
1645
+ .filter(mv => mv.Status === 'Active' &&
1633
1646
  this.isInferenceProvider(mv))
1634
1647
  .sort((a, b) => (b.Priority || 0) - (a.Priority || 0));
1635
1648
  const candidates = [];
@@ -1691,7 +1704,7 @@ export class AIPromptRunner {
1691
1704
  */
1692
1705
  addPromptSpecificCandidates(candidates, promptModels, preferredVendorId) {
1693
1706
  for (const pm of promptModels) {
1694
- const model = AIEngine.Instance.Models.find(m => UUIDsEqual(m.ID, pm.ModelID));
1707
+ const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1695
1708
  if (model && model.IsActive) {
1696
1709
  const modelCandidates = this.createCandidatesForModel(model, 5000, 'prompt-model', preferredVendorId, pm.Priority);
1697
1710
  candidates.push(...modelCandidates);
@@ -1714,7 +1727,7 @@ export class AIPromptRunner {
1714
1727
  LogStatus(`Adding ${parentModels.length} models from parent config "${parentConfig.Name}" as fallback`);
1715
1728
  }
1716
1729
  for (const pm of parentModels) {
1717
- const model = AIEngine.Instance.Models.find(m => UUIDsEqual(m.ID, pm.ModelID));
1730
+ const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1718
1731
  if (model && model.IsActive) {
1719
1732
  // Decrease base priority for each level up the chain (3000, 2500, 2000, etc.)
1720
1733
  const basePriority = 3000 - (i * 500);
@@ -1731,7 +1744,7 @@ export class AIPromptRunner {
1731
1744
  LogStatus(`Adding ${nullConfigModels.length} NULL configuration models as universal fallback`);
1732
1745
  }
1733
1746
  for (const pm of nullConfigModels) {
1734
- const model = AIEngine.Instance.Models.find(m => UUIDsEqual(m.ID, pm.ModelID));
1747
+ const model = AIEngine.Instance.ModelsByID.get(NormalizeUUID(pm.ModelID));
1735
1748
  if (model && model.IsActive) {
1736
1749
  const modelCandidates = this.createCandidatesForModel(model, 1000, // Lowest priority tier
1737
1750
  'prompt-model', preferredVendorId, pm.Priority);
@@ -1759,8 +1772,7 @@ export class AIPromptRunner {
1759
1772
  return AIEngine.Instance.Models.filter(m => m.IsActive &&
1760
1773
  (!prompt.AIModelTypeID || UUIDsEqual(m.AIModelTypeID, prompt.AIModelTypeID)) &&
1761
1774
  (!preferredVendorName ||
1762
- AIEngine.Instance.ModelVendors.some(mv => UUIDsEqual(mv.ModelID, m.ID) &&
1763
- mv.Status === 'Active' &&
1775
+ m.ModelVendors.some(mv => mv.Status === 'Active' &&
1764
1776
  mv.Vendor === preferredVendorName &&
1765
1777
  this.isInferenceProvider(mv))));
1766
1778
  }
@@ -1801,9 +1813,11 @@ export class AIPromptRunner {
1801
1813
  */
1802
1814
  createCandidatesForModel(model, basePriority, source, preferredVendorId, promptModelPriority) {
1803
1815
  const modelCandidates = [];
1804
- // Get all vendors for this model - filter for inference providers only
1805
- const modelVendors = AIEngine.Instance.ModelVendors
1806
- .filter(mv => UUIDsEqual(mv.ModelID, model.ID) && mv.Status === 'Active' && this.isInferenceProvider(mv))
1816
+ // Get all vendors for this model - filter for inference providers only.
1817
+ // Uses the model's precomputed ModelVendors (grouped at engine load) rather than scanning
1818
+ // the global ModelVendors array.
1819
+ const modelVendors = model.ModelVendors
1820
+ .filter(mv => mv.Status === 'Active' && this.isInferenceProvider(mv))
1807
1821
  .sort((a, b) => b.Priority - a.Priority);
1808
1822
  // First, add preferred vendor if it exists
1809
1823
  if (preferredVendorId) {
@@ -1877,8 +1891,7 @@ export class AIPromptRunner {
1877
1891
  return validModels.map(considered => {
1878
1892
  // Find matching model vendor for driver and API info
1879
1893
  const modelVendor = considered.vendor
1880
- ? AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.ModelID, considered.model.ID) &&
1881
- UUIDsEqual(mv.VendorID, considered.vendor.ID))
1894
+ ? considered.model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, considered.vendor.ID))
1882
1895
  : undefined;
1883
1896
  return {
1884
1897
  model: considered.model,
@@ -1908,8 +1921,33 @@ export class AIPromptRunner {
1908
1921
  // Key format: "driverClass:modelId:vendorId" to properly cache credential hierarchy
1909
1922
  const credentialCache = new Map();
1910
1923
  const consideredModels = [];
1911
- // Check ALL candidates to build complete list of valid and invalid options
1924
+ // DECISION (performance): candidates are ordered by priority, and we only need the
1925
+ // highest-priority candidate that has working credentials. So once we find that first
1926
+ // hit, we STOP credential-probing the remaining candidates and record them as
1927
+ // "not-evaluated" rather than running a `hasCredentialsAvailable` check (which does
1928
+ // env-var lookups + binding scans) for every configured model on every prompt run.
1929
+ // The remaining candidates are still kept in `consideredModels` (and in the returned
1930
+ // `allCandidates` from selectModel, which is the FULL ordered list) so failover and the
1931
+ // ordering are unaffected — only the per-candidate availability *telemetry* for the tail
1932
+ // is skipped. Callers that need a complete availability report (e.g. an admin diagnostic)
1933
+ // can set `AIPromptParams.forceFullModelEvaluation = true` to probe every candidate.
1934
+ const forceFullEval = params?.forceFullModelEvaluation === true;
1935
+ let selected;
1912
1936
  for (const candidate of candidates) {
1937
+ const vendorEntity = candidate.vendorId
1938
+ ? AIEngine.Instance.VendorsByID.get(NormalizeUUID(candidate.vendorId))
1939
+ : undefined;
1940
+ // Short-circuit: a usable candidate is already selected and full evaluation wasn't requested.
1941
+ if (selected && !forceFullEval) {
1942
+ consideredModels.push({
1943
+ model: candidate.model,
1944
+ vendor: vendorEntity,
1945
+ priority: candidate.priority,
1946
+ available: false,
1947
+ unavailableReason: AIPromptRunner.NOT_EVALUATED_REASON
1948
+ });
1949
+ continue;
1950
+ }
1913
1951
  // Build cache key including model and vendor for proper credential resolution
1914
1952
  const cacheKey = `${candidate.driverClass}:${candidate.model.ID}:${candidate.vendorId || 'default'}`;
1915
1953
  // Check cache first
@@ -1922,22 +1960,20 @@ export class AIPromptRunner {
1922
1960
  hasCredentials = this.hasCredentialsAvailable(candidate.driverClass, promptId, candidate.model.ID, candidate.vendorId, params);
1923
1961
  credentialCache.set(cacheKey, hasCredentials);
1924
1962
  }
1925
- // Get vendor entity from AIEngine cache if vendorId is available
1926
- let vendorEntity;
1927
- if (candidate.vendorId) {
1928
- vendorEntity = AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, candidate.vendorId));
1929
- }
1930
1963
  // Track this model as considered with availability status
1931
- consideredModels.push({
1964
+ const considered = {
1932
1965
  model: candidate.model,
1933
1966
  vendor: vendorEntity,
1934
1967
  priority: candidate.priority,
1935
1968
  available: hasCredentials,
1936
1969
  unavailableReason: hasCredentials ? undefined : `No credentials configured for driver ${candidate.driverClass}`
1937
- });
1970
+ };
1971
+ consideredModels.push(considered);
1972
+ // Record the first available candidate as the selection (highest priority with credentials)
1973
+ if (hasCredentials && !selected) {
1974
+ selected = considered;
1975
+ }
1938
1976
  }
1939
- // Select the first available candidate (highest priority with API key)
1940
- const selected = consideredModels.find(m => m.available);
1941
1977
  const selectedCandidate = selected ? candidates.find(c => UUIDsEqual(c.model.ID, selected.model.ID) &&
1942
1978
  UUIDsEqual(c.vendorId, selected.vendor?.ID)) : null;
1943
1979
  if (selectedCandidate) {
@@ -1990,6 +2026,75 @@ export class AIPromptRunner {
1990
2026
  /**
1991
2027
  * Creates an AIPromptRun entity for execution tracking
1992
2028
  */
2029
+ /**
2030
+ * Resolves the scalar inference parameters for a run: each value is the per-request override
2031
+ * from `additionalParameters` when supplied, otherwise the prompt's configured default. This
2032
+ * is the single source of truth for parameter precedence so {@link executeModel} (ChatParams)
2033
+ * and {@link createPromptRun} (the persisted record) stay in lockstep. Stop sequences and
2034
+ * assistant prefill are intentionally excluded — their representations differ per target.
2035
+ */
2036
+ resolveScalarInferenceParams(prompt, additionalParameters) {
2037
+ const pick = (override, promptDefault) => override !== undefined ? override : (promptDefault != null ? promptDefault : undefined);
2038
+ const ap = additionalParameters;
2039
+ return {
2040
+ temperature: pick(ap?.temperature, prompt.Temperature),
2041
+ topP: pick(ap?.topP, prompt.TopP),
2042
+ topK: pick(ap?.topK, prompt.TopK),
2043
+ minP: pick(ap?.minP, prompt.MinP),
2044
+ frequencyPenalty: pick(ap?.frequencyPenalty, prompt.FrequencyPenalty),
2045
+ presencePenalty: pick(ap?.presencePenalty, prompt.PresencePenalty),
2046
+ seed: pick(ap?.seed, prompt.Seed),
2047
+ includeLogProbs: pick(ap?.includeLogProbs, prompt.IncludeLogProbs),
2048
+ topLogProbs: pick(ap?.topLogProbs, prompt.TopLogProbs),
2049
+ };
2050
+ }
2051
+ /**
2052
+ * Queues a fire-and-forget `Save()` for a prompt-run entity. Saves for the same instance are
2053
+ * chained (via {@link _promptRunSaveChains}) so the initial INSERT always completes before any
2054
+ * finalize UPDATE — guaranteeing a slow INSERT can't overwrite the finalized row. The whole
2055
+ * chain runs independently of the execution flow (callers do NOT await it), so the model call
2056
+ * is never delayed by a DB write.
2057
+ *
2058
+ * Save failures are logged (non-fatal): the AIPromptRun record is observability, not part of the
2059
+ * prompt's success contract, so a rare persistence failure must not fail the prompt. The chained
2060
+ * promise is tracked in {@link _pendingPromptRunSaves} so {@link WaitForPendingPromptRunSaves}
2061
+ * can flush them when determinism is required (e.g. tests, or a caller that needs the rows
2062
+ * durably written). Returns that promise.
2063
+ */
2064
+ queuePromptRunSave(promptRun) {
2065
+ const previous = this._promptRunSaveChains.get(promptRun) ?? Promise.resolve(true);
2066
+ const current = previous
2067
+ .then(async () => {
2068
+ const ok = await promptRun.Save();
2069
+ if (!ok) {
2070
+ this.logError(`Failed to save AIPromptRun ${promptRun.ID || '(unsaved)'}: ${promptRun.LatestResult?.CompleteMessage || 'Unknown error'}`, {
2071
+ category: 'PromptRunSave',
2072
+ metadata: { promptRunId: promptRun.ID }
2073
+ });
2074
+ }
2075
+ return ok;
2076
+ })
2077
+ .catch((err) => {
2078
+ // Infrastructure-level throw (network, etc.) — log and swallow so the fire-and-forget
2079
+ // promise never surfaces as an unhandled rejection.
2080
+ this.logError(err instanceof Error ? err : new Error(String(err)), {
2081
+ category: 'PromptRunSave',
2082
+ metadata: { promptRunId: promptRun.ID }
2083
+ });
2084
+ return false;
2085
+ });
2086
+ this._promptRunSaveChains.set(promptRun, current);
2087
+ this._pendingPromptRunSaves.push(current);
2088
+ return current;
2089
+ }
2090
+ /**
2091
+ * Awaits all in-flight prompt-run saves queued by this runner instance. The normal execution
2092
+ * path does NOT call this — prompt-run persistence is intentionally fire-and-forget. Exposed for
2093
+ * tests and for callers that need the AIPromptRun rows durably written before proceeding.
2094
+ */
2095
+ async WaitForPendingPromptRunSaves() {
2096
+ await Promise.allSettled(this._pendingPromptRunSaves);
2097
+ }
1993
2098
  async createPromptRun(prompt, model, params, systemPromptText, startTime, vendorId, modelSelectionInfo) {
1994
2099
  const provider = params.provider ?? Metadata.Provider;
1995
2100
  const promptRun = await provider.GetEntityObject('MJ: AI Prompt Runs', params.contextUser);
@@ -2058,8 +2163,8 @@ export class AIPromptRunner {
2058
2163
  }
2059
2164
  else {
2060
2165
  // Fallback: grab the highest priority AI Model Vendor record for this model (inference providers only)
2061
- const modelVendors = AIEngine.Instance.ModelVendors
2062
- .filter((mv) => UUIDsEqual(mv.ModelID, model.ID) && mv.Status === 'Active' && this.isInferenceProvider(mv))
2166
+ const modelVendors = model.ModelVendors
2167
+ .filter((mv) => mv.Status === 'Active' && this.isInferenceProvider(mv))
2063
2168
  .sort((a, b) => b.Priority - a.Priority);
2064
2169
  if (modelVendors.length > 0) {
2065
2170
  promptRun.VendorID = modelVendors[0].VendorID;
@@ -2091,62 +2196,35 @@ export class AIPromptRunner {
2091
2196
  if (prompt.ResponseFormat && prompt.ResponseFormat !== 'Any') {
2092
2197
  promptRun.ResponseFormat = prompt.ResponseFormat;
2093
2198
  }
2094
- // Save the actual values that will be used (either from prompt defaults or additionalParameters)
2095
- // First, apply defaults from prompt entity
2096
- if (prompt.Temperature != null)
2097
- promptRun.Temperature = prompt.Temperature;
2098
- if (prompt.TopP != null)
2099
- promptRun.TopP = prompt.TopP;
2100
- if (prompt.TopK != null)
2101
- promptRun.TopK = prompt.TopK;
2102
- if (prompt.MinP != null)
2103
- promptRun.MinP = prompt.MinP;
2104
- if (prompt.FrequencyPenalty != null)
2105
- promptRun.FrequencyPenalty = prompt.FrequencyPenalty;
2106
- if (prompt.PresencePenalty != null)
2107
- promptRun.PresencePenalty = prompt.PresencePenalty;
2108
- if (prompt.Seed != null)
2109
- promptRun.Seed = prompt.Seed;
2199
+ // Save the actual values that will be used (prompt defaults overridden by additionalParameters).
2200
+ // Uses the shared resolver so the persisted record matches what executeModel sends to the model.
2201
+ const resolvedParams = this.resolveScalarInferenceParams(prompt, params.additionalParameters);
2202
+ if (resolvedParams.temperature !== undefined)
2203
+ promptRun.Temperature = resolvedParams.temperature;
2204
+ if (resolvedParams.topP !== undefined)
2205
+ promptRun.TopP = resolvedParams.topP;
2206
+ if (resolvedParams.topK !== undefined)
2207
+ promptRun.TopK = resolvedParams.topK;
2208
+ if (resolvedParams.minP !== undefined)
2209
+ promptRun.MinP = resolvedParams.minP;
2210
+ if (resolvedParams.frequencyPenalty !== undefined)
2211
+ promptRun.FrequencyPenalty = resolvedParams.frequencyPenalty;
2212
+ if (resolvedParams.presencePenalty !== undefined)
2213
+ promptRun.PresencePenalty = resolvedParams.presencePenalty;
2214
+ if (resolvedParams.seed !== undefined)
2215
+ promptRun.Seed = resolvedParams.seed;
2216
+ if (resolvedParams.includeLogProbs !== undefined)
2217
+ promptRun.LogProbs = resolvedParams.includeLogProbs;
2218
+ if (resolvedParams.topLogProbs !== undefined)
2219
+ promptRun.TopLogProbs = resolvedParams.topLogProbs;
2220
+ // Stop sequences + assistant prefill: stored from the prompt, with the additionalParameters
2221
+ // array (JSON-encoded) taking precedence when supplied.
2110
2222
  if (prompt.StopSequences)
2111
2223
  promptRun.StopSequences = prompt.StopSequences;
2112
2224
  if (prompt.AssistantPrefill)
2113
2225
  promptRun.AssistantPrefill = prompt.AssistantPrefill;
2114
- if (prompt.IncludeLogProbs != null)
2115
- promptRun.LogProbs = prompt.IncludeLogProbs;
2116
- if (prompt.TopLogProbs != null)
2117
- promptRun.TopLogProbs = prompt.TopLogProbs;
2118
- // Then override with additionalParameters if provided
2119
- if (params.additionalParameters) {
2120
- if (params.additionalParameters.temperature !== undefined) {
2121
- promptRun.Temperature = params.additionalParameters.temperature;
2122
- }
2123
- if (params.additionalParameters.topP !== undefined) {
2124
- promptRun.TopP = params.additionalParameters.topP;
2125
- }
2126
- if (params.additionalParameters.topK !== undefined) {
2127
- promptRun.TopK = params.additionalParameters.topK;
2128
- }
2129
- if (params.additionalParameters.minP !== undefined) {
2130
- promptRun.MinP = params.additionalParameters.minP;
2131
- }
2132
- if (params.additionalParameters.frequencyPenalty !== undefined) {
2133
- promptRun.FrequencyPenalty = params.additionalParameters.frequencyPenalty;
2134
- }
2135
- if (params.additionalParameters.presencePenalty !== undefined) {
2136
- promptRun.PresencePenalty = params.additionalParameters.presencePenalty;
2137
- }
2138
- if (params.additionalParameters.seed !== undefined) {
2139
- promptRun.Seed = params.additionalParameters.seed;
2140
- }
2141
- if (params.additionalParameters.stopSequences !== undefined && params.additionalParameters.stopSequences.length > 0) {
2142
- promptRun.StopSequences = JSON.stringify(params.additionalParameters.stopSequences);
2143
- }
2144
- if (params.additionalParameters.includeLogProbs !== undefined) {
2145
- promptRun.LogProbs = params.additionalParameters.includeLogProbs;
2146
- }
2147
- if (params.additionalParameters.topLogProbs !== undefined) {
2148
- promptRun.TopLogProbs = params.additionalParameters.topLogProbs;
2149
- }
2226
+ if (params.additionalParameters?.stopSequences !== undefined && params.additionalParameters.stopSequences.length > 0) {
2227
+ promptRun.StopSequences = JSON.stringify(params.additionalParameters.stopSequences);
2150
2228
  }
2151
2229
  // Store the input data/context as JSON in Messages field
2152
2230
  if (params.data || params.templateData || systemPromptText) {
@@ -2179,21 +2257,13 @@ export class AIPromptRunner {
2179
2257
  promptRun.ValidationAttemptCount = 0; // Will be updated during execution
2180
2258
  promptRun.SuccessfulValidationCount = 0;
2181
2259
  promptRun.FinalValidationPassed = false; // Will be updated after execution
2182
- const saveResult = await promptRun.Save();
2183
- if (!saveResult) {
2184
- const error = `Failed to save AIPromptRun: ${promptRun.LatestResult?.CompleteMessage || 'Unknown error'}`;
2185
- this.logError(error, {
2186
- category: 'PromptRunCreation',
2187
- metadata: {
2188
- promptId: prompt.ID,
2189
- modelId: model.ID,
2190
- vendorId
2191
- },
2192
- maxErrorLength: params.maxErrorLength
2193
- });
2194
- throw new Error(error);
2195
- }
2196
- // Invoke callback if provided
2260
+ // Persist the initial 'Running' record fire-and-forget. The ID was already assigned by
2261
+ // NewRecord() above, so callers (and the onPromptRunCreated callback) have it immediately —
2262
+ // we don't block the model call on the INSERT. The finalize UPDATE chains after this INSERT
2263
+ // via the instance-keyed save queue, so ordering is guaranteed.
2264
+ this.queuePromptRunSave(promptRun);
2265
+ // Invoke callback if provided. The ID is available without awaiting the save (client-generated
2266
+ // by NewRecord()), so agent-run/step linking that depends on it works immediately.
2197
2267
  if (params.onPromptRunCreated) {
2198
2268
  try {
2199
2269
  await params.onPromptRunCreated(promptRun.ID);
@@ -2530,7 +2600,7 @@ export class AIPromptRunner {
2530
2600
  supportsEffortLevel = model.SupportsEffortLevel ?? false;
2531
2601
  if (vendorId) {
2532
2602
  // Find the AIModelVendor record for this specific vendor - must be an inference provider
2533
- const modelVendor = AIEngine.Instance.ModelVendors.find((mv) => UUIDsEqual(mv.ModelID, model.ID) && UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active' && this.isInferenceProvider(mv));
2603
+ const modelVendor = model.ModelVendors.find((mv) => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active' && this.isInferenceProvider(mv));
2534
2604
  if (modelVendor) {
2535
2605
  driverClass = modelVendor.DriverClass || driverClass;
2536
2606
  apiName = modelVendor.APIName || apiName;
@@ -2554,63 +2624,34 @@ export class AIPromptRunner {
2554
2624
  }
2555
2625
  chatParams.model = apiName;
2556
2626
  chatParams.cancellationToken = cancellationToken;
2557
- // Apply defaults from prompt entity first (if they exist)
2558
- // These can be overridden by additionalParameters
2559
- if (prompt.Temperature != null)
2560
- chatParams.temperature = prompt.Temperature;
2561
- if (prompt.TopP != null)
2562
- chatParams.topP = prompt.TopP;
2563
- if (prompt.TopK != null)
2564
- chatParams.topK = prompt.TopK;
2565
- if (prompt.MinP != null)
2566
- chatParams.minP = prompt.MinP;
2567
- if (prompt.FrequencyPenalty != null)
2568
- chatParams.frequencyPenalty = prompt.FrequencyPenalty;
2569
- if (prompt.PresencePenalty != null)
2570
- chatParams.presencePenalty = prompt.PresencePenalty;
2571
- if (prompt.Seed != null)
2572
- chatParams.seed = prompt.Seed;
2627
+ // Apply scalar inference params (prompt defaults overridden by additionalParameters) via the
2628
+ // shared resolver so ChatParams and the persisted AIPromptRun never drift.
2629
+ const resolvedParams = this.resolveScalarInferenceParams(prompt, params.additionalParameters);
2630
+ if (resolvedParams.temperature !== undefined)
2631
+ chatParams.temperature = resolvedParams.temperature;
2632
+ if (resolvedParams.topP !== undefined)
2633
+ chatParams.topP = resolvedParams.topP;
2634
+ if (resolvedParams.topK !== undefined)
2635
+ chatParams.topK = resolvedParams.topK;
2636
+ if (resolvedParams.minP !== undefined)
2637
+ chatParams.minP = resolvedParams.minP;
2638
+ if (resolvedParams.frequencyPenalty !== undefined)
2639
+ chatParams.frequencyPenalty = resolvedParams.frequencyPenalty;
2640
+ if (resolvedParams.presencePenalty !== undefined)
2641
+ chatParams.presencePenalty = resolvedParams.presencePenalty;
2642
+ if (resolvedParams.seed !== undefined)
2643
+ chatParams.seed = resolvedParams.seed;
2644
+ if (resolvedParams.includeLogProbs !== undefined)
2645
+ chatParams.includeLogProbs = resolvedParams.includeLogProbs;
2646
+ if (resolvedParams.topLogProbs !== undefined)
2647
+ chatParams.topLogProbs = resolvedParams.topLogProbs;
2648
+ // Stop sequences are handled separately: the prompt value is comma-delimited and gated by
2649
+ // driver support; additionalParameters supplies a ready-made array that overrides it.
2573
2650
  if (prompt.StopSequences && this.shouldApplyStopSequences(prompt, model, vendorId, llm)) {
2574
- // Parse comma-delimited stop sequences
2575
2651
  chatParams.stopSequences = prompt.StopSequences.split(',').map((s) => s.replace(AIPromptRunner.STOP_SEQUENCE_TRIM_REGEX, '')).filter((s) => s.length > 0);
2576
2652
  }
2577
- if (prompt.IncludeLogProbs != null)
2578
- chatParams.includeLogProbs = prompt.IncludeLogProbs;
2579
- if (prompt.TopLogProbs != null)
2580
- chatParams.topLogProbs = prompt.TopLogProbs;
2581
- // Apply additional parameters if provided (these override prompt defaults)
2582
- if (params.additionalParameters) {
2583
- // Apply chat-specific parameters from additionalParameters
2584
- if (params.additionalParameters.temperature !== undefined) {
2585
- chatParams.temperature = params.additionalParameters.temperature;
2586
- }
2587
- if (params.additionalParameters.topP !== undefined) {
2588
- chatParams.topP = params.additionalParameters.topP;
2589
- }
2590
- if (params.additionalParameters.topK !== undefined) {
2591
- chatParams.topK = params.additionalParameters.topK;
2592
- }
2593
- if (params.additionalParameters.minP !== undefined) {
2594
- chatParams.minP = params.additionalParameters.minP;
2595
- }
2596
- if (params.additionalParameters.frequencyPenalty !== undefined) {
2597
- chatParams.frequencyPenalty = params.additionalParameters.frequencyPenalty;
2598
- }
2599
- if (params.additionalParameters.presencePenalty !== undefined) {
2600
- chatParams.presencePenalty = params.additionalParameters.presencePenalty;
2601
- }
2602
- if (params.additionalParameters.seed !== undefined) {
2603
- chatParams.seed = params.additionalParameters.seed;
2604
- }
2605
- if (params.additionalParameters.stopSequences !== undefined) {
2606
- chatParams.stopSequences = params.additionalParameters.stopSequences;
2607
- }
2608
- if (params.additionalParameters.includeLogProbs !== undefined) {
2609
- chatParams.includeLogProbs = params.additionalParameters.includeLogProbs;
2610
- }
2611
- if (params.additionalParameters.topLogProbs !== undefined) {
2612
- chatParams.topLogProbs = params.additionalParameters.topLogProbs;
2613
- }
2653
+ if (params.additionalParameters?.stopSequences !== undefined) {
2654
+ chatParams.stopSequences = params.additionalParameters.stopSequences;
2614
2655
  }
2615
2656
  // Apply effortLevel with precedence hierarchy
2616
2657
  // 1. params.effortLevel (runtime override - highest priority)
@@ -2819,8 +2860,13 @@ export class AIPromptRunner {
2819
2860
  const lower = mimeType.toLowerCase();
2820
2861
  return caps.SupportedMimeTypes.some((pattern) => {
2821
2862
  const p = pattern.toLowerCase();
2863
+ // Wildcard on EITHER side must match (the requested mime is often a modality
2864
+ // probe like 'image/*' — e.g. an image_url block with no explicit mimeType —
2865
+ // and must match a driver that declares any concrete 'image/<x>' type).
2822
2866
  if (p.endsWith('/*'))
2823
2867
  return lower.startsWith(p.slice(0, -1));
2868
+ if (lower.endsWith('/*'))
2869
+ return p.startsWith(lower.slice(0, -1));
2824
2870
  return lower === p;
2825
2871
  });
2826
2872
  }
@@ -2976,7 +3022,7 @@ export class AIPromptRunner {
2976
3022
  }
2977
3023
  // Vendor-level override (null = inherit)
2978
3024
  if (vendorId) {
2979
- const modelVendor = AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.ModelID, model.ID) && UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
3025
+ const modelVendor = model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
2980
3026
  if (modelVendor?.SupportsPrefill != null) {
2981
3027
  supportsPrefill = modelVendor.SupportsPrefill;
2982
3028
  }
@@ -2990,7 +3036,7 @@ export class AIPromptRunner {
2990
3036
  */
2991
3037
  resolvePrefillFallbackText(model, vendorId) {
2992
3038
  // Start with model type default
2993
- const modelType = AIEngine.Instance.ModelTypes.find(mt => UUIDsEqual(mt.ID, model.AIModelTypeID));
3039
+ const modelType = AIEngine.Instance.ModelTypesByID.get(NormalizeUUID(model.AIModelTypeID));
2994
3040
  let fallbackText = modelType?.PrefillFallbackText ?? null;
2995
3041
  // Model-level override
2996
3042
  if (model.PrefillFallbackText != null) {
@@ -2998,7 +3044,7 @@ export class AIPromptRunner {
2998
3044
  }
2999
3045
  // Vendor-level override
3000
3046
  if (vendorId) {
3001
- const modelVendor = AIEngine.Instance.ModelVendors.find(mv => UUIDsEqual(mv.ModelID, model.ID) && UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
3047
+ const modelVendor = model.ModelVendors.find(mv => UUIDsEqual(mv.VendorID, vendorId) && mv.Status === 'Active');
3002
3048
  if (modelVendor?.PrefillFallbackText != null) {
3003
3049
  fallbackText = modelVendor.PrefillFallbackText;
3004
3050
  }
@@ -3230,7 +3276,7 @@ export class AIPromptRunner {
3230
3276
  const filteredCandidates = allCandidates.filter(c => (c.vendorId || 'default') !== failedVendorId);
3231
3277
  const removedCount = beforeCount - filteredCandidates.length;
3232
3278
  if (removedCount > 0) {
3233
- const vendorName = AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, failedVendorId))?.Name || failedVendorId;
3279
+ const vendorName = AIEngine.Instance.VendorsByID.get(NormalizeUUID(failedVendorId))?.Name || failedVendorId;
3234
3280
  const remainingCount = filteredCandidates.length;
3235
3281
  // Log appropriate message based on error type
3236
3282
  let reason;
@@ -3271,7 +3317,7 @@ export class AIPromptRunner {
3271
3317
  if (shouldRetry) {
3272
3318
  const modelName = currentModel.Name;
3273
3319
  const vendorName = currentVendorId
3274
- ? AIEngine.Instance.Vendors.find(v => UUIDsEqual(v.ID, currentVendorId))?.Name || 'default'
3320
+ ? AIEngine.Instance.VendorsByID.get(NormalizeUUID(currentVendorId))?.Name || 'default'
3275
3321
  : 'default';
3276
3322
  this.logStatus(` ⏳ Rate limit hit - retrying ${modelName} (${vendorName}) with backoff (attempt ${rateLimitRetryCount}/${maxRetries})`, true);
3277
3323
  this.logFailoverAttempt(prompt.ID, failoverAttempt, true);
@@ -3746,28 +3792,28 @@ export class AIPromptRunner {
3746
3792
  jsonToParse = CleanJSON(rawOutput);
3747
3793
  }
3748
3794
  catch (cleanError) {
3749
- if (params.verbose) {
3750
- this.logError(cleanError, {
3751
- category: 'JSONCleaningFailed',
3752
- metadata: {
3753
- originalError: originalError.message,
3754
- rawOutput: rawOutput.substring(0, 500)
3755
- },
3756
- maxErrorLength: params.maxErrorLength
3757
- });
3758
- }
3795
+ this.logError(cleanError, {
3796
+ category: 'JSONCleaningFailed',
3797
+ metadata: {
3798
+ originalError: originalError.message,
3799
+ rawOutput: rawOutput.substring(0, 500)
3800
+ },
3801
+ maxErrorLength: params.maxErrorLength
3802
+ });
3759
3803
  }
3760
3804
  const json5Result = JSON5.parse(jsonToParse);
3761
- if (params.verbose) {
3762
- this.logStatus(' ✅ JSON5 successfully parsed the malformed JSON', true, params);
3763
- }
3805
+ this.logStatus(' ✅ JSON5 successfully parsed the malformed JSON', true, params);
3806
+ currentPromptRun._jsonRepairInfo = {
3807
+ repaired: true,
3808
+ method: 'JSON5',
3809
+ originalError: originalError.message,
3810
+ rawOutputPrefix: rawOutput.substring(0, 200)
3811
+ };
3764
3812
  return json5Result;
3765
3813
  }
3766
3814
  catch (json5Error) {
3767
3815
  // Step 2: Use AI to repair the JSON
3768
- if (params.verbose) {
3769
- this.logStatus(' 🤖 JSON5 failed, attempting AI-based JSON repair...', true, params);
3770
- }
3816
+ this.logStatus(' 🤖 JSON5 failed, attempting AI-based JSON repair...', true, params);
3771
3817
  try {
3772
3818
  // Find the "Repair JSON" prompt in the "MJ: System" category
3773
3819
  const repairPrompt = AIEngine.Instance.Prompts.find(p => p.Name.trim().toLowerCase() === 'repair json' && p.Category.trim().toLowerCase() === 'mj: system');
@@ -3797,39 +3843,61 @@ export class AIPromptRunner {
3797
3843
  }
3798
3844
  // if we get here, we successfully repaired the JSON!!!
3799
3845
  this.logStatus(' ✅ AI successfully repaired the JSON', true, params);
3846
+ currentPromptRun._jsonRepairInfo = {
3847
+ repaired: true,
3848
+ method: 'AIRepair',
3849
+ originalError: originalError.message,
3850
+ rawOutputPrefix: rawOutput.substring(0, 200),
3851
+ repairPromptRunId: repairResult.promptRun?.ID
3852
+ };
3800
3853
  return repairedJSON;
3801
3854
  }
3802
3855
  catch (aiRepairError) {
3803
- // Both repair attempts failed
3804
- if (params.verbose) {
3805
- this.logError(aiRepairError, {
3806
- category: 'JSONRepairFailed',
3807
- metadata: {
3808
- originalError: originalError.message,
3809
- json5Error: json5Error.message,
3810
- aiError: aiRepairError.message,
3811
- rawOutput: rawOutput.substring(0, 500)
3812
- },
3813
- maxErrorLength: params.maxErrorLength
3814
- });
3815
- }
3856
+ // Both repair attempts failed — always log, this is unexpected LLM behavior
3857
+ this.logError(aiRepairError, {
3858
+ category: 'JSONRepairFailed',
3859
+ metadata: {
3860
+ originalError: originalError.message,
3861
+ json5Error: json5Error.message,
3862
+ aiError: aiRepairError.message,
3863
+ rawOutput: rawOutput.substring(0, 500)
3864
+ },
3865
+ maxErrorLength: params.maxErrorLength
3866
+ });
3816
3867
  throw new Error(`JSON repair failed after both JSON5 and AI attempts: ${originalError.message}`);
3817
3868
  }
3818
3869
  }
3819
3870
  }
3871
+ /**
3872
+ * Returns the parsed form of a prompt's `OutputExample` JSON, memoized by content.
3873
+ * Parsing happens at most once per distinct example string for the life of the process;
3874
+ * parse failures are cached too (so malformed examples aren't re-parsed every attempt).
3875
+ */
3876
+ getParsedOutputExample(outputExample) {
3877
+ const cached = AIPromptRunner._outputExampleCache.get(outputExample);
3878
+ if (cached) {
3879
+ return cached;
3880
+ }
3881
+ let entry;
3882
+ try {
3883
+ entry = { parsed: JSON.parse(outputExample) };
3884
+ }
3885
+ catch (parseError) {
3886
+ entry = { error: parseError instanceof Error ? parseError.message : String(parseError) };
3887
+ }
3888
+ AIPromptRunner._outputExampleCache.set(outputExample, entry);
3889
+ return entry;
3890
+ }
3820
3891
  /**
3821
3892
  * Validates parsed result against JSON schema derived from OutputExample
3822
3893
  */
3823
3894
  async validateAgainstSchema(parsedResult, outputExample, promptId) {
3824
3895
  const validationErrors = [];
3825
3896
  try {
3826
- // Parse the output example
3827
- let exampleObject;
3828
- try {
3829
- exampleObject = JSON.parse(outputExample);
3830
- }
3831
- catch (parseError) {
3832
- const error = new ValidationErrorInfo('outputExample', `Invalid OutputExample JSON: ${parseError.message}`, outputExample, ValidationErrorType.Failure);
3897
+ // Parse the output example (cached by content — it's a static string reused across runs/retries)
3898
+ const { parsed: exampleObject, error: exampleParseError } = this.getParsedOutputExample(outputExample);
3899
+ if (exampleParseError) {
3900
+ const error = new ValidationErrorInfo('outputExample', `Invalid OutputExample JSON: ${exampleParseError}`, outputExample, ValidationErrorType.Failure);
3833
3901
  validationErrors.push(error);
3834
3902
  return validationErrors;
3835
3903
  }
@@ -4058,7 +4126,8 @@ export class AIPromptRunner {
4058
4126
  type: e.Type,
4059
4127
  value: e.Value
4060
4128
  })) || [],
4061
- validationDecision: this.getValidationDecisionDescription(parsedResult.validationResult?.Success || false, validationAttempts.length, promptRun.ValidationBehavior || 'Warn')
4129
+ validationDecision: this.getValidationDecisionDescription(parsedResult.validationResult?.Success || false, validationAttempts.length, promptRun.ValidationBehavior || 'Warn'),
4130
+ jsonRepairInfo: promptRun._jsonRepairInfo || null
4062
4131
  });
4063
4132
  }
4064
4133
  else {
@@ -4068,6 +4137,12 @@ export class AIPromptRunner {
4068
4137
  promptRun.FinalValidationPassed = parsedResult.validationResult?.Success !== false;
4069
4138
  promptRun.LastAttemptAt = endTime;
4070
4139
  promptRun.TotalRetryDurationMS = 0;
4140
+ // Even without validation, persist JSON repair info if a repair occurred
4141
+ if (promptRun._jsonRepairInfo) {
4142
+ promptRun.ValidationSummary = JSON.stringify({
4143
+ jsonRepairInfo: promptRun._jsonRepairInfo
4144
+ });
4145
+ }
4071
4146
  }
4072
4147
  // Set Success flag based on validation result
4073
4148
  promptRun.Success = modelResult.success && (parsedResult.validationResult?.Success !== false);
@@ -4093,28 +4168,10 @@ export class AIPromptRunner {
4093
4168
  if (promptRun.Cost !== undefined) {
4094
4169
  promptRun.TotalCost = promptRun.Cost;
4095
4170
  }
4096
- const saveResult = await promptRun.Save();
4097
- if (!saveResult) {
4098
- // Safely extract error message using CompleteMessage getter
4099
- let errorMsg = 'Unknown error';
4100
- try {
4101
- if (promptRun.LatestResult?.CompleteMessage) {
4102
- errorMsg = typeof promptRun.LatestResult.CompleteMessage === 'string'
4103
- ? promptRun.LatestResult.CompleteMessage
4104
- : String(promptRun.LatestResult.CompleteMessage);
4105
- }
4106
- }
4107
- catch (msgError) {
4108
- errorMsg = 'Error accessing error message';
4109
- }
4110
- this.logError(`Failed to update AIPromptRun with results: ${errorMsg}`, {
4111
- category: 'PromptRunUpdate',
4112
- metadata: {
4113
- promptRunId: promptRun.ID,
4114
- updateError: errorMsg
4115
- }
4116
- });
4117
- }
4171
+ // Finalize fire-and-forget. Chains after the initial INSERT (same entity instance) so the
4172
+ // 'Running' INSERT can never overwrite this finalized 'Completed'/'Failed' state. The
4173
+ // execution flow does NOT await this — see queuePromptRunSave / WaitForPendingPromptRunSaves.
4174
+ this.queuePromptRunSave(promptRun);
4118
4175
  }
4119
4176
  catch (error) {
4120
4177
  this.logError(error, {