@memberjunction/ai-prompts 6.1.0-edge.6 → 6.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/AIPromptRunner.d.ts +86 -5
- package/dist/AIPromptRunner.d.ts.map +1 -1
- package/dist/AIPromptRunner.js +283 -13
- package/dist/AIPromptRunner.js.map +1 -1
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +4 -0
- package/dist/index.js.map +1 -1
- package/dist/nativeToolCallingGate.d.ts +108 -0
- package/dist/nativeToolCallingGate.d.ts.map +1 -0
- package/dist/nativeToolCallingGate.js +120 -0
- package/dist/nativeToolCallingGate.js.map +1 -0
- package/package.json +12 -12
package/dist/AIPromptRunner.d.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import { ChatResult, ChatMessage } from '@memberjunction/ai';
|
|
1
|
+
import { ChatResult, ChatMessage, AIPromptConfiguration } from '@memberjunction/ai';
|
|
2
|
+
import { NativeToolCallingDecision } from './nativeToolCallingGate.js';
|
|
2
3
|
import { AIModelRunner } from './AIModelRunner.js';
|
|
3
4
|
import { AIPromptRunResult } from '@memberjunction/ai-core-plus';
|
|
4
5
|
import { IMetadataProvider } from '@memberjunction/core';
|
|
@@ -34,14 +35,24 @@ interface ModelVendorCandidate {
|
|
|
34
35
|
apiName?: string;
|
|
35
36
|
supportsEffortLevel?: boolean;
|
|
36
37
|
effortLevel?: number;
|
|
38
|
+
/**
|
|
39
|
+
* The `PromptConfiguration` bag of the `AIPromptModel` row this candidate came from, when it came
|
|
40
|
+
* from one. Threaded like `effortLevel` rather than looked up later, because a candidate sourced
|
|
41
|
+
* from power-rank or model-type has NO prompt-model row and must contribute no override.
|
|
42
|
+
*/
|
|
43
|
+
promptModelConfiguration?: AIPromptConfiguration | null;
|
|
37
44
|
isPreferredVendor: boolean;
|
|
38
45
|
priority: number;
|
|
39
46
|
source: 'explicit' | 'prompt-model' | 'model-type' | 'power-rank' | 'power-match-fallback';
|
|
40
47
|
}
|
|
41
48
|
/**
|
|
42
|
-
* Configuration for failover behavior when primary model fails
|
|
49
|
+
* Configuration for failover behavior when primary model fails.
|
|
50
|
+
*
|
|
51
|
+
* Exported because it is the return type of `AIPromptRunner.getFailoverConfiguration`, a
|
|
52
|
+
* `protected` method documented as an override point — a subclass cannot name its own return type
|
|
53
|
+
* otherwise, which made the documented extension point unusable from outside this package.
|
|
43
54
|
*/
|
|
44
|
-
interface FailoverConfiguration {
|
|
55
|
+
export interface FailoverConfiguration {
|
|
45
56
|
strategy: 'SameModelDifferentVendor' | 'NextBestModel' | 'PowerRank' | 'None';
|
|
46
57
|
maxAttempts: number;
|
|
47
58
|
delaySeconds: number;
|
|
@@ -476,7 +487,7 @@ export declare class AIPromptRunner {
|
|
|
476
487
|
* - updatePromptRunWithFailoverFailure: Records failed failover metadata
|
|
477
488
|
* - createFailoverErrorResult: Creates standardized error response
|
|
478
489
|
*/
|
|
479
|
-
protected executeModelWithFailover(model: MJAIModelEntityExtended, renderedPrompt: string, prompt: MJAIPromptEntityExtended, params: AIPromptParams, vendorId: string | null, conversationMessages?: ChatMessage[], templateMessageRole?: TemplateMessageRole, cancellationToken?: AbortSignal, allCandidates?: ModelVendorCandidate[], promptRun?: MJAIPromptRunEntityExtended, vendorDriverClass?: string, vendorApiName?: string, vendorSupportsEffortLevel?: boolean, modelEffortLevel?: number, credentialAvailability?: Map<string, boolean
|
|
490
|
+
protected executeModelWithFailover(model: MJAIModelEntityExtended, renderedPrompt: string, prompt: MJAIPromptEntityExtended, params: AIPromptParams, vendorId: string | null, conversationMessages?: ChatMessage[], templateMessageRole?: TemplateMessageRole, cancellationToken?: AbortSignal, allCandidates?: ModelVendorCandidate[], promptRun?: MJAIPromptRunEntityExtended, vendorDriverClass?: string, vendorApiName?: string, vendorSupportsEffortLevel?: boolean, modelEffortLevel?: number, credentialAvailability?: Map<string, boolean>, promptModelConfiguration?: AIPromptConfiguration | null): Promise<ChatResult>;
|
|
480
491
|
/**
|
|
481
492
|
* Builds failover candidates for a prompt based on available models and type restrictions
|
|
482
493
|
*/
|
|
@@ -504,7 +515,60 @@ export declare class AIPromptRunner {
|
|
|
504
515
|
* resolution, driver selection, ChatParams construction, prefill, media handling, and streaming
|
|
505
516
|
* all live here ONCE. Do not duplicate this logic elsewhere.
|
|
506
517
|
*/
|
|
507
|
-
|
|
518
|
+
/**
|
|
519
|
+
* Resolves the native tool-calling gate for THIS (model, vendor) and, when it opens, copies the
|
|
520
|
+
* caller's ephemeral tool surface onto the outgoing request.
|
|
521
|
+
*
|
|
522
|
+
* Called per model call rather than once per run: failover can move the run to a different
|
|
523
|
+
* (model, vendor) whose capability differs, and a decision made before failover would be wrong.
|
|
524
|
+
*
|
|
525
|
+
* @param chatParams The request being assembled (mutated in place)
|
|
526
|
+
* @param prompt The prompt being run — supplies the prompt-layer configuration bag
|
|
527
|
+
* @param params The caller's params — supplies the tool declarations, if any
|
|
528
|
+
* @param model The selected model
|
|
529
|
+
* @param vendorId The selected vendor (`MJ: AI Vendors` ID), or null
|
|
530
|
+
* @param promptModelConfiguration The selected candidate's `AIPromptModel` bag, when it came from one
|
|
531
|
+
*/
|
|
532
|
+
/**
|
|
533
|
+
* Resolves the native tool-calling gate WITHOUT touching a request.
|
|
534
|
+
*
|
|
535
|
+
* Split out because the answer is needed twice and must be the same both times: once before the
|
|
536
|
+
* template renders — the loop template drops its action catalog and the `'Actions'` step type
|
|
537
|
+
* when native mode is on (plan §8.4), and it can only do that truthfully if it knows the real
|
|
538
|
+
* decision rather than the caller's intent — and once when the request is assembled.
|
|
539
|
+
*
|
|
540
|
+
* Never throws: the gate is an opt-in enhancement and must not be able to fail a run that would
|
|
541
|
+
* otherwise succeed, so any configuration problem resolves to the path that has always worked.
|
|
542
|
+
*/
|
|
543
|
+
resolveNativeToolCallingDecision(prompt: MJAIPromptEntityExtended, params: AIPromptParams, model: MJAIModelEntityExtended, vendorId: string | null, promptModelConfiguration?: AIPromptConfiguration | null): NativeToolCallingDecision;
|
|
544
|
+
private applyNativeToolCalling;
|
|
545
|
+
/**
|
|
546
|
+
* Whether a failed result failed for a TOOLS-specific reason, and so is worth one retry with the
|
|
547
|
+
* declarations stripped.
|
|
548
|
+
*
|
|
549
|
+
* Deliberately narrow. A rate limit, a context-length error or a network failure has nothing to do
|
|
550
|
+
* with tools, and retrying those here would burn the fallback and mask the real cause from the
|
|
551
|
+
* existing retry/failover logic — which already handles them properly.
|
|
552
|
+
*
|
|
553
|
+
* @param result The result of a native-mode call
|
|
554
|
+
* @returns true when the failure looks tool-related
|
|
555
|
+
*/
|
|
556
|
+
private isToolSpecificFailure;
|
|
557
|
+
/**
|
|
558
|
+
* Retries a native call once on today's exact path, with the tool declarations stripped.
|
|
559
|
+
*
|
|
560
|
+
* The retry reuses the SAME execution bound rather than opening a fresh one, so the two attempts
|
|
561
|
+
* share one timeout budget. That is deliberate: the bound exists to cap how long a single model
|
|
562
|
+
* call may take from the caller's point of view, and a fallback is still that one call. It does
|
|
563
|
+
* mean a native attempt that burned most of the budget leaves the retry little — but the
|
|
564
|
+
* alternative, silently doubling the caller's timeout, is worse.
|
|
565
|
+
*
|
|
566
|
+
* @param reason What the provider said, for the warning — a misconfiguration should be visible
|
|
567
|
+
*/
|
|
568
|
+
private retryWithToolsStripped;
|
|
569
|
+
/** Scans provider prose for a tools marker. Shared by the thrown-error and failed-result paths. */
|
|
570
|
+
private isToolSpecificFailureText;
|
|
571
|
+
protected executeModel(model: MJAIModelEntityExtended, renderedPrompt: string, prompt: MJAIPromptEntityExtended, params: AIPromptParams, vendorId: string | null, conversationMessages?: ChatMessage[], templateMessageRole?: TemplateMessageRole, cancellationToken?: AbortSignal, vendorDriverClass?: string, vendorApiName?: string, vendorSupportsEffortLevel?: boolean, modelEffortLevel?: number, promptModelConfiguration?: AIPromptConfiguration | null): Promise<ChatResult>;
|
|
508
572
|
/**
|
|
509
573
|
* Engine-level default model-call timeout, in milliseconds, applied when the caller supplies no
|
|
510
574
|
* `AIPromptParams.timeoutMS`. `undefined` (the default) means NO implicit bound — a prompt run
|
|
@@ -617,6 +681,18 @@ export declare class AIPromptRunner {
|
|
|
617
681
|
* opening fence instead — producing an empty response for non-native prefill providers.
|
|
618
682
|
*/
|
|
619
683
|
private static readonly STOP_SEQUENCE_TRIM_REGEX;
|
|
684
|
+
/**
|
|
685
|
+
* Substrings that mark a provider failure as TOOLS-specific, so the native call is worth one
|
|
686
|
+
* envelope retry (see {@link AIPromptRunner.isToolSpecificFailure}). Drawn from how the
|
|
687
|
+
* tool-capable providers word a rejected `tools` payload, an unusable tool call, or a turn whose output
|
|
688
|
+
* they discarded for a tool-related reason.
|
|
689
|
+
*
|
|
690
|
+
* Substring matching over provider prose is inherently approximate. It is deliberately biased
|
|
691
|
+
* toward MISSING a tool failure rather than catching an unrelated one: a missed match just means
|
|
692
|
+
* the existing retry/failover logic handles the error as it does today, whereas a false positive
|
|
693
|
+
* would silently strip tools from a run that should have kept them.
|
|
694
|
+
*/
|
|
695
|
+
private static readonly TOOL_FAILURE_MARKERS;
|
|
620
696
|
/**
|
|
621
697
|
* Resolves whether the current model/vendor supports native assistant prefill.
|
|
622
698
|
*
|
|
@@ -739,6 +815,11 @@ export declare class AIPromptRunner {
|
|
|
739
815
|
* @param params - Optional prompt parameters containing additional configuration like attemptJSONRepair
|
|
740
816
|
* @returns Parsed result with optional validation results and errors
|
|
741
817
|
*/
|
|
818
|
+
/**
|
|
819
|
+
* Whether `text` parses as JSON once CleanJSON has stripped fences and prose around it — the test
|
|
820
|
+
* for "is this an envelope or plain prose?" under implicit control flow. Never throws.
|
|
821
|
+
*/
|
|
822
|
+
private parsesAsJSON;
|
|
742
823
|
private parseAndValidateResultEnhanced;
|
|
743
824
|
/**
|
|
744
825
|
* Parses a string output value.
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"AIPromptRunner.d.ts","sourceRoot":"","sources":["../src/AIPromptRunner.ts"],"names":[],"mappings":"AAAA,OAAO,EAAuB,UAAU,EAAmB,WAAW,EAAqE,MAAM,oBAAoB,CAAC;
|
|
1
|
+
{"version":3,"file":"AIPromptRunner.d.ts","sourceRoot":"","sources":["../src/AIPromptRunner.ts"],"names":[],"mappings":"AAAA,OAAO,EAAuB,UAAU,EAAmB,WAAW,EAAqE,qBAAqB,EAAyB,MAAM,oBAAoB,CAAC;AACpN,OAAO,EAA8C,yBAAyB,EAA8E,MAAM,yBAAyB,CAAC;AAC5L,OAAO,EAAE,aAAa,EAAE,MAAM,iBAAiB,CAAC;AAChD,OAAO,EAAqB,iBAAiB,EAAwB,MAAM,8BAA8B,CAAC;AAC1G,OAAO,EAAwG,iBAAiB,EAAE,MAAM,sBAAsB,CAAC;AAG/J,OAAO,EAAE,uBAAuB,EAAE,wBAAwB,EAAE,2BAA2B,EAAE,MAAM,8BAA8B,CAAC;AAM9H,OAAO,EAAyB,KAAK,6BAA6B,EAAE,MAAM,qBAAqB,CAAC;AAIhG,OAAO,EACH,mBAAmB,EAEnB,cAAc,EACjB,MAAM,8BAA8B,CAAC;AAsBtC;;;;;;;GAOG;AACH,MAAM,WAAW,cAAc;IAC3B,kGAAkG;IAClG,MAAM,CAAC,EAAE,WAAW,CAAC;IACrB,6DAA6D;IAC7D,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,4FAA4F;IAC5F,QAAQ,EAAE,MAAM,OAAO,CAAC;IACxB,qFAAqF;IACrF,OAAO,EAAE,MAAM,IAAI,CAAC;CACvB;AAwFD;;GAEG;AACH,UAAU,oBAAoB;IAC5B,KAAK,EAAE,uBAAuB,CAAC;IAC/B,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,WAAW,EAAE,MAAM,CAAC;IACpB,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,mBAAmB,CAAC,EAAE,OAAO,CAAC;IAC9B,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB;;;;OAIG;IACH,wBAAwB,CAAC,EAAE,qBAAqB,GAAG,IAAI,CAAC;IACxD,iBAAiB,EAAE,OAAO,CAAC;IAC3B,QAAQ,EAAE,MAAM,CAAC;IACjB,MAAM,EAAE,UAAU,GAAG,cAAc,GAAG,YAAY,GAAG,YAAY,GAAG,sBAAsB,CAAC;CAC5F;AAGD;;;;;;GAMG;AACH,MAAM,WAAW,qBAAqB;IACpC,QAAQ,EAAE,0BAA0B,GAAG,eAAe,GAAG,WAAW,GAAG,MAAM,CAAC;IAC9E,WAAW,EAAE,MAAM,CAAC;IACpB,YAAY,EAAE,MAAM,CAAC;IACrB,aAAa,CAAC,EAAE,iBAAiB,GAAG,sBAAsB,GAAG,kBAAkB,CAAC;IAChF,UAAU,CAAC,EAAE,KAAK,GAAG,aAAa,GAAG,eAAe,GAAG,kBAAkB,CAAC;CAC3E;AAED;;GAEG;AACH,UAAU,eAAe;IACvB,aAAa,EAAE,MAAM,CAAC;IACtB,OAAO,EAAE,MAAM,CAAC;IAChB,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,KAAK,EAAE,KAAK,CAAC;IACb,SAAS,EAAE,MAAM,CAAC;IAClB,QAAQ,EAAE,MAAM,CAAC;IACjB,SAAS,EAAE,IAAI,CAAC;CACjB;AAqBD,qBAAa,cAAc;IACzB,OAAO,CAAC,SAAS,CAAW;IAC5B,OAAO,CAAC,eAAe,CAAuB;IAC9C,OAAO,CAAC,iBAAiB,CAAmB;IAC5C,OAAO,CAAC,oBAAoB,CAAC,CAAgC;IAC7D,OAAO,CAAC,cAAc,CAAgB;IACtC,OAAO,CAAC,YAAY,CAAgB;IACpC,OAAO,CAAC,SAAS,CAAkC;IAEnD;;;;;;OAMG;IACH,OAAO,CAAC,eAAe,CAEpB;IAEH;;;;;;;OAOG;IACH,OAAO,CAAC,MAAM,CAAC,QAAQ,CAAC,mBAAmB,CAA2D;IAEtG;;;;OAIG;IACH,OAAO,CAAC,MAAM,CAAC,QAAQ,CAAC,oBAAoB,CAAgI;IAE5K;;;;OAIG;IACH,IAAW,QAAQ,IAAI,iBAAiB,CAEvC;IACD,IAAW,QAAQ,CAAC,KAAK,EAAE,iBAAiB,GAAG,IAAI,EAElD;;IAUD,wGAAwG;IACxG,OAAO,CAAC,MAAM,CAAC,QAAQ,CAAC,wBAAwB,CAAkC;IAElF;;;;;;;;;;;OAWG;IACH,SAAS,KAAK,mBAAmB,IAAI,6BAA6B,CAiBjE;IAED;;;OAGG;IACH,IAAW,WAAW,IAAI,aAAa,CAEtC;IAED;;;OAGG;IACH,OAAO,CAAC,aAAa;IAUrB;;;;;OAKG;IACH,SAAS,CAAC,SAAS,CAAC,OAAO,EAAE,MAAM,EAAE,WAAW,GAAE,OAAe,EAAE,MAAM,CAAC,EAAE,cAAc,GAAG,IAAI;IAYjG;;OAEG;IACH,SAAS,CAAC,QAAQ,CAAC,KAAK,EAAE,KAAK,GAAG,MAAM,EAAE,OAAO,CAAC,EAAE;QAClD,QAAQ,CAAC,EAAE,MAAM,CAAC;QAClB,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC;QAC/B,MAAM,CAAC,EAAE,wBAAwB,CAAC;QAClC,KAAK,CAAC,EAAE,uBAAuB,CAAC;QAChC,QAAQ,CAAC,EAAE,SAAS,GAAG,OAAO,GAAG,UAAU,CAAC;QAC5C,cAAc,CAAC,EAAE,MAAM,CAAC;KACzB,GAAG,IAAI;IAmCR;;;;;;;OAOG;IACH,OAAO,CAAC,mBAAmB;IAI3B;;;;;;;;;;;;;;;;;;;;OAoBG;YACW,6BAA6B;IAqE3C;;;OAGG;YACW,iCAAiC;IAiC/C;;OAEG;YACW,oBAAoB;IA4DlC;;;OAGG;YACW,qBAAqB;IA0BnC;;OAEG;IACH,OAAO,CAAC,2BAA2B;IAUnC;;;;;;;;;;;;;;;;;;;;OAoBG;IACH,OAAO,CAAC,uBAAuB;IAuD/B;;;;;;;;;;;;;;;;;;;;;;;;;OAyBG;IACU,aAAa,CAAC,CAAC,GAAG,OAAO,EAAE,MAAM,EAAE,cAAc,GAAG,OAAO,CAAC,iBAAiB,CAAC,CAAC,CAAC,CAAC;IAmO9F;;;;;;;;OAQG;YACW,mBAAmB;IAuIjC;;;;;;;;OAQG;YACW,uBAAuB;IAqRrC;;OAEG;YACW,YAAY;IAkB1B;;;;;;;;OAQG;IACH,OAAO,CAAC,4BAA4B;IAoBpC;;;;;;;OAOG;YACW,0BAA0B;IA8LxC;;;;;;;OAOG;YACW,8BAA8B;IAmE5C;;;;OAIG;YACW,WAAW;IAqLzB;;;;;;;;;;;;;OAaG;IACH,OAAO,CAAC,0BAA0B;IAsBlC;;;OAGG;IACH,OAAO,CAAC,+BAA+B;IAoBvC;;;;;OAKG;IACH,OAAO,CAAC,kCAAkC;IAkD1C;;;;;;;;OAQG;IACH,OAAO,CAAC,oCAAoC;IA6C5C;;;;;OAKG;IACH,OAAO,CAAC,sBAAsB;IAmB9B;;;OAGG;IACH,OAAO,CAAC,oBAAoB;IAY5B;;;OAGG;IACH,OAAO,CAAC,kCAAkC;IAsC1C;;;OAGG;IACH,OAAO,CAAC,iCAAiC;IAoBzC;;;;OAIG;IACH,OAAO,CAAC,mCAAmC;IA2B3C;;;OAGG;IACH,OAAO,CAAC,+BAA+B;IA4BvC;;OAEG;IACH,OAAO,CAAC,gCAAgC;IA8BxC;;OAEG;IACH,OAAO,CAAC,6BAA6B;IA2CrC;;;;OAIG;IACH,OAAO,CAAC,yBAAyB;IAIjC;;;;OAIG;IACH,OAAO,CAAC,+BAA+B;IAiCvC;;OAEG;IACH,OAAO,CAAC,2BAA2B;IAoBnC;;;OAGG;IACH,OAAO,CAAC,kCAAkC;IAiE1C;;OAEG;IACH,OAAO,CAAC,0BAA0B;IAgBlC;;OAEG;IACH,OAAO,CAAC,uBAAuB;IAgB/B;;OAEG;IACH,OAAO,CAAC,uBAAuB;IAe/B;;OAEG;IACH,OAAO,CAAC,qBAAqB;IAqB7B;;OAEG;IACH,OAAO,CAAC,wBAAwB;IAwEhC;;;OAGG;IACH,OAAO,CAAC,mBAAmB;IAoB3B;;;;;;;;;OASG;YACW,4BAA4B;IAiI1C;;;;OAIG;IACH,OAAO,CAAC,wBAAwB;IAwBhC;;OAEG;IACH;;;;;;OAMG;IACH,OAAO,CAAC,4BAA4B;IAoBpC;;;;OAIG;IACU,4BAA4B,IAAI,OAAO,CAAC,IAAI,CAAC;YAI5C,eAAe;IA8N7B;;OAEG;YACW,oBAAoB;IAsDlC;;;;;;;;;;;;;;OAcG;cACa,wBAAwB,CACtC,KAAK,EAAE,uBAAuB,EAC9B,cAAc,EAAE,MAAM,EACtB,MAAM,EAAE,wBAAwB,EAChC,MAAM,EAAE,cAAc,EACtB,QAAQ,EAAE,MAAM,GAAG,IAAI,EACvB,oBAAoB,CAAC,EAAE,WAAW,EAAE,EACpC,mBAAmB,GAAE,mBAA8B,EACnD,iBAAiB,CAAC,EAAE,WAAW,EAC/B,aAAa,CAAC,EAAE,oBAAoB,EAAE,EACtC,SAAS,CAAC,EAAE,2BAA2B,EACvC,iBAAiB,CAAC,EAAE,MAAM,EAC1B,aAAa,CAAC,EAAE,MAAM,EACtB,yBAAyB,CAAC,EAAE,OAAO,EACnC,gBAAgB,CAAC,EAAE,MAAM,EACzB,sBAAsB,CAAC,EAAE,GAAG,CAAC,MAAM,EAAE,OAAO,CAAC,EAC7C,wBAAwB,CAAC,EAAE,qBAAqB,GAAG,IAAI,GACtD,OAAO,CAAC,UAAU,CAAC;IA8LtB;;OAEG;cACa,uBAAuB,CAAC,MAAM,EAAE,wBAAwB,GAAG,OAAO,CAAC,oBAAoB,EAAE,CAAC;IA2B1G;;OAEG;IACH,SAAS,CAAC,0BAA0B,CAAC,MAAM,EAAE,uBAAuB,EAAE,GAAG,oBAAoB,EAAE;IAuC/F;;OAEG;IACH,SAAS,CAAC,kCAAkC,CAC1C,SAAS,EAAE,2BAA2B,EACtC,gBAAgB,EAAE,eAAe,EAAE,EACnC,YAAY,EAAE,uBAAuB,EACrC,eAAe,EAAE,MAAM,GAAG,IAAI,GAC7B,IAAI;IAoBP;;OAEG;IACH,OAAO,CAAC,kCAAkC;IAe1C;;OAEG;IACH,OAAO,CAAC,yBAAyB;IAiCjC;;;;;;OAMG;IACH;;;;;;;;;;;;;OAaG;IACH;;;;;;;;;;OAUG;IACI,gCAAgC,CACrC,MAAM,EAAE,wBAAwB,EAChC,MAAM,EAAE,cAAc,EACtB,KAAK,EAAE,uBAAuB,EAC9B,QAAQ,EAAE,MAAM,GAAG,IAAI,EACvB,wBAAwB,CAAC,EAAE,qBAAqB,GAAG,IAAI,GACtD,yBAAyB;IA+B5B,OAAO,CAAC,sBAAsB;IAkC9B;;;;;;;;;;OAUG;IACH,OAAO,CAAC,qBAAqB;IAW7B;;;;;;;;;;OAUG;YACW,sBAAsB;IA4BpC,mGAAmG;IACnG,OAAO,CAAC,yBAAyB;cAKjB,YAAY,CAC1B,KAAK,EAAE,uBAAuB,EAC9B,cAAc,EAAE,MAAM,EACtB,MAAM,EAAE,wBAAwB,EAChC,MAAM,EAAE,cAAc,EACtB,QAAQ,EAAE,MAAM,GAAG,IAAI,EACvB,oBAAoB,CAAC,EAAE,WAAW,EAAE,EACpC,mBAAmB,GAAE,mBAA8B,EACnD,iBAAiB,CAAC,EAAE,WAAW,EAC/B,iBAAiB,CAAC,EAAE,MAAM,EAC1B,aAAa,CAAC,EAAE,MAAM,EACtB,yBAAyB,CAAC,EAAE,OAAO,EACnC,gBAAgB,CAAC,EAAE,MAAM,EACzB,wBAAwB,CAAC,EAAE,qBAAqB,GAAG,IAAI,GACtD,OAAO,CAAC,UAAU,CAAC;IAyPtB;;;;;;;;OAQG;IACH,SAAS,KAAK,sBAAsB,IAAI,MAAM,GAAG,SAAS,CAEzD;IAED;;;;;;;;;;;;OAYG;IACH,SAAS,CAAC,qBAAqB,CAAC,MAAM,EAAE,cAAc,GAAG,MAAM,GAAG,SAAS;IAK3E;;;;;;;;;;;;OAYG;IACH,SAAS,CAAC,oBAAoB,CAAC,MAAM,EAAE,wBAAwB,EAAE,MAAM,EAAE,cAAc,EAAE,iBAAiB,CAAC,EAAE,WAAW,GAAG,cAAc;IAyCzI;;;;;;;;;;;;;;OAcG;YACW,wBAAwB;IAoBtC;;;OAGG;IACH,OAAO,CAAC,eAAe;IAQvB;;;;;;;;;;;;;;OAcG;IACH,OAAO,CAAC,2BAA2B;IA0DnC;;;;;;OAMG;IACH,OAAO,CAAC,mCAAmC;IA0C3C;;;;OAIG;IACH,OAAO,CAAC,sBAAsB;IAc9B;;;OAGG;IACH,OAAO,CAAC,sBAAsB;IA+C9B,OAAO,CAAC,iBAAiB;IAqCzB;;;OAGG;IACH,OAAO,CAAC,MAAM,CAAC,QAAQ,CAAC,wBAAwB,CAAoJ;IAEpM;;;;;;;;;;OAUG;IACH,OAAO,CAAC,MAAM,CAAC,QAAQ,CAAC,wBAAwB,CAAsB;IAEtE;;;;;;;;;;OAUG;IACH,OAAO,CAAC,MAAM,CAAC,QAAQ,CAAC,oBAAoB,CAe1C;IAEF;;;;;;;;;;;;;;;OAeG;IACH;;;;;OAKG;IACH,OAAO,CAAC,8BAA8B;IAStC;;;;;;;;;;;;;;;;;;OAkBG;IACH,OAAO,CAAC,wBAAwB;IAchC,OAAO,CAAC,sBAAsB;IA0B9B;;;;OAIG;IACH,OAAO,CAAC,0BAA0B;IA0BlC;;;OAGG;IACH,OAAO,CAAC,qBAAqB;IA8C7B;;OAEG;YACW,4BAA4B;IAgM1C;;OAEG;IACH;;;OAGG;IACH,OAAO,CAAC,mBAAmB;YA8Bb,eAAe;IAO7B;;;;;OAKG;IACH,OAAO,CAAC,sBAAsB;IA6C9B;;;OAGG;YACW,oBAAoB;IAmDlC;;;;;OAKG;YACW,oBAAoB;IAsFlC;;;OAGG;YACW,yBAAyB;IAoDvC;;OAEG;IACH,OAAO,CAAC,gCAAgC;IAuBxC;;OAEG;IACH,OAAO,CAAC,yBAAyB;IA2CjC;;OAEG;IACH,OAAO,CAAC,mBAAmB;IAkB3B;;OAEG;IACH,OAAO,CAAC,yBAAyB;IAoBjC;;OAEG;IACH,OAAO,CAAC,sBAAsB;IA+B9B;;;;;;;;;OASG;IACH;;;OAGG;IACH,OAAO,CAAC,YAAY;YASN,8BAA8B;IA2J5C;;;;;OAKG;IACH,OAAO,CAAC,iBAAiB;IAIzB;;;;;;;;OAQG;IACH,OAAO,CAAC,iBAAiB;IAiBzB;;;;;;;;OAQG;IACH,OAAO,CAAC,kBAAkB;IAkB1B;;;;;;;;OAQG;IACH,OAAO,CAAC,eAAe;IAcvB;;;;;;;;;;;OAWG;YACW,iBAAiB;IAsC/B;;;;;;;;;;;;;;;;OAgBG;IACH,OAAO,CAAC,qBAAqB;IAU7B;;;;;;;;;OASG;YACW,iBAAiB;IA2I/B;;;;OAIG;IACH,OAAO,CAAC,sBAAsB;IAe9B;;OAEG;YACW,qBAAqB;IAkGnC;;OAEG;YACW,eAAe;IAuB7B;;;;OAIG;IACH,OAAO,CAAC,6BAA6B;IAyOrC;;;;;;;OAOG;IAIH;;;;;;;;;;OAUG;IACH,SAAS,CAAC,wBAAwB,CAAC,MAAM,EAAE,wBAAwB,GAAG,qBAAqB;IAU3F;;;;;;;;;;;;OAYG;IACH,SAAS,CAAC,qBAAqB,CAC7B,KAAK,EAAE,KAAK,EACZ,MAAM,EAAE,qBAAqB,EAC7B,aAAa,EAAE,MAAM,GACpB,OAAO;IA6BV;;;;;;OAMG;IACH,OAAO,CAAC,iBAAiB;IAczB;;;;;;;;;;;;OAYG;IACH,SAAS,CAAC,sBAAsB,CAC9B,aAAa,EAAE,MAAM,EACrB,gBAAgB,EAAE,MAAM,GACvB,MAAM;IAaT;;;;;;;;;;;;;;;;;;OAkBG;IACH,SAAS,CAAC,wBAAwB,CAChC,YAAY,EAAE,uBAAuB,EACrC,eAAe,EAAE,MAAM,GAAG,SAAS,EACnC,QAAQ,EAAE,qBAAqB,CAAC,UAAU,CAAC,EAC3C,aAAa,EAAE,qBAAqB,CAAC,eAAe,CAAC,EACrD,aAAa,EAAE,oBAAoB,EAAE,EACrC,cAAc,EAAE,eAAe,EAAE,GAChC,oBAAoB,EAAE;IAgIzB;;;;;;;;;;;OAWG;IACH,SAAS,CAAC,kBAAkB,CAC1B,QAAQ,EAAE,MAAM,EAChB,OAAO,EAAE,eAAe,EACxB,SAAS,EAAE,OAAO,GACjB,IAAI;CA6BR"}
|
package/dist/AIPromptRunner.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import { BaseLLM, ChatParams, ChatMessageRole, GetAIAPIKey, ErrorAnalyzer, ResolveFileInputStrategy } from '@memberjunction/ai';
|
|
1
|
+
import { BaseLLM, ChatParams, ChatMessageRole, GetAIAPIKey, ErrorAnalyzer, ResolveFileInputStrategy, EncodeToolTurnsAsText } from '@memberjunction/ai';
|
|
2
|
+
import { GetToolCallingDecision, GetToolCallingMode, RecordToolCallingDecision, RecordToolCallingMode, ResolveNativeToolCalling } from './nativeToolCallingGate.js';
|
|
2
3
|
import { AIModelRunner } from './AIModelRunner.js';
|
|
3
4
|
import { AIModelSelectionInfo } from '@memberjunction/ai-core-plus';
|
|
4
5
|
import { BaseEntitySaveQueue, LogErrorEx, LogStatus, LogStatusEx, IsVerboseLoggingEnabled, Metadata } from '@memberjunction/core';
|
|
@@ -502,6 +503,29 @@ export class AIPromptRunner {
|
|
|
502
503
|
if (!selection.model) {
|
|
503
504
|
throw new Error(this.buildNoModelFoundMessage(modelSelectionPrompt.Name, selection.selectionInfo));
|
|
504
505
|
}
|
|
506
|
+
// Tell the template which path this run is actually taking, BEFORE it renders. The loop
|
|
507
|
+
// template drops its action catalog and the `'Actions'` step type under native mode
|
|
508
|
+
// (plan §8.4), and that is only safe if the flag is the gate's real answer rather than the
|
|
509
|
+
// caller's intent — a prompt that suppressed its catalog while the model received no tools
|
|
510
|
+
// would leave the agent unable to act at all.
|
|
511
|
+
// Caveat: failover re-resolves the gate per attempt, so a failover onto a model without
|
|
512
|
+
// the capability keeps the already-rendered native wording. The request still degrades
|
|
513
|
+
// correctly (no tools are sent); the prompt is merely quieter than it should be.
|
|
514
|
+
//
|
|
515
|
+
// This must resolve from the SAME inputs the request-time call uses (see
|
|
516
|
+
// `applyNativeToolCalling`), or the template can render for the opposite path: the vendor
|
|
517
|
+
// actually selected — not merely the caller's override, which is usually absent — and the
|
|
518
|
+
// selected candidate's AIPromptModel bag. Resolving without those skips the two layers the
|
|
519
|
+
// capability is normally declared on and silently inverts the decision.
|
|
520
|
+
if (params.tools?.length) {
|
|
521
|
+
const nativeDecision = this.resolveNativeToolCallingDecision(prompt, params, selection.model, selection.selectionInfo?.vendorSelected?.ID ?? params.override?.vendorId ?? null, selection.promptModelConfiguration);
|
|
522
|
+
params.data = {
|
|
523
|
+
...(params.data ?? {}),
|
|
524
|
+
_NATIVE_TOOL_CALLING: nativeDecision.useNativeTools,
|
|
525
|
+
// The template renders the implicit-mode section only when this is the gate's REAL answer.
|
|
526
|
+
_NATIVE_CONTROL_FLOW: nativeDecision.controlFlow
|
|
527
|
+
};
|
|
528
|
+
}
|
|
505
529
|
// Check if we have a system prompt override
|
|
506
530
|
if (params.systemPromptOverride) {
|
|
507
531
|
// Use the override instead of rendering child templates and parent template
|
|
@@ -651,6 +675,7 @@ export class AIPromptRunner {
|
|
|
651
675
|
let vendorApiName = existingSelection?.vendorApiName;
|
|
652
676
|
let vendorSupportsEffortLevel = existingSelection?.vendorSupportsEffortLevel;
|
|
653
677
|
let modelEffortLevel = existingSelection?.modelEffortLevel;
|
|
678
|
+
let promptModelConfiguration = existingSelection?.promptModelConfiguration;
|
|
654
679
|
let allCandidates = existingSelection?.allCandidates ?? [];
|
|
655
680
|
// Credential probes already done during selection — reused by failover so it doesn't
|
|
656
681
|
// recompute hasCredentialsAvailable for the prefix it walks before the selected candidate.
|
|
@@ -668,6 +693,7 @@ export class AIPromptRunner {
|
|
|
668
693
|
vendorApiName = modelResult.vendorApiName;
|
|
669
694
|
vendorSupportsEffortLevel = modelResult.vendorSupportsEffortLevel;
|
|
670
695
|
modelEffortLevel = modelResult.modelEffortLevel;
|
|
696
|
+
promptModelConfiguration = modelResult.promptModelConfiguration;
|
|
671
697
|
modelSelectionInfo = modelResult.selectionInfo;
|
|
672
698
|
allCandidates = modelResult.allCandidates || [];
|
|
673
699
|
credentialAvailability = modelResult.credentialAvailability;
|
|
@@ -687,11 +713,14 @@ export class AIPromptRunner {
|
|
|
687
713
|
}
|
|
688
714
|
// Execute with retry logic for validation failures
|
|
689
715
|
const { modelResult, parsedResult, validationAttempts, cumulativeTokens } = await this.executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, // Pass model-specific effort level
|
|
690
|
-
credentialAvailability // Reuse credential probes from selection
|
|
691
|
-
);
|
|
716
|
+
credentialAvailability, // Reuse credential probes from selection
|
|
717
|
+
promptModelConfiguration);
|
|
692
718
|
// Calculate execution metrics
|
|
693
719
|
const endTime = new Date();
|
|
694
720
|
const executionTimeMS = endTime.getTime() - startTime.getTime();
|
|
721
|
+
// Layer 4 instrumentation: attribute this run to the path it actually took, before the update
|
|
722
|
+
// persists it. Left NULL when no model call happened, which is the honest value.
|
|
723
|
+
promptRun.ToolCallingMode = GetToolCallingMode(modelResult) ?? null;
|
|
695
724
|
// Update the prompt run with results including validation attempts and cumulative tokens
|
|
696
725
|
await this.updatePromptRun(promptRun, prompt, modelResult, parsedResult, endTime, executionTimeMS, validationAttempts, cumulativeTokens);
|
|
697
726
|
const chatResult = modelResult;
|
|
@@ -846,6 +875,11 @@ export class AIPromptRunner {
|
|
|
846
875
|
consolidatedPromptRun.ExecutionTimeMS = parallelResult.totalExecutionTimeMS;
|
|
847
876
|
consolidatedPromptRun.Result = selectedResult.rawResult || '';
|
|
848
877
|
consolidatedPromptRun.TokensUsed = parallelResult.totalTokensUsed;
|
|
878
|
+
// Layer 4 instrumentation, same as the single-model path. This path reaches the same
|
|
879
|
+
// `executeModel` and therefore declares tools, so leaving the column NULL here would tell the
|
|
880
|
+
// agent loop a NativeImplicit turn was not implicit — every control-flow call would then be
|
|
881
|
+
// rejected as an undeclared tool and the loop would retry on tools it declared itself.
|
|
882
|
+
consolidatedPromptRun.ToolCallingMode = GetToolCallingMode(selectedResult.modelResult) ?? null;
|
|
849
883
|
// Extract token and cost info from selected result
|
|
850
884
|
const selectedResultUsage = selectedResult.modelResult?.data?.usage;
|
|
851
885
|
if (selectedResultUsage) {
|
|
@@ -1350,6 +1384,7 @@ export class AIPromptRunner {
|
|
|
1350
1384
|
vendorApiName: selected.apiName,
|
|
1351
1385
|
vendorSupportsEffortLevel: selected.supportsEffortLevel,
|
|
1352
1386
|
modelEffortLevel: selected.effortLevel, // Pass through model-specific effort level
|
|
1387
|
+
promptModelConfiguration: selected.promptModelConfiguration,
|
|
1353
1388
|
allCandidates: candidates,
|
|
1354
1389
|
credentialAvailability,
|
|
1355
1390
|
selectionInfo: this.createSelectionInfo({
|
|
@@ -1659,6 +1694,7 @@ export class AIPromptRunner {
|
|
|
1659
1694
|
apiName: modelVendor.APIName || model.APIName,
|
|
1660
1695
|
supportsEffortLevel: modelVendor.SupportsEffortLevel ?? model.SupportsEffortLevel ?? false,
|
|
1661
1696
|
effortLevel: promptModel.EffortLevel ?? undefined, // Model-specific effort level override
|
|
1697
|
+
promptModelConfiguration: promptModel.PromptConfigurationObject,
|
|
1662
1698
|
isPreferredVendor: false,
|
|
1663
1699
|
priority: computedPriority,
|
|
1664
1700
|
source: 'prompt-model'
|
|
@@ -2147,8 +2183,10 @@ export class AIPromptRunner {
|
|
|
2147
2183
|
}
|
|
2148
2184
|
promptRun.ConfigurationID = params.configurationId;
|
|
2149
2185
|
promptRun.RunAt = startTime;
|
|
2150
|
-
// Resolve and save the effort level used (same precedence as ChatParams resolution)
|
|
2151
|
-
|
|
2186
|
+
// Resolve and save the effort level used (same precedence as ChatParams resolution).
|
|
2187
|
+
// EffortLevel is a numeric column with a CHECK (1-100), so a provider-named level such as
|
|
2188
|
+
// 'xhigh' is deliberately not persisted here — it still reaches the driver via ChatParams.
|
|
2189
|
+
if (typeof params.effortLevel === 'number') {
|
|
2152
2190
|
promptRun.EffortLevel = params.effortLevel;
|
|
2153
2191
|
}
|
|
2154
2192
|
else if (prompt.EffortLevel !== undefined && prompt.EffortLevel !== null) {
|
|
@@ -2332,12 +2370,12 @@ export class AIPromptRunner {
|
|
|
2332
2370
|
* - updatePromptRunWithFailoverFailure: Records failed failover metadata
|
|
2333
2371
|
* - createFailoverErrorResult: Creates standardized error response
|
|
2334
2372
|
*/
|
|
2335
|
-
async executeModelWithFailover(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, allCandidates, promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability) {
|
|
2373
|
+
async executeModelWithFailover(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, allCandidates, promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability, promptModelConfiguration) {
|
|
2336
2374
|
// Get failover configuration (used for errorScope filtering)
|
|
2337
2375
|
const failoverConfig = this.getFailoverConfiguration(prompt);
|
|
2338
2376
|
// If no candidates provided or failover disabled, execute normally with first model
|
|
2339
2377
|
if (!allCandidates || allCandidates.length === 0 || failoverConfig.strategy === 'None') {
|
|
2340
|
-
return this.executeModel(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole, cancellationToken, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel);
|
|
2378
|
+
return this.executeModel(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole, cancellationToken, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, promptModelConfiguration);
|
|
2341
2379
|
}
|
|
2342
2380
|
// Track failover attempts
|
|
2343
2381
|
const failoverAttempts = [];
|
|
@@ -2399,7 +2437,7 @@ export class AIPromptRunner {
|
|
|
2399
2437
|
});
|
|
2400
2438
|
}
|
|
2401
2439
|
// Execute the model with this candidate
|
|
2402
|
-
const result = await this.executeModel(candidate.model, renderedPrompt, prompt, params, candidate.vendorId || null, conversationMessages, templateMessageRole, cancellationToken, candidate.driverClass, candidate.apiName, candidate.supportsEffortLevel, candidate.effortLevel);
|
|
2440
|
+
const result = await this.executeModel(candidate.model, renderedPrompt, prompt, params, candidate.vendorId || null, conversationMessages, templateMessageRole, cancellationToken, candidate.driverClass, candidate.apiName, candidate.supportsEffortLevel, candidate.effortLevel, candidate.promptModelConfiguration);
|
|
2403
2441
|
// CRITICAL FIX: Check if result failed but is retriable (network errors, rate limits, etc.)
|
|
2404
2442
|
// Provider drivers (GeminiLLM, OpenAILLM, etc.) catch errors internally and return ChatResult{success: false}
|
|
2405
2443
|
// instead of throwing, so we must check result.success here.
|
|
@@ -2595,7 +2633,131 @@ export class AIPromptRunner {
|
|
|
2595
2633
|
* resolution, driver selection, ChatParams construction, prefill, media handling, and streaming
|
|
2596
2634
|
* all live here ONCE. Do not duplicate this logic elsewhere.
|
|
2597
2635
|
*/
|
|
2598
|
-
|
|
2636
|
+
/**
|
|
2637
|
+
* Resolves the native tool-calling gate for THIS (model, vendor) and, when it opens, copies the
|
|
2638
|
+
* caller's ephemeral tool surface onto the outgoing request.
|
|
2639
|
+
*
|
|
2640
|
+
* Called per model call rather than once per run: failover can move the run to a different
|
|
2641
|
+
* (model, vendor) whose capability differs, and a decision made before failover would be wrong.
|
|
2642
|
+
*
|
|
2643
|
+
* @param chatParams The request being assembled (mutated in place)
|
|
2644
|
+
* @param prompt The prompt being run — supplies the prompt-layer configuration bag
|
|
2645
|
+
* @param params The caller's params — supplies the tool declarations, if any
|
|
2646
|
+
* @param model The selected model
|
|
2647
|
+
* @param vendorId The selected vendor (`MJ: AI Vendors` ID), or null
|
|
2648
|
+
* @param promptModelConfiguration The selected candidate's `AIPromptModel` bag, when it came from one
|
|
2649
|
+
*/
|
|
2650
|
+
/**
|
|
2651
|
+
* Resolves the native tool-calling gate WITHOUT touching a request.
|
|
2652
|
+
*
|
|
2653
|
+
* Split out because the answer is needed twice and must be the same both times: once before the
|
|
2654
|
+
* template renders — the loop template drops its action catalog and the `'Actions'` step type
|
|
2655
|
+
* when native mode is on (plan §8.4), and it can only do that truthfully if it knows the real
|
|
2656
|
+
* decision rather than the caller's intent — and once when the request is assembled.
|
|
2657
|
+
*
|
|
2658
|
+
* Never throws: the gate is an opt-in enhancement and must not be able to fail a run that would
|
|
2659
|
+
* otherwise succeed, so any configuration problem resolves to the path that has always worked.
|
|
2660
|
+
*/
|
|
2661
|
+
resolveNativeToolCallingDecision(prompt, params, model, vendorId, promptModelConfiguration) {
|
|
2662
|
+
try {
|
|
2663
|
+
return ResolveNativeToolCalling({
|
|
2664
|
+
catalogConfiguration: AIEngine.Instance.GetEffectiveModelConfiguration(model.ID, vendorId
|
|
2665
|
+
// Must be the INFERENCE PROVIDER row, not the Model Developer row: most models carry
|
|
2666
|
+
// two AIModelVendor rows for the same VendorID, and ModelVendors has no guaranteed
|
|
2667
|
+
// order. Picking the developer row merges an empty config layer and silently drops any
|
|
2668
|
+
// per-serving-path LLM.* knob (notably the SupportsNativeToolCalling kill switch).
|
|
2669
|
+
? model.ModelVendors?.find(mv => UUIDsEqual(mv.VendorID, vendorId)
|
|
2670
|
+
&& mv.Status === 'Active' && this.isInferenceProvider(mv))?.ID
|
|
2671
|
+
: undefined),
|
|
2672
|
+
promptConfiguration: prompt.PromptConfigurationObject,
|
|
2673
|
+
promptModelConfiguration,
|
|
2674
|
+
// Action tools and control-flow tools are counted separately: under the hybrid the control
|
|
2675
|
+
// tools are stripped, so on their own they must not open the gate (spec §5).
|
|
2676
|
+
toolsProvided: (params.tools ?? []).some((t) => !(params.controlFlowToolNames ?? []).includes(t.name)),
|
|
2677
|
+
controlToolsProvided: (params.tools ?? []).some((t) => (params.controlFlowToolNames ?? []).includes(t.name))
|
|
2678
|
+
});
|
|
2679
|
+
}
|
|
2680
|
+
catch (error) {
|
|
2681
|
+
console.warn(`AIPromptRunner: could not resolve the native tool-calling gate for prompt "${prompt.Name}" ` +
|
|
2682
|
+
`on model "${model.Name}" — defaulting to the envelope path.`, error);
|
|
2683
|
+
return { useNativeTools: false, mode: 'Envelope', controlFlow: 'envelope', toolResults: false };
|
|
2684
|
+
}
|
|
2685
|
+
}
|
|
2686
|
+
applyNativeToolCalling(chatParams, prompt, params, model, vendorId, promptModelConfiguration) {
|
|
2687
|
+
const decision = this.resolveNativeToolCallingDecision(prompt, params, model, vendorId, promptModelConfiguration);
|
|
2688
|
+
if (decision.warning) {
|
|
2689
|
+
console.warn(`AIPromptRunner: ${decision.warning} (prompt "${prompt.Name}", model "${model.Name}"` +
|
|
2690
|
+
`${vendorId ? `, vendor ${vendorId}` : ''})`);
|
|
2691
|
+
}
|
|
2692
|
+
if (decision.useNativeTools) {
|
|
2693
|
+
const control = new Set(params.controlFlowToolNames ?? []);
|
|
2694
|
+
// Under the hybrid the control tools are stripped: that model's control flow is the envelope,
|
|
2695
|
+
// and offering it ask_user or a sub-agent tool would be a protocol it was never told about.
|
|
2696
|
+
chatParams.tools = decision.controlFlow === 'implicit'
|
|
2697
|
+
? params.tools
|
|
2698
|
+
: params.tools?.filter((t) => !control.has(t.name));
|
|
2699
|
+
chatParams.toolChoice = params.toolChoice;
|
|
2700
|
+
chatParams.parallelToolCalls = params.parallelToolCalls;
|
|
2701
|
+
}
|
|
2702
|
+
// Recorded even on the envelope path, so a run is always attributable to a path. The whole
|
|
2703
|
+
// decision travels so the agent loop can read `toolResults` off the result later.
|
|
2704
|
+
RecordToolCallingDecision(chatParams, decision);
|
|
2705
|
+
}
|
|
2706
|
+
/**
|
|
2707
|
+
* Whether a failed result failed for a TOOLS-specific reason, and so is worth one retry with the
|
|
2708
|
+
* declarations stripped.
|
|
2709
|
+
*
|
|
2710
|
+
* Deliberately narrow. A rate limit, a context-length error or a network failure has nothing to do
|
|
2711
|
+
* with tools, and retrying those here would burn the fallback and mask the real cause from the
|
|
2712
|
+
* existing retry/failover logic — which already handles them properly.
|
|
2713
|
+
*
|
|
2714
|
+
* @param result The result of a native-mode call
|
|
2715
|
+
* @returns true when the failure looks tool-related
|
|
2716
|
+
*/
|
|
2717
|
+
isToolSpecificFailure(result) {
|
|
2718
|
+
if (result.success) {
|
|
2719
|
+
return false;
|
|
2720
|
+
}
|
|
2721
|
+
// A cancellation is the caller's decision, never a tool problem.
|
|
2722
|
+
if (result.errorInfo?.canFailover === false && result.errorInfo?.providerErrorCode === 'request_cancelled') {
|
|
2723
|
+
return false;
|
|
2724
|
+
}
|
|
2725
|
+
return this.isToolSpecificFailureText(`${result.errorMessage ?? ''} ${result.statusText ?? ''}`);
|
|
2726
|
+
}
|
|
2727
|
+
/**
|
|
2728
|
+
* Retries a native call once on today's exact path, with the tool declarations stripped.
|
|
2729
|
+
*
|
|
2730
|
+
* The retry reuses the SAME execution bound rather than opening a fresh one, so the two attempts
|
|
2731
|
+
* share one timeout budget. That is deliberate: the bound exists to cap how long a single model
|
|
2732
|
+
* call may take from the caller's point of view, and a fallback is still that one call. It does
|
|
2733
|
+
* mean a native attempt that burned most of the budget leaves the retry little — but the
|
|
2734
|
+
* alternative, silently doubling the caller's timeout, is worse.
|
|
2735
|
+
*
|
|
2736
|
+
* @param reason What the provider said, for the warning — a misconfiguration should be visible
|
|
2737
|
+
*/
|
|
2738
|
+
async retryWithToolsStripped(llm, chatParams, executionBound, model, vendorId, prompt, reason) {
|
|
2739
|
+
console.warn(`AIPromptRunner: native tool calling failed on ${model.Name}${vendorId ? ` (vendor ${vendorId})` : ''} ` +
|
|
2740
|
+
`for prompt "${prompt.Name}" — retrying once on the envelope path with tools stripped. ` +
|
|
2741
|
+
`Provider error: ${reason}`);
|
|
2742
|
+
chatParams.tools = undefined;
|
|
2743
|
+
chatParams.toolChoice = undefined;
|
|
2744
|
+
chatParams.parallelToolCalls = undefined;
|
|
2745
|
+
// A history that already holds native turns — the assistant's call turn, the tool-result turn —
|
|
2746
|
+
// is refused once the declarations are gone (Gemini also polices their order), so the retry
|
|
2747
|
+
// would fail for a second, different reason and the loop would burn its remaining attempts on
|
|
2748
|
+
// the same request. Show the retry what the envelope path has always
|
|
2749
|
+
// shown: the calls' prose, and each result as an "[Action Result]" user message.
|
|
2750
|
+
chatParams.messages = EncodeToolTurnsAsText(chatParams.messages);
|
|
2751
|
+
const fallbackResult = await this.runChatCompletionBounded(llm, chatParams, executionBound);
|
|
2752
|
+
RecordToolCallingMode(fallbackResult, 'NativeFallback');
|
|
2753
|
+
return fallbackResult;
|
|
2754
|
+
}
|
|
2755
|
+
/** Scans provider prose for a tools marker. Shared by the thrown-error and failed-result paths. */
|
|
2756
|
+
isToolSpecificFailureText(text) {
|
|
2757
|
+
const lowered = text.toLowerCase();
|
|
2758
|
+
return AIPromptRunner.TOOL_FAILURE_MARKERS.some(marker => lowered.includes(marker));
|
|
2759
|
+
}
|
|
2760
|
+
async executeModel(model, renderedPrompt, prompt, params, vendorId, conversationMessages, templateMessageRole = 'system', cancellationToken, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, promptModelConfiguration) {
|
|
2599
2761
|
// define these variables here to ensure they're available in the catch block
|
|
2600
2762
|
let driverClass;
|
|
2601
2763
|
let apiName;
|
|
@@ -2732,8 +2894,22 @@ export class AIPromptRunner {
|
|
|
2732
2894
|
// if response format is not set or set to Any (prompt or override), stay silent
|
|
2733
2895
|
chatParams.responseFormat = undefined;
|
|
2734
2896
|
}
|
|
2897
|
+
// Native tool calling (Layer 3). This is the ONLY place metadata decides whether tools go out.
|
|
2898
|
+
// Resolved HERE rather than once per run because failover may land on a different
|
|
2899
|
+
// (model, vendor) that does not support tools.
|
|
2900
|
+
this.applyNativeToolCalling(chatParams, prompt, params, model, vendorId, promptModelConfiguration);
|
|
2735
2901
|
// Build message array with rendered prompt and conversation messages
|
|
2736
2902
|
chatParams.messages = this.buildMessageArray(renderedPrompt, conversationMessages, templateMessageRole);
|
|
2903
|
+
// Declarations and tool turns must travel TOGETHER. The gate above is re-resolved per
|
|
2904
|
+
// failover attempt, so an attempt can legitimately come back envelope on a history that
|
|
2905
|
+
// earlier turns filled with assistant `toolCalls` and `tool` turns — a candidate whose vendor
|
|
2906
|
+
// row lacks the capability, or a catalog change mid-run. Sending those with no `tools` array
|
|
2907
|
+
// is rejected outright by Anthropic and OpenAI (Gemini also polices their order), which would
|
|
2908
|
+
// make failover — the mechanism meant to rescue a failing run — fail for a second, unrelated
|
|
2909
|
+
// reason. Degrade the turns to text, exactly as the tools-stripped retry does.
|
|
2910
|
+
if (!chatParams.tools?.length) {
|
|
2911
|
+
chatParams.messages = EncodeToolTurnsAsText(chatParams.messages);
|
|
2912
|
+
}
|
|
2737
2913
|
// Resolve native file inputs: check each file against the driver's capabilities
|
|
2738
2914
|
// and inject qualifying files as content blocks in the last user message.
|
|
2739
2915
|
this.injectNativeFileInputs(params, llm, chatParams, verbose);
|
|
@@ -2758,7 +2934,38 @@ export class AIPromptRunner {
|
|
|
2758
2934
|
};
|
|
2759
2935
|
}
|
|
2760
2936
|
// Execute the model bounded by the composed abort signal (caller cancellation + prompt TimeoutMS)
|
|
2761
|
-
|
|
2937
|
+
//
|
|
2938
|
+
// Layer 4 fallback, part one: some providers REJECT a tools payload instead of returning a
|
|
2939
|
+
// failed result. The OpenAI SDK raises a 400 as an exception, so the returned-result check
|
|
2940
|
+
// below never sees it and a native run that should degrade hard-fails instead. Observed on
|
|
2941
|
+
// gpt-5.6-luna: `400 Function tools with reasoning_effort are not supported
|
|
2942
|
+
// for gpt-5.6-luna in /v1/chat/completions`. Both shapes get the same one-shot retry.
|
|
2943
|
+
let chatResult;
|
|
2944
|
+
try {
|
|
2945
|
+
chatResult = await this.runChatCompletionBounded(llm, chatParams, executionBound);
|
|
2946
|
+
}
|
|
2947
|
+
catch (error) {
|
|
2948
|
+
if (!chatParams.tools?.length || !this.isToolSpecificFailureText(error instanceof Error ? error.message : String(error ?? ''))) {
|
|
2949
|
+
throw error;
|
|
2950
|
+
}
|
|
2951
|
+
return await this.retryWithToolsStripped(llm, chatParams, executionBound, model, vendorId, prompt, error instanceof Error ? error.message : String(error));
|
|
2952
|
+
}
|
|
2953
|
+
// Carry the gate's WHOLE decision from the request onto the result, which is what flows back up
|
|
2954
|
+
// to the prompt run (mode) and to the agent loop (`toolResults` — the loop answers native
|
|
2955
|
+
// calls as tool turns only when the result says so). Copying the mode alone resets `toolResults`
|
|
2956
|
+
// to false on the fresh result object, which leaves native tool results inert.
|
|
2957
|
+
// A fallback below overwrites the mode with 'NativeFallback'.
|
|
2958
|
+
const gatedDecision = GetToolCallingDecision(chatParams);
|
|
2959
|
+
if (gatedDecision) {
|
|
2960
|
+
RecordToolCallingDecision(chatResult, gatedDecision);
|
|
2961
|
+
}
|
|
2962
|
+
// Layer 4 fallback, part two: a native-mode call that came back as a failed result for a
|
|
2963
|
+
// TOOLS-specific reason retries the same way. Non-tool failures fall through to the existing
|
|
2964
|
+
// retry/failover machinery untouched.
|
|
2965
|
+
if (chatParams.tools?.length && this.isToolSpecificFailure(chatResult)) {
|
|
2966
|
+
return await this.retryWithToolsStripped(llm, chatParams, executionBound, model, vendorId, prompt, chatResult.errorMessage || chatResult.statusText || 'unspecified');
|
|
2967
|
+
}
|
|
2968
|
+
return chatResult;
|
|
2762
2969
|
}
|
|
2763
2970
|
catch (error) {
|
|
2764
2971
|
const errorInfo = ErrorAnalyzer.analyzeError(error, driverClass);
|
|
@@ -3125,6 +3332,33 @@ export class AIPromptRunner {
|
|
|
3125
3332
|
* opening fence instead — producing an empty response for non-native prefill providers.
|
|
3126
3333
|
*/
|
|
3127
3334
|
static { this.STOP_SEQUENCE_TRIM_REGEX = /^[ \t]+|[ \t]+$/g; }
|
|
3335
|
+
/**
|
|
3336
|
+
* Substrings that mark a provider failure as TOOLS-specific, so the native call is worth one
|
|
3337
|
+
* envelope retry (see {@link AIPromptRunner.isToolSpecificFailure}). Drawn from how the
|
|
3338
|
+
* tool-capable providers word a rejected `tools` payload, an unusable tool call, or a turn whose output
|
|
3339
|
+
* they discarded for a tool-related reason.
|
|
3340
|
+
*
|
|
3341
|
+
* Substring matching over provider prose is inherently approximate. It is deliberately biased
|
|
3342
|
+
* toward MISSING a tool failure rather than catching an unrelated one: a missed match just means
|
|
3343
|
+
* the existing retry/failover logic handles the error as it does today, whereas a false positive
|
|
3344
|
+
* would silently strip tools from a run that should have kept them.
|
|
3345
|
+
*/
|
|
3346
|
+
static { this.TOOL_FAILURE_MARKERS = [
|
|
3347
|
+
'tool_use',
|
|
3348
|
+
'tool use',
|
|
3349
|
+
'tool_call',
|
|
3350
|
+
'tool call',
|
|
3351
|
+
'tool_choice',
|
|
3352
|
+
'tool choice',
|
|
3353
|
+
'tools',
|
|
3354
|
+
'function_call',
|
|
3355
|
+
'function call',
|
|
3356
|
+
'function_declaration',
|
|
3357
|
+
'functiondeclarations',
|
|
3358
|
+
'malformed_function_call',
|
|
3359
|
+
'input_schema',
|
|
3360
|
+
'parametersjsonschema'
|
|
3361
|
+
]; }
|
|
3128
3362
|
/**
|
|
3129
3363
|
* Resolves whether the current model/vendor supports native assistant prefill.
|
|
3130
3364
|
*
|
|
@@ -3259,7 +3493,7 @@ export class AIPromptRunner {
|
|
|
3259
3493
|
/**
|
|
3260
3494
|
* Executes the model with retry logic for validation failures
|
|
3261
3495
|
*/
|
|
3262
|
-
async executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability) {
|
|
3496
|
+
async executeWithValidationRetries(selectedModel, renderedPromptText, prompt, params, promptRun, allCandidates, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability, promptModelConfiguration) {
|
|
3263
3497
|
const validationAttempts = [];
|
|
3264
3498
|
const maxRetries = Math.max(0, prompt.MaxRetries || 0);
|
|
3265
3499
|
let lastError = null;
|
|
@@ -3279,8 +3513,8 @@ export class AIPromptRunner {
|
|
|
3279
3513
|
}
|
|
3280
3514
|
// Execute the AI model with failover support
|
|
3281
3515
|
const modelResult = await this.executeModelWithFailover(selectedModel, renderedPromptText, prompt, params, promptRun.VendorID, params.conversationMessages, params.templateMessageRole || 'system', params.cancellationToken, allCandidates, // Pass the candidates from initial selection
|
|
3282
|
-
promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability // Reuse credential probes from selection
|
|
3283
|
-
);
|
|
3516
|
+
promptRun, vendorDriverClass, vendorApiName, vendorSupportsEffortLevel, modelEffortLevel, credentialAvailability, // Reuse credential probes from selection
|
|
3517
|
+
promptModelConfiguration);
|
|
3284
3518
|
// Check for fatal errors - don't attempt validation/retry on these
|
|
3285
3519
|
// Fatal errors (like ContextLengthExceeded when all models exhausted) cannot be resolved by retrying
|
|
3286
3520
|
if (!modelResult.success && modelResult.errorInfo?.severity === 'Fatal') {
|
|
@@ -3718,6 +3952,19 @@ export class AIPromptRunner {
|
|
|
3718
3952
|
* @param params - Optional prompt parameters containing additional configuration like attemptJSONRepair
|
|
3719
3953
|
* @returns Parsed result with optional validation results and errors
|
|
3720
3954
|
*/
|
|
3955
|
+
/**
|
|
3956
|
+
* Whether `text` parses as JSON once CleanJSON has stripped fences and prose around it — the test
|
|
3957
|
+
* for "is this an envelope or plain prose?" under implicit control flow. Never throws.
|
|
3958
|
+
*/
|
|
3959
|
+
parsesAsJSON(text) {
|
|
3960
|
+
try {
|
|
3961
|
+
JSON.parse(CleanJSON(text) ?? text);
|
|
3962
|
+
return true;
|
|
3963
|
+
}
|
|
3964
|
+
catch {
|
|
3965
|
+
return false;
|
|
3966
|
+
}
|
|
3967
|
+
}
|
|
3721
3968
|
async parseAndValidateResultEnhanced(modelResult, prompt, skipValidation = false, cleanValidationSyntax = false, currentPromptRun, params) {
|
|
3722
3969
|
const validationErrors = [];
|
|
3723
3970
|
let rawOutput;
|
|
@@ -3727,8 +3974,31 @@ export class AIPromptRunner {
|
|
|
3727
3974
|
}
|
|
3728
3975
|
rawOutput = modelResult.data?.choices?.[0]?.message?.content;
|
|
3729
3976
|
if (!rawOutput) {
|
|
3977
|
+
// A native tool call IS the model's answer. Providers return `content: null` on a turn
|
|
3978
|
+
// that is nothing but calls, so reading emptiness as "no output" turns the designed
|
|
3979
|
+
// native response into a validation failure — a warning on every native turn under
|
|
3980
|
+
// `ValidationBehavior: 'Warn'`, and under `'Strict'` a retry that cannot ever succeed,
|
|
3981
|
+
// because retrying asks the same question of a model that already answered it correctly.
|
|
3982
|
+
//
|
|
3983
|
+
// There is also nothing here to parse: `OutputType` describes the shape of TEXT output,
|
|
3984
|
+
// and tool-call arguments arrive already structured and already schema-checked by the
|
|
3985
|
+
// provider. The callers that care read them off `chatResult` directly — the agent loop
|
|
3986
|
+
// via `LoopAgentType`, the eval harness via its own turn extractor.
|
|
3987
|
+
if ((modelResult.data?.choices?.[0]?.message?.toolCalls?.length ?? 0) > 0) {
|
|
3988
|
+
return { result: null };
|
|
3989
|
+
}
|
|
3730
3990
|
throw new Error('No output received from model');
|
|
3731
3991
|
}
|
|
3992
|
+
// Implicit control flow: "reply in plain text when the task is complete" — prose IS the
|
|
3993
|
+
// designed terminal form, so on a prompt whose OutputType is 'object' a reply that is not JSON is
|
|
3994
|
+
// the answer, not a malformed envelope. Validating it as JSON marks every such prompt run
|
|
3995
|
+
// Failed and spends a "Repair JSON" model call trying to fix prose.
|
|
3996
|
+
// A reply that does parse as JSON — an honoured envelope — takes the normal path.
|
|
3997
|
+
if (prompt.OutputType === 'object' && GetToolCallingDecision(modelResult)?.controlFlow === 'implicit' && !this.parsesAsJSON(rawOutput)) {
|
|
3998
|
+
const accepted = new ValidationResult();
|
|
3999
|
+
accepted.Success = true;
|
|
4000
|
+
return { result: rawOutput, validationResult: accepted };
|
|
4001
|
+
}
|
|
3732
4002
|
// Parse based on output type
|
|
3733
4003
|
let parsedResult = rawOutput;
|
|
3734
4004
|
try {
|