@memberjunction/testing-engine 6.1.0-edge.5 → 6.1.0-edge.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/drivers/AgentPromptComposer.d.ts +50 -0
- package/dist/drivers/AgentPromptComposer.d.ts.map +1 -0
- package/dist/drivers/AgentPromptComposer.js +60 -0
- package/dist/drivers/AgentPromptComposer.js.map +1 -0
- package/dist/drivers/PinnedVendorPromptRunner.d.ts +30 -0
- package/dist/drivers/PinnedVendorPromptRunner.d.ts.map +1 -0
- package/dist/drivers/PinnedVendorPromptRunner.js +30 -0
- package/dist/drivers/PinnedVendorPromptRunner.js.map +1 -0
- package/dist/drivers/PromptEvalDriver.d.ts +158 -0
- package/dist/drivers/PromptEvalDriver.d.ts.map +1 -0
- package/dist/drivers/PromptEvalDriver.js +365 -0
- package/dist/drivers/PromptEvalDriver.js.map +1 -0
- package/dist/drivers/index.d.ts +1 -0
- package/dist/drivers/index.d.ts.map +1 -1
- package/dist/drivers/index.js +1 -0
- package/dist/drivers/index.js.map +1 -1
- package/dist/engine/TestEngine.d.ts.map +1 -1
- package/dist/engine/TestEngine.js +5 -0
- package/dist/engine/TestEngine.js.map +1 -1
- package/dist/eval/corpus.d.ts +91 -0
- package/dist/eval/corpus.d.ts.map +1 -0
- package/dist/eval/corpus.js +141 -0
- package/dist/eval/corpus.js.map +1 -0
- package/dist/eval/decision.d.ts +156 -0
- package/dist/eval/decision.d.ts.map +1 -0
- package/dist/eval/decision.js +257 -0
- package/dist/eval/decision.js.map +1 -0
- package/dist/eval/expectation.d.ts +76 -0
- package/dist/eval/expectation.d.ts.map +1 -0
- package/dist/eval/expectation.js +160 -0
- package/dist/eval/expectation.js.map +1 -0
- package/dist/eval/history.d.ts +17 -0
- package/dist/eval/history.d.ts.map +1 -0
- package/dist/eval/history.js +19 -0
- package/dist/eval/history.js.map +1 -0
- package/dist/eval/index.d.ts +15 -0
- package/dist/eval/index.d.ts.map +1 -0
- package/dist/eval/index.js +15 -0
- package/dist/eval/index.js.map +1 -0
- package/dist/eval/matchers.d.ts +88 -0
- package/dist/eval/matchers.d.ts.map +1 -0
- package/dist/eval/matchers.js +119 -0
- package/dist/eval/matchers.js.map +1 -0
- package/dist/eval/subAgentTrace.d.ts +52 -0
- package/dist/eval/subAgentTrace.d.ts.map +1 -0
- package/dist/eval/subAgentTrace.js +61 -0
- package/dist/eval/subAgentTrace.js.map +1 -0
- package/dist/eval/wellFormed.d.ts +41 -0
- package/dist/eval/wellFormed.d.ts.map +1 -0
- package/dist/eval/wellFormed.js +83 -0
- package/dist/eval/wellFormed.js.map +1 -0
- package/dist/index.d.ts +5 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5 -0
- package/dist/index.js.map +1 -1
- package/dist/oracles/AgentDecisionOracle.d.ts +54 -0
- package/dist/oracles/AgentDecisionOracle.d.ts.map +1 -0
- package/dist/oracles/AgentDecisionOracle.js +107 -0
- package/dist/oracles/AgentDecisionOracle.js.map +1 -0
- package/dist/oracles/TraceSubAgentValidatorOracle.d.ts +44 -0
- package/dist/oracles/TraceSubAgentValidatorOracle.d.ts.map +1 -0
- package/dist/oracles/TraceSubAgentValidatorOracle.js +105 -0
- package/dist/oracles/TraceSubAgentValidatorOracle.js.map +1 -0
- package/dist/oracles/index.d.ts +2 -0
- package/dist/oracles/index.d.ts.map +1 -1
- package/dist/oracles/index.js +2 -0
- package/dist/oracles/index.js.map +1 -1
- package/package.json +10 -10
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Composes an agent's prompt the way production does, for evaluation.
|
|
3
|
+
* @module @memberjunction/testing-engine
|
|
4
|
+
*/
|
|
5
|
+
import { BaseAgent } from '@memberjunction/ai-agents';
|
|
6
|
+
import type { NativeToolBinding } from '@memberjunction/ai-agents';
|
|
7
|
+
import type { AgentConfiguration, AIPromptParams, ExecuteAgentParams, MJAIAgentEntityExtended } from '@memberjunction/ai-core-plus';
|
|
8
|
+
/**
|
|
9
|
+
* Exposes `BaseAgent`'s prompt composition so an evaluation driver can build the SAME prompt the
|
|
10
|
+
* runtime would build.
|
|
11
|
+
*
|
|
12
|
+
* **Why this exists rather than the driver assembling a prompt itself.** An MJ agent's prompt is
|
|
13
|
+
* not its own prompt record. `BaseAgent` runs the *agent type's* system prompt as the parent and
|
|
14
|
+
* injects the agent's prompt as a child at `agentType.AgentPromptPlaceholder`, and the parent is
|
|
15
|
+
* what carries the response contract, the action catalog and the sub-agent list. A driver that
|
|
16
|
+
* executes the agent's own prompt alone therefore measures something production never runs:
|
|
17
|
+
* a baseline run scored the Demo Loop Agent at 100% malformed with a 626-character
|
|
18
|
+
* system prompt containing no action catalog, so the model invented an action name it had never
|
|
19
|
+
* been shown. Codesmith, whose own prompt happens to restate everything, scored 88% on the same
|
|
20
|
+
* models — the gap was the harness, not the agents.
|
|
21
|
+
*
|
|
22
|
+
* Reimplementing the composition here would reproduce that class of error the moment `BaseAgent`
|
|
23
|
+
* changed. Subclassing and widening the two protected members keeps exactly one implementation:
|
|
24
|
+
* whatever production does, the evaluation inherits.
|
|
25
|
+
*
|
|
26
|
+
* Nothing is overridden and no agent is executed — this class is used only to build parameters.
|
|
27
|
+
*/
|
|
28
|
+
export declare class AgentPromptComposer extends BaseAgent {
|
|
29
|
+
/** Resolves the agent's type, system prompt and child prompt. */
|
|
30
|
+
LoadConfiguration(agent: MJAIAgentEntityExtended): Promise<AgentConfiguration>;
|
|
31
|
+
/**
|
|
32
|
+
* Builds the composed `AIPromptParams` — parent system prompt, child agent prompt, and the
|
|
33
|
+
* template data that renders the action catalog and sub-agent list.
|
|
34
|
+
*/
|
|
35
|
+
ComposeParams<P>(config: AgentConfiguration, payload: P, params: ExecuteAgentParams): Promise<AIPromptParams>;
|
|
36
|
+
/**
|
|
37
|
+
* The tool-name → Action map produced by the most recent {@link ComposeParams} call.
|
|
38
|
+
*
|
|
39
|
+
* A native tool call comes back under the SANITIZED name (`execute_code`), while the corpus
|
|
40
|
+
* states its expectation as the Action's real name (`Execute Code`) — test plan §1.2 requires
|
|
41
|
+
* a case to say what the agent should decide, never how the wire spells it. Without this map
|
|
42
|
+
* the eval harness would score every correct native call as a wrong action, which is not a
|
|
43
|
+
* measurement of the feature but of the sanitizer.
|
|
44
|
+
*
|
|
45
|
+
* Exposing the framework's own map rather than re-deriving it is the point: the harness reads
|
|
46
|
+
* the same table the loop dispatches from, so the two cannot drift.
|
|
47
|
+
*/
|
|
48
|
+
get NativeToolBindings(): ReadonlyMap<string, NativeToolBinding> | undefined;
|
|
49
|
+
}
|
|
50
|
+
//# sourceMappingURL=AgentPromptComposer.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"AgentPromptComposer.d.ts","sourceRoot":"","sources":["../../src/drivers/AgentPromptComposer.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH,OAAO,EAAE,SAAS,EAAE,MAAM,2BAA2B,CAAC;AACtD,OAAO,KAAK,EAAE,iBAAiB,EAAE,MAAM,2BAA2B,CAAC;AACnE,OAAO,KAAK,EAAE,kBAAkB,EAAE,cAAc,EAAE,kBAAkB,EAAE,uBAAuB,EAAE,MAAM,8BAA8B,CAAC;AAEpI;;;;;;;;;;;;;;;;;;;GAmBG;AACH,qBAAa,mBAAoB,SAAQ,SAAS;IAC9C,iEAAiE;IACpD,iBAAiB,CAAC,KAAK,EAAE,uBAAuB,GAAG,OAAO,CAAC,kBAAkB,CAAC;IAI3F;;;OAGG;IACU,aAAa,CAAC,CAAC,EACxB,MAAM,EAAE,kBAAkB,EAC1B,OAAO,EAAE,CAAC,EACV,MAAM,EAAE,kBAAkB,GAC3B,OAAO,CAAC,cAAc,CAAC;IAU1B;;;;;;;;;;;OAWG;IACH,IAAW,kBAAkB,IAAI,WAAW,CAAC,MAAM,EAAE,iBAAiB,CAAC,GAAG,SAAS,CAElF;CACJ"}
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Composes an agent's prompt the way production does, for evaluation.
|
|
3
|
+
* @module @memberjunction/testing-engine
|
|
4
|
+
*/
|
|
5
|
+
import { BaseAgent } from '@memberjunction/ai-agents';
|
|
6
|
+
/**
|
|
7
|
+
* Exposes `BaseAgent`'s prompt composition so an evaluation driver can build the SAME prompt the
|
|
8
|
+
* runtime would build.
|
|
9
|
+
*
|
|
10
|
+
* **Why this exists rather than the driver assembling a prompt itself.** An MJ agent's prompt is
|
|
11
|
+
* not its own prompt record. `BaseAgent` runs the *agent type's* system prompt as the parent and
|
|
12
|
+
* injects the agent's prompt as a child at `agentType.AgentPromptPlaceholder`, and the parent is
|
|
13
|
+
* what carries the response contract, the action catalog and the sub-agent list. A driver that
|
|
14
|
+
* executes the agent's own prompt alone therefore measures something production never runs:
|
|
15
|
+
* a baseline run scored the Demo Loop Agent at 100% malformed with a 626-character
|
|
16
|
+
* system prompt containing no action catalog, so the model invented an action name it had never
|
|
17
|
+
* been shown. Codesmith, whose own prompt happens to restate everything, scored 88% on the same
|
|
18
|
+
* models — the gap was the harness, not the agents.
|
|
19
|
+
*
|
|
20
|
+
* Reimplementing the composition here would reproduce that class of error the moment `BaseAgent`
|
|
21
|
+
* changed. Subclassing and widening the two protected members keeps exactly one implementation:
|
|
22
|
+
* whatever production does, the evaluation inherits.
|
|
23
|
+
*
|
|
24
|
+
* Nothing is overridden and no agent is executed — this class is used only to build parameters.
|
|
25
|
+
*/
|
|
26
|
+
export class AgentPromptComposer extends BaseAgent {
|
|
27
|
+
/** Resolves the agent's type, system prompt and child prompt. */
|
|
28
|
+
async LoadConfiguration(agent) {
|
|
29
|
+
return this.loadAgentConfiguration(agent);
|
|
30
|
+
}
|
|
31
|
+
/**
|
|
32
|
+
* Builds the composed `AIPromptParams` — parent system prompt, child agent prompt, and the
|
|
33
|
+
* template data that renders the action catalog and sub-agent list.
|
|
34
|
+
*/
|
|
35
|
+
async ComposeParams(config, payload, params) {
|
|
36
|
+
// The runtime resolves the agent TYPE before its first turn, and `BaseAgent` now consults it
|
|
37
|
+
// when deciding whether Actions may be declared as tools (`SupportsNativeToolCalls`). A
|
|
38
|
+
// composer that skipped this step would compose a prompt for a type-less agent, declare
|
|
39
|
+
// nothing, and every "native" eval cell would quietly measure the envelope — the exact
|
|
40
|
+
// mislabelling the scorecard's attribution guard exists to catch. Same call the loop makes.
|
|
41
|
+
await this.initializeAgentType(params, config);
|
|
42
|
+
return this.preparePromptParams(config, payload, params);
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* The tool-name → Action map produced by the most recent {@link ComposeParams} call.
|
|
46
|
+
*
|
|
47
|
+
* A native tool call comes back under the SANITIZED name (`execute_code`), while the corpus
|
|
48
|
+
* states its expectation as the Action's real name (`Execute Code`) — test plan §1.2 requires
|
|
49
|
+
* a case to say what the agent should decide, never how the wire spells it. Without this map
|
|
50
|
+
* the eval harness would score every correct native call as a wrong action, which is not a
|
|
51
|
+
* measurement of the feature but of the sanitizer.
|
|
52
|
+
*
|
|
53
|
+
* Exposing the framework's own map rather than re-deriving it is the point: the harness reads
|
|
54
|
+
* the same table the loop dispatches from, so the two cannot drift.
|
|
55
|
+
*/
|
|
56
|
+
get NativeToolBindings() {
|
|
57
|
+
return this._nativeToolBindings;
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
//# sourceMappingURL=AgentPromptComposer.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"AgentPromptComposer.js","sourceRoot":"","sources":["../../src/drivers/AgentPromptComposer.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH,OAAO,EAAE,SAAS,EAAE,MAAM,2BAA2B,CAAC;AAItD;;;;;;;;;;;;;;;;;;;GAmBG;AACH,MAAM,OAAO,mBAAoB,SAAQ,SAAS;IAC9C,iEAAiE;IAC1D,KAAK,CAAC,iBAAiB,CAAC,KAA8B;QACzD,OAAO,IAAI,CAAC,sBAAsB,CAAC,KAAK,CAAC,CAAC;IAC9C,CAAC;IAED;;;OAGG;IACI,KAAK,CAAC,aAAa,CACtB,MAA0B,EAC1B,OAAU,EACV,MAA0B;QAE1B,6FAA6F;QAC7F,wFAAwF;QACxF,wFAAwF;QACxF,uFAAuF;QACvF,4FAA4F;QAC5F,MAAM,IAAI,CAAC,mBAAmB,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;QAC/C,OAAO,IAAI,CAAC,mBAAmB,CAAC,MAAM,EAAE,OAAO,EAAE,MAAM,CAAC,CAAC;IAC7D,CAAC;IAED;;;;;;;;;;;OAWG;IACH,IAAW,kBAAkB;QACzB,OAAO,IAAI,CAAC,mBAAmB,CAAC;IACpC,CAAC;CACJ"}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview A prompt runner that will not fail over away from the cell's pinned vendor.
|
|
3
|
+
* @module @memberjunction/testing-engine
|
|
4
|
+
*/
|
|
5
|
+
import { AIPromptRunner } from '@memberjunction/ai-prompts';
|
|
6
|
+
import type { FailoverConfiguration } from '@memberjunction/ai-prompts';
|
|
7
|
+
import type { MJAIPromptEntityExtended } from '@memberjunction/ai-core-plus';
|
|
8
|
+
/**
|
|
9
|
+
* Disables cross-vendor failover for a matrix cell that pinned a (model, vendor).
|
|
10
|
+
*
|
|
11
|
+
* **Why an eval run must not fail over.** A cell IS a vendor: `gpt-oss-120b · Cerebras · native`
|
|
12
|
+
* means "this model, on Cerebras, with tools". The Loop agent-type system prompt ships
|
|
13
|
+
* `FailoverStrategy = 'SameModelDifferentVendor'`, which is the right production behavior — a
|
|
14
|
+
* transient Cerebras outage should not fail a user's agent run — and exactly the wrong measurement
|
|
15
|
+
* behavior, because the row still lands in the results table under the Cerebras label.
|
|
16
|
+
*
|
|
17
|
+
* That is not hypothetical. An aborted comparison run recorded 27 results whose actual
|
|
18
|
+
* failure was `Invalid Vertex AI credentials JSON`: a transient error on the pinned Google vendor
|
|
19
|
+
* failed over to Vertex, whose credential in this environment is malformed. Those rows scored as
|
|
20
|
+
* Google cells producing no output. A pinned run that fails is a data point; a pinned run that
|
|
21
|
+
* silently measures somewhere else is a corrupted one.
|
|
22
|
+
*
|
|
23
|
+
* `getFailoverConfiguration` is `protected` and documented as the override point for exactly this,
|
|
24
|
+
* so no production surface changes — the harness declines a behavior rather than the runner
|
|
25
|
+
* growing a switch for it.
|
|
26
|
+
*/
|
|
27
|
+
export declare class PinnedVendorPromptRunner extends AIPromptRunner {
|
|
28
|
+
protected getFailoverConfiguration(prompt: MJAIPromptEntityExtended): FailoverConfiguration;
|
|
29
|
+
}
|
|
30
|
+
//# sourceMappingURL=PinnedVendorPromptRunner.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"PinnedVendorPromptRunner.d.ts","sourceRoot":"","sources":["../../src/drivers/PinnedVendorPromptRunner.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH,OAAO,EAAE,cAAc,EAAE,MAAM,4BAA4B,CAAC;AAC5D,OAAO,KAAK,EAAE,qBAAqB,EAAE,MAAM,4BAA4B,CAAC;AACxE,OAAO,KAAK,EAAE,wBAAwB,EAAE,MAAM,8BAA8B,CAAC;AAE7E;;;;;;;;;;;;;;;;;;GAkBG;AACH,qBAAa,wBAAyB,SAAQ,cAAc;cACrC,wBAAwB,CAAC,MAAM,EAAE,wBAAwB,GAAG,qBAAqB;CAGvG"}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview A prompt runner that will not fail over away from the cell's pinned vendor.
|
|
3
|
+
* @module @memberjunction/testing-engine
|
|
4
|
+
*/
|
|
5
|
+
import { AIPromptRunner } from '@memberjunction/ai-prompts';
|
|
6
|
+
/**
|
|
7
|
+
* Disables cross-vendor failover for a matrix cell that pinned a (model, vendor).
|
|
8
|
+
*
|
|
9
|
+
* **Why an eval run must not fail over.** A cell IS a vendor: `gpt-oss-120b · Cerebras · native`
|
|
10
|
+
* means "this model, on Cerebras, with tools". The Loop agent-type system prompt ships
|
|
11
|
+
* `FailoverStrategy = 'SameModelDifferentVendor'`, which is the right production behavior — a
|
|
12
|
+
* transient Cerebras outage should not fail a user's agent run — and exactly the wrong measurement
|
|
13
|
+
* behavior, because the row still lands in the results table under the Cerebras label.
|
|
14
|
+
*
|
|
15
|
+
* That is not hypothetical. An aborted comparison run recorded 27 results whose actual
|
|
16
|
+
* failure was `Invalid Vertex AI credentials JSON`: a transient error on the pinned Google vendor
|
|
17
|
+
* failed over to Vertex, whose credential in this environment is malformed. Those rows scored as
|
|
18
|
+
* Google cells producing no output. A pinned run that fails is a data point; a pinned run that
|
|
19
|
+
* silently measures somewhere else is a corrupted one.
|
|
20
|
+
*
|
|
21
|
+
* `getFailoverConfiguration` is `protected` and documented as the override point for exactly this,
|
|
22
|
+
* so no production surface changes — the harness declines a behavior rather than the runner
|
|
23
|
+
* growing a switch for it.
|
|
24
|
+
*/
|
|
25
|
+
export class PinnedVendorPromptRunner extends AIPromptRunner {
|
|
26
|
+
getFailoverConfiguration(prompt) {
|
|
27
|
+
return { ...super.getFailoverConfiguration(prompt), strategy: 'None' };
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
//# sourceMappingURL=PinnedVendorPromptRunner.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"PinnedVendorPromptRunner.js","sourceRoot":"","sources":["../../src/drivers/PinnedVendorPromptRunner.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAEH,OAAO,EAAE,cAAc,EAAE,MAAM,4BAA4B,CAAC;AAI5D;;;;;;;;;;;;;;;;;;GAkBG;AACH,MAAM,OAAO,wBAAyB,SAAQ,cAAc;IACrC,wBAAwB,CAAC,MAAgC;QACxE,OAAO,EAAE,GAAG,KAAK,CAAC,wBAAwB,CAAC,MAAM,CAAC,EAAE,QAAQ,EAAE,MAAM,EAAE,CAAC;IAC3E,CAAC;CACJ"}
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Prompt-eval driver — one model decision, evaluated deterministically.
|
|
3
|
+
* @module @memberjunction/testing-engine
|
|
4
|
+
*/
|
|
5
|
+
import type { MJTestEntity } from '@memberjunction/core-entities';
|
|
6
|
+
import type { ChatMessage, ChatToolChoice } from '@memberjunction/ai';
|
|
7
|
+
import { BaseTestDriver } from './BaseTestDriver.js';
|
|
8
|
+
import { DriverExecutionContext, DriverExecutionResult, ValidationResult } from '../types.js';
|
|
9
|
+
/** One oracle to run, as it appears in the test's `Configuration`. */
|
|
10
|
+
export interface PromptEvalOracleConfig {
|
|
11
|
+
type: string;
|
|
12
|
+
weight?: number;
|
|
13
|
+
config?: Record<string, unknown>;
|
|
14
|
+
}
|
|
15
|
+
/** `Configuration` for a Prompt Eval test. */
|
|
16
|
+
export interface PromptEvalConfig {
|
|
17
|
+
/** The prompt to run. Exactly one of `promptId` / `agentId` is required. */
|
|
18
|
+
promptId?: string;
|
|
19
|
+
/** Resolve the prompt from an agent's active prompt binding instead of naming it. */
|
|
20
|
+
agentId?: string;
|
|
21
|
+
/**
|
|
22
|
+
* Same, by agent NAME. The corpus states its agent by name — golden files stay readable and
|
|
23
|
+
* diffable, and the suite generator runs offline with no database to resolve a GUID against.
|
|
24
|
+
*/
|
|
25
|
+
agentName?: string;
|
|
26
|
+
oracles: PromptEvalOracleConfig[];
|
|
27
|
+
/** Per-cell model pinning — this is the matrix axis (§3.3). */
|
|
28
|
+
modelId?: string;
|
|
29
|
+
vendorId?: string;
|
|
30
|
+
configurationId?: string;
|
|
31
|
+
/**
|
|
32
|
+
* Per-cell reasoning effort. Numeric 1-100 is MJ's cross-provider scale; a provider-named
|
|
33
|
+
* level (OpenAI's `'xhigh'` / `'none'`) is passed through verbatim for the levels that scale
|
|
34
|
+
* cannot express. Part of the cell identity, so two effort levels are two cells, not noise
|
|
35
|
+
* pooled into one.
|
|
36
|
+
*/
|
|
37
|
+
effortLevel?: number | string;
|
|
38
|
+
/**
|
|
39
|
+
* Which wire encoding this cell observes.
|
|
40
|
+
*
|
|
41
|
+
* The composer attaches the agent's Actions as native tool declarations unconditionally,
|
|
42
|
+
* because that is what `BaseAgent` does and the whole point of composing through it is to send
|
|
43
|
+
* what production sends. A cell measuring the ENVELOPE therefore has to take them back off:
|
|
44
|
+
* `toolsProvided` is one of the three terms in the runner's gate, so withholding declarations
|
|
45
|
+
* is exactly how a real envelope-mode call differs from a native one. Nothing else about the
|
|
46
|
+
* request changes, which is what keeps the two arms comparable.
|
|
47
|
+
*
|
|
48
|
+
* Defaults to `'envelope'`, so a record written before this axis existed —
|
|
49
|
+
* keeps measuring the baseline it was generated to measure.
|
|
50
|
+
*/
|
|
51
|
+
toolCallingMode?: 'envelope' | 'native';
|
|
52
|
+
/**
|
|
53
|
+
* Control-flow arm for a native cell. `'implicit'` keeps the control tools the composer
|
|
54
|
+
* declared (the model's catalog must also say `NativeControlFlow: 'implicit'` for the runner to
|
|
55
|
+
* send them); `'envelope'` (default) strips them here so the cell measures the hybrid on a
|
|
56
|
+
* model whose catalog has been switched to implicit for the gate run.
|
|
57
|
+
*/
|
|
58
|
+
nativeControlFlow?: 'envelope' | 'implicit';
|
|
59
|
+
/**
|
|
60
|
+
* Native-tool-results arm: send tool-form history (assistant toolCalls + tool turns) as native tool turns.
|
|
61
|
+
* Otherwise the driver renders that history as the corpus's `[Action Result] …` prose, which is
|
|
62
|
+
* what the loop shows models that do not return results natively.
|
|
63
|
+
*/
|
|
64
|
+
nativeToolResults?: boolean;
|
|
65
|
+
/** `tool_choice` for a native cell. Omitted accepts the agent's own default (`'auto'`). */
|
|
66
|
+
toolChoice?: ChatToolChoice;
|
|
67
|
+
/** Replaces the composed system prompt wholesale. Used by the trimmed-prompt variant cells. */
|
|
68
|
+
systemPromptOverride?: string;
|
|
69
|
+
maxExecutionTime?: number;
|
|
70
|
+
scoringWeights?: Record<string, number>;
|
|
71
|
+
}
|
|
72
|
+
/** `InputDefinition` for a Prompt Eval test — the frozen state the model decides from. */
|
|
73
|
+
export interface PromptEvalInput {
|
|
74
|
+
/** Prior conversation, including any action-result annotations for mid-loop cases. */
|
|
75
|
+
conversationMessages?: ChatMessage[];
|
|
76
|
+
/** The starting payload. Surfaced to the template as `_CURRENT_PAYLOAD` (§3.1). */
|
|
77
|
+
payload?: Record<string, unknown>;
|
|
78
|
+
/** Anything else the template needs. */
|
|
79
|
+
templateData?: Record<string, unknown>;
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* Runs ONE prompt and evaluates the single decision it produced.
|
|
83
|
+
*
|
|
84
|
+
* This is the corpus workhorse (test plan §3.1). Given frozen mid-loop state — a payload, a
|
|
85
|
+
* conversation, a system prompt — it asks the model for exactly one decision and hands the raw turn
|
|
86
|
+
* to the decision oracles. **No action ever executes.** `AIPromptRunner.ExecutePrompt` returns the
|
|
87
|
+
* model's reply and nothing is dispatched, which is what makes the evaluation a plain data
|
|
88
|
+
* assertion rather than a side-effecting integration test.
|
|
89
|
+
*
|
|
90
|
+
* Why a separate driver rather than reusing `AgentEvalDriver`: an agent run is a *loop*, and a loop
|
|
91
|
+
* confounds the measurement. If the corpus is asking "does this model, in this state, choose action
|
|
92
|
+
* B", then a run that recovered on iteration three answers a different question. Loop-level cases
|
|
93
|
+
* still go through `AgentEvalDriver`; this driver isolates the single decision.
|
|
94
|
+
*
|
|
95
|
+
* Three execution details are load-bearing and easy to get wrong:
|
|
96
|
+
* - **explicit `contextUser`** — the CLI provider's `CurrentUser` is null (#3251), so an omitted
|
|
97
|
+
* context user fails deep inside template rendering with an unrelated-looking error;
|
|
98
|
+
* - **`AIEngine.Config()` before the first run** — otherwise prompt/model metadata is empty;
|
|
99
|
+
* - **`WaitForPendingPromptRunSaves()` before oracles read back** — prompt-run persistence is
|
|
100
|
+
* fire-and-forget, so an oracle asserting on `AIPromptRun` would race it.
|
|
101
|
+
*/
|
|
102
|
+
export declare class PromptEvalDriver extends BaseTestDriver {
|
|
103
|
+
private static readonly AI_PROMPT_RUNS_ENTITY_NAME;
|
|
104
|
+
private _promptRunsEntityId;
|
|
105
|
+
Execute(context: DriverExecutionContext): Promise<DriverExecutionResult>;
|
|
106
|
+
/**
|
|
107
|
+
* Turns the runner's result into the encoding-neutral shape the decision oracles read.
|
|
108
|
+
*
|
|
109
|
+
* Both channels are carried, unjudged: the raw text and any native tool calls. Deciding which
|
|
110
|
+
* one *counts* is the normalizer's job, not the driver's — that separation is what lets the
|
|
111
|
+
* same corpus case score an envelope run and a native run.
|
|
112
|
+
*/
|
|
113
|
+
/** Which protocol a cell measures — the normalizer reads plain text differently under implicit. */
|
|
114
|
+
private protocolOf;
|
|
115
|
+
private extractTurn;
|
|
116
|
+
private buildParams;
|
|
117
|
+
/**
|
|
118
|
+
* Builds the prompt parameters, composing the agent the way the runtime composes it.
|
|
119
|
+
*
|
|
120
|
+
* An MJ agent's prompt is not its own prompt record: `BaseAgent` runs the AGENT TYPE's system
|
|
121
|
+
* prompt as the parent and injects the agent's prompt as a child, and only the parent carries
|
|
122
|
+
* the response contract, the action catalog and the sub-agent list. Executing the agent's own
|
|
123
|
+
* prompt alone measures a prompt production never sends — see {@link AgentPromptComposer}.
|
|
124
|
+
*
|
|
125
|
+
* `promptId` still runs that one prompt bare, which is the point of naming a prompt rather
|
|
126
|
+
* than an agent: it is the only way to evaluate a prompt outside any agent.
|
|
127
|
+
*/
|
|
128
|
+
private buildComposedParams;
|
|
129
|
+
/** The matrix axes (§3.3): model/vendor pinning, effort, configuration — applied last so a cell wins. */
|
|
130
|
+
private applyCellOverrides;
|
|
131
|
+
/**
|
|
132
|
+
* Applies the cell's tool-calling mode by adding or withholding the tool declarations.
|
|
133
|
+
*
|
|
134
|
+
* This driver never decides whether tools GO OUT — `AIPromptRunner`'s gate does, from metadata,
|
|
135
|
+
* exactly as it does in production. All the cell controls is whether any were offered. That
|
|
136
|
+
* split matters for what a comparison number means: a `native` cell whose model or prompt is not
|
|
137
|
+
* configured for native mode does not quietly become a native run, it records an envelope one,
|
|
138
|
+
* and the mismatch shows up here rather than as an unexplained delta in the scorecard.
|
|
139
|
+
*/
|
|
140
|
+
private applyToolCallingMode;
|
|
141
|
+
/** Resolves the prompt either directly or through the agent's active prompt binding. */
|
|
142
|
+
private resolvePrompt;
|
|
143
|
+
/** Agent name -> ID. Returns undefined when no name was given, so `agentId` can be absent too. */
|
|
144
|
+
private resolveAgentIdByName;
|
|
145
|
+
private findAgentPromptId;
|
|
146
|
+
/**
|
|
147
|
+
* Runs the configured oracles.
|
|
148
|
+
*
|
|
149
|
+
* A missing oracle becomes a FAILED result naming it, rather than being skipped. Skipping is
|
|
150
|
+
* how `trace-validate-sub-agents` went unimplemented while the tests referencing it reported
|
|
151
|
+
* green — a quarter of their weight silently left the score. A weighted check that cannot run
|
|
152
|
+
* is a failure of the test, not an absence of one.
|
|
153
|
+
*/
|
|
154
|
+
private runOracles;
|
|
155
|
+
Validate(test: MJTestEntity): Promise<ValidationResult>;
|
|
156
|
+
private getPromptRunsEntityId;
|
|
157
|
+
}
|
|
158
|
+
//# sourceMappingURL=PromptEvalDriver.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"PromptEvalDriver.d.ts","sourceRoot":"","sources":["../../src/drivers/PromptEvalDriver.ts"],"names":[],"mappings":"AAAA;;;GAGG;AAOH,OAAO,KAAK,EAAE,YAAY,EAAyB,MAAM,+BAA+B,CAAC;AAMzF,OAAO,KAAK,EAAE,WAAW,EAAE,cAAc,EAAE,MAAM,oBAAoB,CAAC;AACtE,OAAO,EAAE,cAAc,EAAE,MAAM,kBAAkB,CAAC;AAClD,OAAO,EAAE,sBAAsB,EAAE,qBAAqB,EAAe,gBAAgB,EAAE,MAAM,UAAU,CAAC;AAIxG,sEAAsE;AACtE,MAAM,WAAW,sBAAsB;IACnC,IAAI,EAAE,MAAM,CAAC;IACb,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,MAAM,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CACpC;AAED,8CAA8C;AAC9C,MAAM,WAAW,gBAAgB;IAC7B,4EAA4E;IAC5E,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,qFAAqF;IACrF,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB;;;OAGG;IACH,SAAS,CAAC,EAAE,MAAM,CAAC;IACnB,OAAO,EAAE,sBAAsB,EAAE,CAAC;IAClC,+DAA+D;IAC/D,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB;;;;;OAKG;IACH,WAAW,CAAC,EAAE,MAAM,GAAG,MAAM,CAAC;IAC9B;;;;;;;;;;;;OAYG;IACH,eAAe,CAAC,EAAE,UAAU,GAAG,QAAQ,CAAC;IACxC;;;;;OAKG;IACH,iBAAiB,CAAC,EAAE,UAAU,GAAG,UAAU,CAAC;IAC5C;;;;OAIG;IACH,iBAAiB,CAAC,EAAE,OAAO,CAAC;IAC5B,2FAA2F;IAC3F,UAAU,CAAC,EAAE,cAAc,CAAC;IAC5B,+FAA+F;IAC/F,oBAAoB,CAAC,EAAE,MAAM,CAAC;IAC9B,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,cAAc,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;CAC3C;AAiBD,0FAA0F;AAC1F,MAAM,WAAW,eAAe;IAC5B,sFAAsF;IACtF,oBAAoB,CAAC,EAAE,WAAW,EAAE,CAAC;IACrC,mFAAmF;IACnF,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;IAClC,wCAAwC;IACxC,YAAY,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CAC1C;AAED;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,qBACa,gBAAiB,SAAQ,cAAc;IAChD,OAAO,CAAC,MAAM,CAAC,QAAQ,CAAC,0BAA0B,CAAwB;IAC1E,OAAO,CAAC,mBAAmB,CAAuB;IAErC,OAAO,CAAC,OAAO,EAAE,sBAAsB,GAAG,OAAO,CAAC,qBAAqB,CAAC;IA2CrF;;;;;;OAMG;IACH,mGAAmG;IACnG,OAAO,CAAC,UAAU;IAKlB,OAAO,CAAC,WAAW;IAoBnB,OAAO,CAAC,WAAW;IAwBnB;;;;;;;;;;OAUG;YACW,mBAAmB;IA2DjC,yGAAyG;IACzG,OAAO,CAAC,kBAAkB;IAkB1B;;;;;;;;OAQG;IACH,OAAO,CAAC,oBAAoB;IAsC5B,wFAAwF;YAC1E,aAAa;IAU3B,kGAAkG;YACpF,oBAAoB;YAsBpB,iBAAiB;IAiB/B;;;;;;;OAOG;YACW,UAAU;IAwCF,QAAQ,CAAC,IAAI,EAAE,YAAY,GAAG,OAAO,CAAC,gBAAgB,CAAC;IAwB7E,OAAO,CAAC,qBAAqB;CAQhC"}
|