@directive-run/ai 1.13.0 → 1.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/anthropic.d.cts +1 -1
- package/dist/anthropic.d.ts +1 -1
- package/dist/{chunk-XV2QSBBE.cjs → chunk-2TD4ZSGZ.cjs} +7 -7
- package/dist/{chunk-XV2QSBBE.cjs.map → chunk-2TD4ZSGZ.cjs.map} +1 -1
- package/dist/chunk-5JQ2A3JK.js +2 -0
- package/dist/chunk-5JQ2A3JK.js.map +1 -0
- package/dist/chunk-A22KLB23.cjs +67 -0
- package/dist/chunk-A22KLB23.cjs.map +1 -0
- package/dist/chunk-A5K77UDX.cjs +4 -0
- package/dist/chunk-A5K77UDX.cjs.map +1 -0
- package/dist/chunk-DN5NAJM6.js +6 -0
- package/dist/chunk-DN5NAJM6.js.map +1 -0
- package/dist/chunk-FABDFT74.cjs +6 -0
- package/dist/chunk-FABDFT74.cjs.map +1 -0
- package/dist/chunk-FBT73WFY.js +2 -0
- package/dist/chunk-FBT73WFY.js.map +1 -0
- package/dist/chunk-FFRWQNK7.cjs +7 -0
- package/dist/chunk-FFRWQNK7.cjs.map +1 -0
- package/dist/chunk-IR3IHBVQ.cjs +2 -0
- package/dist/chunk-IR3IHBVQ.cjs.map +1 -0
- package/dist/chunk-IUGSMTBE.js +16 -0
- package/dist/{chunk-Q3PQLWBR.js.map → chunk-IUGSMTBE.js.map} +1 -1
- package/dist/chunk-J2Q5KKPN.js +37 -0
- package/dist/chunk-J2Q5KKPN.js.map +1 -0
- package/dist/chunk-K64WKZ22.cjs +2 -0
- package/dist/chunk-K64WKZ22.cjs.map +1 -0
- package/dist/chunk-LKY4K5TV.cjs +11 -0
- package/dist/chunk-LKY4K5TV.cjs.map +1 -0
- package/dist/chunk-LXUMJKGJ.js +67 -0
- package/dist/chunk-LXUMJKGJ.js.map +1 -0
- package/dist/chunk-NNAQ4ZH2.js +11 -0
- package/dist/chunk-NNAQ4ZH2.js.map +1 -0
- package/dist/chunk-PD772MSE.cjs +30 -0
- package/dist/chunk-PD772MSE.cjs.map +1 -0
- package/dist/chunk-POBEEJR6.js +30 -0
- package/dist/chunk-POBEEJR6.js.map +1 -0
- package/dist/chunk-QRXWLD6H.js +4 -0
- package/dist/chunk-QRXWLD6H.js.map +1 -0
- package/dist/chunk-UR4ZGO7V.js +7 -0
- package/dist/chunk-UR4ZGO7V.js.map +1 -0
- package/dist/chunk-UR5BMWEN.js +2 -0
- package/dist/chunk-UR5BMWEN.js.map +1 -0
- package/dist/chunk-WOFIBIPW.cjs +2 -0
- package/dist/chunk-WOFIBIPW.cjs.map +1 -0
- package/dist/chunk-XN5LUOVS.cjs +37 -0
- package/dist/chunk-XN5LUOVS.cjs.map +1 -0
- package/dist/debug-timeline-DpnRMnLU.d.cts +87 -0
- package/dist/debug-timeline-L13P-U2I.d.ts +87 -0
- package/dist/devtools.cjs +2 -0
- package/dist/devtools.cjs.map +1 -0
- package/dist/devtools.d.cts +354 -0
- package/dist/devtools.d.ts +354 -0
- package/dist/devtools.js +2 -0
- package/dist/devtools.js.map +1 -0
- package/dist/evals.cjs +2 -0
- package/dist/evals.cjs.map +1 -0
- package/dist/evals.d.cts +361 -0
- package/dist/evals.d.ts +361 -0
- package/dist/evals.js +2 -0
- package/dist/evals.js.map +1 -0
- package/dist/gemini.d.cts +1 -1
- package/dist/gemini.d.ts +1 -1
- package/dist/guardrails.cjs +2 -0
- package/dist/guardrails.cjs.map +1 -0
- package/dist/guardrails.d.cts +618 -0
- package/dist/guardrails.d.ts +618 -0
- package/dist/guardrails.js +2 -0
- package/dist/guardrails.js.map +1 -0
- package/dist/health-monitor-C6xoXrQz.d.cts +55 -0
- package/dist/health-monitor-qL9RNMH3.d.ts +55 -0
- package/dist/index.cjs +21 -99
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +1197 -4735
- package/dist/index.d.ts +1197 -4735
- package/dist/index.js +21 -99
- package/dist/index.js.map +1 -1
- package/dist/mcp.cjs +2 -0
- package/dist/mcp.cjs.map +1 -0
- package/dist/mcp.d.cts +450 -0
- package/dist/mcp.d.ts +450 -0
- package/dist/mcp.js +2 -0
- package/dist/mcp.js.map +1 -0
- package/dist/multi-agent-orchestrator-QWWQKKGX.js +2 -0
- package/dist/{multi-agent-orchestrator-4PXNYRHB.js.map → multi-agent-orchestrator-QWWQKKGX.js.map} +1 -1
- package/dist/multi-agent-orchestrator-Y5U4JCFO.cjs +2 -0
- package/dist/{multi-agent-orchestrator-KFGTEGE5.cjs.map → multi-agent-orchestrator-Y5U4JCFO.cjs.map} +1 -1
- package/dist/multi-agent.cjs +2 -0
- package/dist/multi-agent.cjs.map +1 -0
- package/dist/multi-agent.d.cts +1429 -0
- package/dist/multi-agent.d.ts +1429 -0
- package/dist/multi-agent.js +2 -0
- package/dist/multi-agent.js.map +1 -0
- package/dist/ollama.d.cts +1 -1
- package/dist/ollama.d.ts +1 -1
- package/dist/openai.d.cts +2 -2
- package/dist/openai.d.ts +2 -2
- package/dist/{orchestrator-types-CTfIKk0W.d.ts → orchestrator-types-DGBhL6mc.d.ts} +5 -138
- package/dist/{orchestrator-types-Bh8r3_Sq.d.cts → orchestrator-types-DeIMRLR7.d.cts} +5 -138
- package/dist/predicate.cjs +2 -0
- package/dist/predicate.cjs.map +1 -0
- package/dist/predicate.d.cts +371 -0
- package/dist/predicate.d.ts +371 -0
- package/dist/predicate.js +2 -0
- package/dist/predicate.js.map +1 -0
- package/dist/{semantic-cache-nBpQqILc.d.cts → semantic-cache-DM7ev7NQ.d.cts} +1 -1
- package/dist/{semantic-cache-nBpQqILc.d.ts → semantic-cache-DM7ev7NQ.d.ts} +1 -1
- package/dist/testing.cjs +1 -1
- package/dist/testing.cjs.map +1 -1
- package/dist/testing.d.cts +4 -2
- package/dist/testing.d.ts +4 -2
- package/dist/testing.js +1 -1
- package/dist/testing.js.map +1 -1
- package/dist/{types-CRmwFnVk.d.cts → types-DJ09LjZX.d.cts} +1 -1
- package/dist/{types-CRmwFnVk.d.ts → types-DJ09LjZX.d.ts} +1 -1
- package/package.json +32 -2
- package/dist/chunk-Q3PQLWBR.js +0 -16
- package/dist/chunk-RW4R3O5P.js +0 -72
- package/dist/chunk-RW4R3O5P.js.map +0 -1
- package/dist/chunk-X3VQ5F7D.cjs +0 -72
- package/dist/chunk-X3VQ5F7D.cjs.map +0 -1
- package/dist/multi-agent-orchestrator-4PXNYRHB.js +0 -2
- package/dist/multi-agent-orchestrator-KFGTEGE5.cjs +0 -2
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":[],"names":[],"mappings":"","file":"devtools.js"}
|
package/dist/evals.cjs
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
1
|
+
'use strict';var chunkXN5LUOVS_cjs=require('./chunk-XN5LUOVS.cjs');require('./chunk-3WO4MWJM.cjs');Object.defineProperty(exports,"createEvalSuite",{enumerable:true,get:function(){return chunkXN5LUOVS_cjs.k}});Object.defineProperty(exports,"evalAssert",{enumerable:true,get:function(){return chunkXN5LUOVS_cjs.l}});Object.defineProperty(exports,"evalCoherence",{enumerable:true,get:function(){return chunkXN5LUOVS_cjs.j}});Object.defineProperty(exports,"evalCost",{enumerable:true,get:function(){return chunkXN5LUOVS_cjs.a}});Object.defineProperty(exports,"evalFaithfulness",{enumerable:true,get:function(){return chunkXN5LUOVS_cjs.h}});Object.defineProperty(exports,"evalJudge",{enumerable:true,get:function(){return chunkXN5LUOVS_cjs.f}});Object.defineProperty(exports,"evalLatency",{enumerable:true,get:function(){return chunkXN5LUOVS_cjs.b}});Object.defineProperty(exports,"evalMatch",{enumerable:true,get:function(){return chunkXN5LUOVS_cjs.g}});Object.defineProperty(exports,"evalOutputLength",{enumerable:true,get:function(){return chunkXN5LUOVS_cjs.c}});Object.defineProperty(exports,"evalRelevance",{enumerable:true,get:function(){return chunkXN5LUOVS_cjs.i}});Object.defineProperty(exports,"evalSafety",{enumerable:true,get:function(){return chunkXN5LUOVS_cjs.d}});Object.defineProperty(exports,"evalStructure",{enumerable:true,get:function(){return chunkXN5LUOVS_cjs.e}});//# sourceMappingURL=evals.cjs.map
|
|
2
|
+
//# sourceMappingURL=evals.cjs.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":[],"names":[],"mappings":"","file":"evals.cjs"}
|
package/dist/evals.d.cts
ADDED
|
@@ -0,0 +1,361 @@
|
|
|
1
|
+
import { D as DebugTimeline } from './debug-timeline-DpnRMnLU.cjs';
|
|
2
|
+
import { q as RunResult, b as AgentLike, c as AgentRunner, R as RunOptions } from './types-DJ09LjZX.cjs';
|
|
3
|
+
import '@directive-run/core';
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Evaluation Framework — Constraint-driven agent evaluation.
|
|
7
|
+
*
|
|
8
|
+
* Define eval criteria as composable functions. Run agents against datasets
|
|
9
|
+
* and score their outputs across multiple dimensions. Results integrate with
|
|
10
|
+
* the debug timeline for DevTools visualization.
|
|
11
|
+
*
|
|
12
|
+
* @example
|
|
13
|
+
* ```typescript
|
|
14
|
+
* const suite = createEvalSuite({
|
|
15
|
+
* criteria: {
|
|
16
|
+
* safe: evalSafety({ categories: ["pii"] }),
|
|
17
|
+
* costEfficient: evalCost({ maxTokensPerRun: 5000 }),
|
|
18
|
+
* fast: evalLatency({ maxMs: 3000 }),
|
|
19
|
+
* },
|
|
20
|
+
* agents: [researchAgent, writerAgent],
|
|
21
|
+
* runner: myRunner,
|
|
22
|
+
* dataset: [
|
|
23
|
+
* { id: "case-1", input: "What is AI?", expected: "explanation about AI" },
|
|
24
|
+
* ],
|
|
25
|
+
* });
|
|
26
|
+
*
|
|
27
|
+
* const results = await suite.run();
|
|
28
|
+
* // results.summary — pass/fail per criterion per agent
|
|
29
|
+
* // results.details — per-case breakdown
|
|
30
|
+
* ```
|
|
31
|
+
*
|
|
32
|
+
* @module
|
|
33
|
+
*/
|
|
34
|
+
|
|
35
|
+
/** Single test case in the eval dataset */
|
|
36
|
+
interface EvalCase {
|
|
37
|
+
/** Unique identifier for tracking across runs */
|
|
38
|
+
id?: string;
|
|
39
|
+
/** Input to feed the agent */
|
|
40
|
+
input: string;
|
|
41
|
+
/** Expected output or reference answer (for comparison-based criteria) */
|
|
42
|
+
expected?: string;
|
|
43
|
+
/** Reference context for faithfulness evaluation */
|
|
44
|
+
context?: string;
|
|
45
|
+
/** Tags for filtering and grouping results */
|
|
46
|
+
tags?: string[];
|
|
47
|
+
/** Additional context passed to criteria */
|
|
48
|
+
metadata?: Record<string, unknown>;
|
|
49
|
+
}
|
|
50
|
+
/** Result of evaluating a single criterion on a single case */
|
|
51
|
+
interface EvalScore {
|
|
52
|
+
/** Score from 0.0 to 1.0 */
|
|
53
|
+
score: number;
|
|
54
|
+
/** Whether this score passes the criterion threshold */
|
|
55
|
+
passed: boolean;
|
|
56
|
+
/** Reason for the score */
|
|
57
|
+
reason?: string;
|
|
58
|
+
/** Duration of evaluation (ms) */
|
|
59
|
+
durationMs: number;
|
|
60
|
+
}
|
|
61
|
+
/** Context passed to eval criterion functions */
|
|
62
|
+
interface EvalContext {
|
|
63
|
+
/** The agent being evaluated */
|
|
64
|
+
agent: AgentLike;
|
|
65
|
+
/** The test case */
|
|
66
|
+
testCase: EvalCase;
|
|
67
|
+
/** The agent's run result */
|
|
68
|
+
result: RunResult<unknown>;
|
|
69
|
+
/** Duration of the agent run (ms) */
|
|
70
|
+
runDurationMs: number;
|
|
71
|
+
}
|
|
72
|
+
/** Eval criterion function — scores an agent's output */
|
|
73
|
+
type EvalCriterionFn = (context: EvalContext) => EvalScore | Promise<EvalScore>;
|
|
74
|
+
/** Named eval criterion */
|
|
75
|
+
interface EvalCriterion {
|
|
76
|
+
name: string;
|
|
77
|
+
fn: EvalCriterionFn;
|
|
78
|
+
/** Score threshold for passing. Default: 0.5 */
|
|
79
|
+
threshold?: number;
|
|
80
|
+
/** Weight for aggregation. Default: 1.0 */
|
|
81
|
+
weight?: number;
|
|
82
|
+
}
|
|
83
|
+
/** Per-case detail result */
|
|
84
|
+
interface EvalCaseResult {
|
|
85
|
+
/** Test case that was evaluated */
|
|
86
|
+
testCase: EvalCase;
|
|
87
|
+
/** Agent that was evaluated */
|
|
88
|
+
agentName: string;
|
|
89
|
+
/** Agent run result */
|
|
90
|
+
runResult: RunResult<unknown>;
|
|
91
|
+
/** Score per criterion */
|
|
92
|
+
scores: Record<string, EvalScore>;
|
|
93
|
+
/** Overall weighted score (0.0-1.0) */
|
|
94
|
+
overallScore: number;
|
|
95
|
+
/** Whether all criteria passed */
|
|
96
|
+
allPassed: boolean;
|
|
97
|
+
/** Agent run duration (ms) */
|
|
98
|
+
runDurationMs: number;
|
|
99
|
+
}
|
|
100
|
+
/** Per-agent summary */
|
|
101
|
+
interface EvalAgentSummary {
|
|
102
|
+
agentName: string;
|
|
103
|
+
/** Average score per criterion */
|
|
104
|
+
criterionAverages: Record<string, number>;
|
|
105
|
+
/** Pass rate per criterion (0.0-1.0) */
|
|
106
|
+
criterionPassRates: Record<string, number>;
|
|
107
|
+
/** Overall weighted average score */
|
|
108
|
+
overallScore: number;
|
|
109
|
+
/** Overall pass rate */
|
|
110
|
+
passRate: number;
|
|
111
|
+
/** Total tokens consumed */
|
|
112
|
+
totalTokens: number;
|
|
113
|
+
/** Average latency per run (ms) */
|
|
114
|
+
avgLatencyMs: number;
|
|
115
|
+
/** Total cases evaluated */
|
|
116
|
+
totalCases: number;
|
|
117
|
+
/** Cases that passed all criteria */
|
|
118
|
+
passedCases: number;
|
|
119
|
+
}
|
|
120
|
+
/** Complete eval suite results */
|
|
121
|
+
interface EvalResults {
|
|
122
|
+
/** Summary per agent */
|
|
123
|
+
summary: Record<string, EvalAgentSummary>;
|
|
124
|
+
/** Detailed per-case results */
|
|
125
|
+
details: EvalCaseResult[];
|
|
126
|
+
/** Total duration (ms) */
|
|
127
|
+
durationMs: number;
|
|
128
|
+
/** Total tokens consumed across all agents and cases */
|
|
129
|
+
totalTokens: number;
|
|
130
|
+
/** Timestamp when the eval started */
|
|
131
|
+
startedAt: number;
|
|
132
|
+
/** Timestamp when the eval completed */
|
|
133
|
+
completedAt: number;
|
|
134
|
+
}
|
|
135
|
+
/** Configuration for createEvalSuite */
|
|
136
|
+
interface EvalSuiteConfig {
|
|
137
|
+
/** Named criteria to evaluate */
|
|
138
|
+
criteria: Record<string, EvalCriterionFn | EvalCriterion>;
|
|
139
|
+
/** Agents to evaluate */
|
|
140
|
+
agents: AgentLike[];
|
|
141
|
+
/** Agent runner function */
|
|
142
|
+
runner: AgentRunner;
|
|
143
|
+
/** Dataset of test cases */
|
|
144
|
+
dataset: EvalCase[];
|
|
145
|
+
/** Run options passed to the runner */
|
|
146
|
+
runOptions?: Omit<RunOptions, "signal">;
|
|
147
|
+
/** Maximum concurrent agent runs. Default: 5 */
|
|
148
|
+
concurrency?: number;
|
|
149
|
+
/** Optional debug timeline for recording eval events */
|
|
150
|
+
timeline?: DebugTimeline;
|
|
151
|
+
/** Callback fired on each case completion */
|
|
152
|
+
onCaseComplete?: (result: EvalCaseResult) => void;
|
|
153
|
+
/** Callback fired on each agent completion */
|
|
154
|
+
onAgentComplete?: (summary: EvalAgentSummary) => void;
|
|
155
|
+
/** Abort signal */
|
|
156
|
+
signal?: AbortSignal;
|
|
157
|
+
}
|
|
158
|
+
/** Eval suite instance */
|
|
159
|
+
interface EvalSuite {
|
|
160
|
+
/** Run the full evaluation */
|
|
161
|
+
run(): Promise<EvalResults>;
|
|
162
|
+
/** Run evaluation for a specific agent only */
|
|
163
|
+
runAgent(agentName: string): Promise<EvalAgentSummary>;
|
|
164
|
+
/** Get the list of agents being evaluated */
|
|
165
|
+
getAgents(): AgentLike[];
|
|
166
|
+
/** Get the list of criteria */
|
|
167
|
+
getCriteria(): string[];
|
|
168
|
+
/** Get the dataset */
|
|
169
|
+
getDataset(): EvalCase[];
|
|
170
|
+
}
|
|
171
|
+
/** Options for cost evaluation */
|
|
172
|
+
interface EvalCostOptions {
|
|
173
|
+
/** Maximum tokens per run */
|
|
174
|
+
maxTokensPerRun: number;
|
|
175
|
+
}
|
|
176
|
+
/**
|
|
177
|
+
* Evaluate cost efficiency — scores based on token usage relative to a budget.
|
|
178
|
+
*
|
|
179
|
+
* Score = 1.0 when tokens \<= maxTokensPerRun * 0.5,
|
|
180
|
+
* Score = 0.0 when tokens \>= maxTokensPerRun * 2.
|
|
181
|
+
* Linear interpolation between.
|
|
182
|
+
*
|
|
183
|
+
* @param options - Cost evaluation options including `maxTokensPerRun`.
|
|
184
|
+
* @returns An eval criterion that scores token usage against the budget.
|
|
185
|
+
*/
|
|
186
|
+
declare function evalCost(options: EvalCostOptions): EvalCriterion;
|
|
187
|
+
/** Options for latency evaluation */
|
|
188
|
+
interface EvalLatencyOptions {
|
|
189
|
+
/** Maximum acceptable latency (ms) */
|
|
190
|
+
maxMs: number;
|
|
191
|
+
}
|
|
192
|
+
/**
|
|
193
|
+
* Evaluate latency — scores based on agent run duration.
|
|
194
|
+
*
|
|
195
|
+
* Score = 1.0 when duration \<= maxMs * 0.5,
|
|
196
|
+
* Score = 0.0 when duration \>= maxMs * 2.
|
|
197
|
+
* Linear interpolation between.
|
|
198
|
+
*
|
|
199
|
+
* @param options - Latency evaluation options including `maxMs`.
|
|
200
|
+
* @returns An eval criterion that scores run duration against the limit.
|
|
201
|
+
*/
|
|
202
|
+
declare function evalLatency(options: EvalLatencyOptions): EvalCriterion;
|
|
203
|
+
/** Options for output length evaluation */
|
|
204
|
+
interface EvalOutputLengthOptions {
|
|
205
|
+
/** Minimum output length (chars) */
|
|
206
|
+
minLength?: number;
|
|
207
|
+
/** Maximum output length (chars) */
|
|
208
|
+
maxLength?: number;
|
|
209
|
+
}
|
|
210
|
+
/**
|
|
211
|
+
* Evaluate output length — ensures output is within an acceptable range.
|
|
212
|
+
*/
|
|
213
|
+
declare function evalOutputLength(options: EvalOutputLengthOptions): EvalCriterion;
|
|
214
|
+
/** Options for safety evaluation */
|
|
215
|
+
interface EvalSafetyOptions {
|
|
216
|
+
/** Patterns to check for in output (overrides categories) */
|
|
217
|
+
blockedPatterns?: RegExp[];
|
|
218
|
+
/** Categories of content to check: "pii", "violence", "self_harm", "illegal" */
|
|
219
|
+
categories?: Array<"pii" | "violence" | "self_harm" | "illegal">;
|
|
220
|
+
}
|
|
221
|
+
/**
|
|
222
|
+
* Evaluate safety — checks output for blocked patterns or category-based content.
|
|
223
|
+
*
|
|
224
|
+
* When `categories` is provided, uses built-in pattern sets for each category.
|
|
225
|
+
* When `blockedPatterns` is provided, uses those directly (overrides categories).
|
|
226
|
+
* When neither is provided, defaults to all safety categories.
|
|
227
|
+
*
|
|
228
|
+
* Score = 1.0 when no blocked patterns found.
|
|
229
|
+
* Score = 0.0 when any blocked pattern matches.
|
|
230
|
+
*/
|
|
231
|
+
declare function evalSafety(options?: EvalSafetyOptions): EvalCriterion;
|
|
232
|
+
/** Options for output structure evaluation */
|
|
233
|
+
interface EvalStructureOptions {
|
|
234
|
+
/** Expected output type */
|
|
235
|
+
type?: "json" | "string";
|
|
236
|
+
/** Required keys if type is "json" */
|
|
237
|
+
requiredKeys?: string[];
|
|
238
|
+
}
|
|
239
|
+
/**
|
|
240
|
+
* Evaluate output structure — checks that output matches an expected format.
|
|
241
|
+
*/
|
|
242
|
+
declare function evalStructure(options: EvalStructureOptions): EvalCriterion;
|
|
243
|
+
/**
|
|
244
|
+
* Evaluate with a custom LLM judge — uses a runner to grade the output.
|
|
245
|
+
*
|
|
246
|
+
* The judge agent receives the input, output, and expected answer, and
|
|
247
|
+
* returns a JSON score.
|
|
248
|
+
*/
|
|
249
|
+
interface EvalJudgeOptions {
|
|
250
|
+
/** Runner to use for the judge */
|
|
251
|
+
runner: AgentRunner;
|
|
252
|
+
/** Judge agent */
|
|
253
|
+
judge: AgentLike;
|
|
254
|
+
/** Custom grading prompt template. {{input}}, {{output}}, {{expected}} are replaced. */
|
|
255
|
+
promptTemplate?: string;
|
|
256
|
+
/** Optional abort signal */
|
|
257
|
+
signal?: AbortSignal;
|
|
258
|
+
/** Timeout for the judge call in ms. Default: 30_000 */
|
|
259
|
+
timeoutMs?: number;
|
|
260
|
+
}
|
|
261
|
+
/**
|
|
262
|
+
* Evaluate output quality by delegating to a judge agent that scores from 0.0 to 1.0.
|
|
263
|
+
*
|
|
264
|
+
* @param options - Judge evaluation options including `runner`, `judge` agent, and optional `promptTemplate`.
|
|
265
|
+
* @returns An eval criterion that runs a judge agent and returns its score.
|
|
266
|
+
*/
|
|
267
|
+
declare function evalJudge(options: EvalJudgeOptions): EvalCriterion;
|
|
268
|
+
/**
|
|
269
|
+
* Evaluate exact or substring match against expected output.
|
|
270
|
+
*/
|
|
271
|
+
interface EvalMatchOptions {
|
|
272
|
+
/** Match mode. Default: "contains" */
|
|
273
|
+
mode?: "exact" | "contains" | "regex";
|
|
274
|
+
/** Case-insensitive matching. Default: true */
|
|
275
|
+
caseInsensitive?: boolean;
|
|
276
|
+
}
|
|
277
|
+
/**
|
|
278
|
+
* Evaluate exact or substring match against expected output.
|
|
279
|
+
*
|
|
280
|
+
* @param options - Match evaluation options including `mode` and `caseInsensitive`.
|
|
281
|
+
* @returns An eval criterion that checks output against the expected value.
|
|
282
|
+
*/
|
|
283
|
+
declare function evalMatch(options?: EvalMatchOptions): EvalCriterion;
|
|
284
|
+
/** Options for LLM-based semantic evaluation criteria */
|
|
285
|
+
interface EvalSemanticOptions {
|
|
286
|
+
/** Runner to use for the judge LLM */
|
|
287
|
+
runner: AgentRunner;
|
|
288
|
+
/** Judge agent (model to use for evaluation) */
|
|
289
|
+
judge: AgentLike;
|
|
290
|
+
/** Optional abort signal */
|
|
291
|
+
signal?: AbortSignal;
|
|
292
|
+
/** Timeout for the judge call in ms. Default: 30_000 */
|
|
293
|
+
timeoutMs?: number;
|
|
294
|
+
}
|
|
295
|
+
/**
|
|
296
|
+
* Evaluate faithfulness — whether the output is grounded in the provided context.
|
|
297
|
+
*
|
|
298
|
+
* Requires `context` field on the EvalCase. Uses an LLM judge internally
|
|
299
|
+
* to extract and verify claims against the reference context.
|
|
300
|
+
*/
|
|
301
|
+
declare function evalFaithfulness(options: EvalSemanticOptions): EvalCriterion;
|
|
302
|
+
/**
|
|
303
|
+
* Evaluate relevance — whether the output directly addresses the input question.
|
|
304
|
+
*
|
|
305
|
+
* Uses an LLM judge to assess how well the agent's output answers
|
|
306
|
+
* the original question.
|
|
307
|
+
*/
|
|
308
|
+
declare function evalRelevance(options: EvalSemanticOptions): EvalCriterion;
|
|
309
|
+
/**
|
|
310
|
+
* Evaluate coherence — whether the output is logically consistent and well-structured.
|
|
311
|
+
*
|
|
312
|
+
* Uses an LLM judge to assess the internal coherence, logical flow,
|
|
313
|
+
* and consistency of the output.
|
|
314
|
+
*/
|
|
315
|
+
declare function evalCoherence(options: EvalSemanticOptions): EvalCriterion;
|
|
316
|
+
/**
|
|
317
|
+
* Create an evaluation suite for testing agents against a dataset.
|
|
318
|
+
*
|
|
319
|
+
* @example
|
|
320
|
+
* ```typescript
|
|
321
|
+
* const suite = createEvalSuite({
|
|
322
|
+
* criteria: {
|
|
323
|
+
* fast: evalLatency({ maxMs: 3000 }),
|
|
324
|
+
* cheap: evalCost({ maxTokensPerRun: 5000 }),
|
|
325
|
+
* },
|
|
326
|
+
* agents: [researchAgent, writerAgent],
|
|
327
|
+
* runner: myRunner,
|
|
328
|
+
* dataset: [{ input: "What is AI?" }],
|
|
329
|
+
* });
|
|
330
|
+
*
|
|
331
|
+
* const results = await suite.run();
|
|
332
|
+
* ```
|
|
333
|
+
*/
|
|
334
|
+
declare function createEvalSuite(config: EvalSuiteConfig): EvalSuite;
|
|
335
|
+
/** Options for eval assertions in CI */
|
|
336
|
+
interface EvalAssertOptions {
|
|
337
|
+
/** Minimum weighted overall score required (0.0-1.0) */
|
|
338
|
+
minScore?: number;
|
|
339
|
+
/** Minimum pass rate required (0.0-1.0) */
|
|
340
|
+
minPassRate?: number;
|
|
341
|
+
/** Criteria that must achieve 100% pass rate */
|
|
342
|
+
failOn?: string[];
|
|
343
|
+
}
|
|
344
|
+
/**
|
|
345
|
+
* Assert eval results meet requirements — designed for CI pipelines.
|
|
346
|
+
*
|
|
347
|
+
* Throws an error with details if any assertion fails.
|
|
348
|
+
*
|
|
349
|
+
* @example
|
|
350
|
+
* ```typescript
|
|
351
|
+
* const results = await suite.run();
|
|
352
|
+
* evalAssert(results, {
|
|
353
|
+
* minScore: 0.8,
|
|
354
|
+
* minPassRate: 0.9,
|
|
355
|
+
* failOn: ["safety"],
|
|
356
|
+
* });
|
|
357
|
+
* ```
|
|
358
|
+
*/
|
|
359
|
+
declare function evalAssert(results: EvalResults, options: EvalAssertOptions): void;
|
|
360
|
+
|
|
361
|
+
export { type EvalAgentSummary, type EvalAssertOptions, type EvalCase, type EvalCaseResult, type EvalContext, type EvalCostOptions, type EvalCriterion, type EvalCriterionFn, type EvalJudgeOptions, type EvalLatencyOptions, type EvalMatchOptions, type EvalOutputLengthOptions, type EvalResults, type EvalSafetyOptions, type EvalScore, type EvalSemanticOptions, type EvalStructureOptions, type EvalSuite, type EvalSuiteConfig, createEvalSuite, evalAssert, evalCoherence, evalCost, evalFaithfulness, evalJudge, evalLatency, evalMatch, evalOutputLength, evalRelevance, evalSafety, evalStructure };
|