agentv 5.3.1-next.1 → 5.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. package/README.md +52 -44
  2. package/dist/{artifact-writer-7NBCOAYC.js → artifact-writer-KJEOROKQ.js} +5 -5
  3. package/dist/{chunk-ELCJ23K4.js → chunk-6262OKXM.js} +2856 -2404
  4. package/dist/chunk-6262OKXM.js.map +1 -0
  5. package/dist/{chunk-LKGARI3W.js → chunk-6RFRY7X2.js} +812 -570
  6. package/dist/chunk-6RFRY7X2.js.map +1 -0
  7. package/dist/chunk-AQ5BIAXF.js +604 -0
  8. package/dist/chunk-AQ5BIAXF.js.map +1 -0
  9. package/dist/chunk-BV5VQLI2.js +2 -0
  10. package/dist/{chunk-LXBI3SPX.js → chunk-JGSRUJZQ.js} +46 -20
  11. package/dist/chunk-JGSRUJZQ.js.map +1 -0
  12. package/dist/chunk-MMDLXYBX.js +721 -0
  13. package/dist/chunk-MMDLXYBX.js.map +1 -0
  14. package/dist/chunk-TEEXVJWM.js +2 -0
  15. package/dist/{chunk-RKE7SSET.js → chunk-WIAJ7JAV.js} +186 -64
  16. package/dist/chunk-WIAJ7JAV.js.map +1 -0
  17. package/dist/cli.d.ts +1 -0
  18. package/dist/cli.js +17517 -10
  19. package/dist/cli.js.map +1 -1
  20. package/dist/config.d.ts +2 -0
  21. package/dist/config.js +14 -0
  22. package/dist/config.js.map +1 -0
  23. package/dist/contracts-DsZmZLl8.d.ts +742 -0
  24. package/dist/contracts.d.ts +2 -0
  25. package/dist/contracts.js +28 -0
  26. package/dist/contracts.js.map +1 -0
  27. package/dist/dashboard/assets/{index-CbEMiJSb.js → index-BfqOlLOF.js} +1 -1
  28. package/dist/dashboard/assets/index-Bh8lpGce.css +1 -0
  29. package/dist/dashboard/assets/index-oPR3ywQb.js +121 -0
  30. package/dist/dashboard/index.html +2 -2
  31. package/dist/{dist-NMXMI5SK.js → dist-A3SGR7TW.js} +16 -16
  32. package/dist/dist-A3SGR7TW.js.map +1 -0
  33. package/dist/index.d.ts +4 -0
  34. package/dist/index.js +137 -18
  35. package/dist/{interactive-BN527UV3.js → interactive-WPQJDJKZ.js} +24 -24
  36. package/dist/interactive-WPQJDJKZ.js.map +1 -0
  37. package/dist/provider.d.ts +2 -0
  38. package/dist/provider.js +24 -0
  39. package/dist/provider.js.map +1 -0
  40. package/dist/sdk.d.ts +802 -0
  41. package/dist/sdk.js +145 -0
  42. package/dist/sdk.js.map +1 -0
  43. package/dist/skills/agentv-eval-migrations/references/breaking-changes.md +17 -18
  44. package/dist/skills/agentv-eval-writer/SKILL.md +27 -15
  45. package/dist/skills/agentv-eval-writer/references/custom-evaluators.md +5 -5
  46. package/dist/skills/agentv-eval-writer/references/eval.schema.json +5402 -4639
  47. package/dist/skills/agentv-eval-writer/references/python-helpers.md +2 -2
  48. package/dist/skills/agentv-eval-writer/references/rubric-evaluator.md +19 -2
  49. package/dist/templates/.agentv/providers.yaml +42 -0
  50. package/dist/templates/.env.example +2 -2
  51. package/dist/{ts-eval-loader-2RFVZHCT-7CZ3DCDD.js → ts-eval-loader-3G5GEAEC-6F52JLPP.js} +3 -3
  52. package/dist/ts-eval-loader-3G5GEAEC-6F52JLPP.js.map +1 -0
  53. package/package.json +29 -4
  54. package/dist/chunk-ASIGJIOJ.js +0 -17993
  55. package/dist/chunk-ASIGJIOJ.js.map +0 -1
  56. package/dist/chunk-ELCJ23K4.js.map +0 -1
  57. package/dist/chunk-LKGARI3W.js.map +0 -1
  58. package/dist/chunk-LXBI3SPX.js.map +0 -1
  59. package/dist/chunk-RKE7SSET.js.map +0 -1
  60. package/dist/dashboard/assets/index-DTA6-l7q.js +0 -121
  61. package/dist/dashboard/assets/index-D_bokML8.css +0 -1
  62. package/dist/interactive-BN527UV3.js.map +0 -1
  63. package/dist/templates/.agentv/targets.yaml +0 -97
  64. /package/dist/{artifact-writer-7NBCOAYC.js.map → artifact-writer-KJEOROKQ.js.map} +0 -0
  65. /package/dist/{dist-NMXMI5SK.js.map → chunk-BV5VQLI2.js.map} +0 -0
  66. /package/dist/{ts-eval-loader-2RFVZHCT-7CZ3DCDD.js.map → chunk-TEEXVJWM.js.map} +0 -0
package/dist/sdk.d.ts ADDED
@@ -0,0 +1,802 @@
1
+ export { A as AgentVConfig, d as defineConfig, z } from './contracts-DsZmZLl8.js';
2
+ import { ScriptGraderInput, TokenUsage, ScriptGraderResult, ScriptGraderHandler } from '@agentv/core/script-grader';
3
+ export { CodeGraderHandler, CodeGraderInput, CodeGraderInputSchema, CodeGraderResult, CodeGraderResultSchema, Content, ContentFile, ContentFileSchema, ContentImage, ContentImageSchema, ContentSchema, ContentText, ContentTextSchema, Message, MessageSchema, PromptTemplateInput, PromptTemplateInputSchema, ScriptGraderCheck, ScriptGraderCheckSchema, ScriptGraderHandler, ScriptGraderInput, ScriptGraderInputSchema, ScriptGraderResult, ScriptGraderResultSchema, TRACE_EVENT_TYPES, TRACE_REDACTION_LEVELS, TRACE_SOURCE_KINDS, TRACE_TOOL_STATUSES, TokenUsage, TokenUsageSchema, ToolCall, ToolCallSchema, Trace, TraceArtifact, TraceArtifactSchema, TraceBranch, TraceBranchSchema, TraceError, TraceErrorSchema, TraceEvent, TraceEventSchema, TraceMessage, TraceMessageSchema, TraceModel, TraceModelSchema, TraceRawEvidence, TraceRawEvidenceSchema, TraceRedactionState, TraceRedactionStateSchema, TraceSchema, TraceSession, TraceSessionSchema, TraceSource, TraceSourceRef, TraceSourceRefSchema, TraceSourceSchema, TraceSummary, TraceSummarySchema, TraceTool, TraceToolSchema, VitestWorkspaceGraderOptions, defineVitestWorkspaceGrader, runCodeGrader, runScriptGrader, runVitestWorkspaceGrader, vitestReportToCodeGraderResult, vitestReportToScriptGraderResult } from '@agentv/core/script-grader';
4
+ export { a3 as AssertEntry, a4 as ConversationTurnInput, a8 as EvalAssertionInput, a6 as EvalRunArtifacts, a9 as EvalRunResult, aa as EvalSummary, a7 as EvalTestInput, a2 as evaluate } from './ts-eval-loader-kf6J7wSc.js';
5
+
6
+ declare const EVAL_SUITE_SYMBOL: unique symbol;
7
+ declare const TO_EVAL_YAML_OBJECT_SYMBOL: unique symbol;
8
+ declare const KNOWN_SNAKE_CASE_KEYS: {
9
+ readonly afterAll: "after_all";
10
+ readonly afterEach: "after_each";
11
+ readonly argsMatch: "args_match";
12
+ readonly beforeAll: "before_all";
13
+ readonly beforeEach: "before_each";
14
+ readonly budgetUsd: "budget_usd";
15
+ readonly conversationId: "conversation_id";
16
+ readonly dependsOn: "depends_on";
17
+ readonly expectedOutput: "expected_output";
18
+ readonly explorationTolerance: "exploration_tolerance";
19
+ readonly failOnError: "fail_on_error";
20
+ readonly inputFiles: "input_files";
21
+ readonly keepWorkspaces: "keep_workspaces";
22
+ readonly maxCalls: "max_calls";
23
+ readonly maxCostUsd: "max_cost_usd";
24
+ readonly maxDurationMs: "max_duration_ms";
25
+ readonly maxInput: "max_input";
26
+ readonly maxLlmCalls: "max_llm_calls";
27
+ readonly maxOutput: "max_output";
28
+ readonly maxSteps: "max_steps";
29
+ readonly maxTokens: "max_tokens";
30
+ readonly maxToolCalls: "max_tool_calls";
31
+ readonly minScore: "min_score";
32
+ readonly onDependencyFailure: "on_dependency_failure";
33
+ readonly onTurnFailure: "on_turn_failure";
34
+ readonly outputPath: "output_path";
35
+ readonly readOnly: "read_only";
36
+ readonly reasoningEffort: "reasoning_effort";
37
+ readonly rubricPrompt: "rubric_prompt";
38
+ readonly scoreRange: "score_range";
39
+ readonly scoreRanges: "score_ranges";
40
+ readonly skipDefaults: "skip_defaults";
41
+ readonly targetExplorationRatio: "target_exploration_ratio";
42
+ readonly timeoutMs: "timeout_ms";
43
+ readonly timeoutSeconds: "timeout_seconds";
44
+ readonly useTarget: "use_target";
45
+ readonly defaultTest: "default_test";
46
+ readonly windowSize: "window_size";
47
+ };
48
+ type KnownSnakeCaseKeyMap = typeof KNOWN_SNAKE_CASE_KEYS;
49
+ type LowerEvalKey<Key extends string> = Key extends keyof KnownSnakeCaseKeyMap ? KnownSnakeCaseKeyMap[Key] : Key;
50
+ type LowerEvalYamlValue<Value> = Value extends readonly (infer Item)[] ? LowerEvalYamlValue<Item>[] : Value extends object ? {
51
+ [Key in keyof Value as Key extends string ? LowerEvalKey<Key> : never]: LowerEvalYamlValue<Value[Key]>;
52
+ } : Value;
53
+ type EvalMessageContent = string | Readonly<Record<string, unknown>> | readonly (string | Readonly<Record<string, unknown>>)[];
54
+ interface EvalMessage {
55
+ readonly role: 'system' | 'user' | 'assistant' | 'tool';
56
+ readonly content: EvalMessageContent;
57
+ readonly [key: string]: unknown;
58
+ }
59
+ interface EvalAssertionConfig {
60
+ readonly type: string;
61
+ readonly provider?: string | true | object;
62
+ readonly [key: string]: unknown;
63
+ }
64
+ interface EvalLifecycleHook {
65
+ readonly command?: string | readonly string[];
66
+ readonly timeoutMs?: number;
67
+ readonly cwd?: string;
68
+ readonly reset?: 'none' | 'fast' | 'strict';
69
+ readonly [key: string]: unknown;
70
+ }
71
+ interface EvalLifecycleHooks {
72
+ readonly enabled?: boolean;
73
+ readonly beforeAll?: EvalLifecycleHook;
74
+ readonly beforeEach?: EvalLifecycleHook;
75
+ readonly afterEach?: EvalLifecycleHook;
76
+ readonly afterAll?: EvalLifecycleHook;
77
+ }
78
+ interface EvalEnvironmentSetup {
79
+ readonly command: readonly string[];
80
+ readonly cwd?: string;
81
+ readonly timeoutMs?: number;
82
+ }
83
+ interface EvalDockerEnvironmentMount {
84
+ readonly source: string;
85
+ readonly target: string;
86
+ readonly access?: 'ro' | 'rw';
87
+ readonly readOnly?: boolean;
88
+ }
89
+ interface EvalDockerEnvironmentResources {
90
+ readonly cpus?: number;
91
+ readonly memory?: string;
92
+ readonly disk?: string;
93
+ readonly gpu?: boolean | string;
94
+ }
95
+ interface EvalHostEnvironment {
96
+ readonly type: 'host';
97
+ readonly workdir: string;
98
+ readonly setup?: EvalEnvironmentSetup;
99
+ readonly env?: Readonly<Record<string, string>>;
100
+ }
101
+ interface EvalDockerEnvironment {
102
+ readonly type: 'docker';
103
+ readonly workdir: string;
104
+ readonly context?: string;
105
+ readonly dockerfile?: string;
106
+ readonly image?: string;
107
+ readonly setup?: EvalEnvironmentSetup;
108
+ readonly env?: Readonly<Record<string, string>>;
109
+ readonly resources?: EvalDockerEnvironmentResources;
110
+ readonly mounts?: readonly EvalDockerEnvironmentMount[];
111
+ readonly secrets?: Readonly<Record<string, string>>;
112
+ }
113
+ type EvalEnvironment = EvalHostEnvironment | EvalDockerEnvironment;
114
+ interface EvalProviderRef {
115
+ readonly label: string;
116
+ readonly id?: string;
117
+ readonly useProvider?: string;
118
+ readonly hooks?: EvalLifecycleHooks;
119
+ }
120
+ interface EvalProviderConfig {
121
+ readonly extends?: string;
122
+ readonly id: string;
123
+ readonly label?: string;
124
+ readonly model?: string;
125
+ readonly config?: Readonly<Record<string, unknown>>;
126
+ readonly prompts?: unknown;
127
+ readonly transform?: unknown;
128
+ readonly delay?: number;
129
+ readonly inputs?: unknown;
130
+ readonly env?: Readonly<Record<string, string>>;
131
+ readonly reasoningEffort?: string;
132
+ readonly hooks?: EvalLifecycleHooks;
133
+ readonly [key: string]: unknown;
134
+ }
135
+ type EvalProviderMap = Readonly<Record<string, Omit<EvalProviderConfig, 'id'> | EvalProviderRef | Readonly<Record<string, unknown>>>>;
136
+ type EvalProviderEntry = string | EvalProviderRef | EvalProviderConfig | EvalProviderMap;
137
+ interface EvalDefaultsConfig {
138
+ readonly provider?: string;
139
+ readonly grader?: string;
140
+ readonly [key: string]: unknown;
141
+ }
142
+ interface EvalTestOptions {
143
+ readonly provider?: string;
144
+ readonly transform?: unknown;
145
+ readonly repeat?: EvalRepeat;
146
+ readonly rubricPrompt?: unknown;
147
+ readonly [key: string]: unknown;
148
+ }
149
+ interface EvalDefaultTest {
150
+ readonly vars?: Readonly<Record<string, unknown>>;
151
+ readonly assert?: readonly (string | EvalAssertionConfig)[];
152
+ readonly options?: EvalTestOptions;
153
+ readonly [key: string]: unknown;
154
+ }
155
+ type EvalRepeat = number;
156
+ interface EvalExecution {
157
+ readonly provider?: string;
158
+ readonly providers?: readonly EvalProviderEntry[];
159
+ readonly assert?: readonly EvalAssertionConfig[];
160
+ readonly skipDefaults?: boolean;
161
+ readonly cache?: boolean;
162
+ readonly trials?: never;
163
+ readonly budgetUsd?: number;
164
+ readonly failOnError?: boolean;
165
+ readonly threshold?: number;
166
+ readonly [key: string]: unknown;
167
+ }
168
+ interface EvalTurn {
169
+ readonly input: EvalMessageContent;
170
+ readonly expectedOutput?: EvalMessageContent;
171
+ readonly assert?: readonly (string | EvalAssertionConfig)[];
172
+ }
173
+ interface EvalTest {
174
+ readonly id: string;
175
+ readonly vars?: Readonly<Record<string, unknown>>;
176
+ readonly criteria?: string;
177
+ readonly inputFiles?: readonly string[];
178
+ readonly expectedOutput?: string | Readonly<Record<string, unknown>> | readonly EvalMessage[];
179
+ readonly assert?: readonly EvalAssertionConfig[];
180
+ readonly options?: EvalTestOptions;
181
+ readonly execution?: EvalExecution;
182
+ readonly environment?: EvalEnvironment | string;
183
+ readonly metadata?: Readonly<Record<string, unknown>>;
184
+ readonly conversationId?: string;
185
+ readonly suite?: string;
186
+ readonly dependsOn?: readonly string[];
187
+ readonly onDependencyFailure?: 'skip' | 'fail' | 'run';
188
+ readonly mode?: 'conversation';
189
+ readonly turns?: readonly EvalTurn[];
190
+ readonly aggregation?: 'mean' | 'min' | 'max';
191
+ readonly onTurnFailure?: 'continue' | 'stop';
192
+ readonly windowSize?: number;
193
+ }
194
+ interface EvalRequires {
195
+ readonly agentv?: string;
196
+ readonly [key: string]: unknown;
197
+ }
198
+ interface EvalConfig {
199
+ readonly $schema?: string;
200
+ readonly name?: string;
201
+ readonly description?: string;
202
+ readonly category?: string;
203
+ readonly version?: string;
204
+ readonly author?: string;
205
+ /**
206
+ * Suite tags. Either the selection list form (`string[]`, drives
207
+ * `select.tags` / `--tag name` filtering) or the promptfoo-shaped
208
+ * `Record<string,string>` map. In the map form the reserved `experiment` key
209
+ * labels the run/experiment (grouped by the Dashboard), matching
210
+ * `tags.experiment` in YAML evals.
211
+ */
212
+ readonly tags?: readonly string[] | Readonly<Record<string, string>>;
213
+ readonly license?: string;
214
+ readonly requires?: EvalRequires;
215
+ readonly inputFiles?: readonly string[];
216
+ readonly prompts?: unknown;
217
+ readonly providers?: readonly EvalProviderEntry[];
218
+ readonly defaults?: EvalDefaultsConfig;
219
+ readonly defaultTest?: EvalDefaultTest | string;
220
+ readonly tests: readonly EvalTest[] | string;
221
+ /** @deprecated Rejected. Use `tags: { experiment: '<name>' }` or CLI --experiment. */
222
+ readonly experiment?: never;
223
+ readonly repeat?: EvalRepeat;
224
+ readonly timeoutSeconds?: number;
225
+ readonly threshold?: number;
226
+ readonly budgetUsd?: number;
227
+ readonly assert?: readonly EvalAssertionConfig[];
228
+ readonly environment?: EvalEnvironment | string;
229
+ }
230
+ interface DefinedEvalSuite {
231
+ readonly [EVAL_SUITE_SYMBOL]: true;
232
+ readonly [TO_EVAL_YAML_OBJECT_SYMBOL]: () => Record<string, unknown>;
233
+ }
234
+ /**
235
+ * Define a YAML-aligned eval suite in TypeScript.
236
+ *
237
+ * The returned object preserves the TypeScript authoring shape and carries a
238
+ * non-enumerable lowering hook so AgentV can materialize the canonical
239
+ * snake_case eval contract when the suite is loaded from a `.eval.ts` file.
240
+ */
241
+ declare function defineEval<T extends EvalConfig>(definition: T): T & DefinedEvalSuite;
242
+ /**
243
+ * Lower a TypeScript-authored eval suite into the canonical snake_case object
244
+ * contract used by YAML files and the runtime loader.
245
+ *
246
+ * Only known AgentV wire keys are converted. Unknown keys are preserved as-is
247
+ * so opaque assertion, provider, and metadata payloads are not corrupted.
248
+ */
249
+ declare function toEvalYamlObject<T extends EvalConfig | DefinedEvalSuite>(definition: T): LowerEvalYamlValue<T>;
250
+ /**
251
+ * Serialize an eval suite to canonical YAML.
252
+ */
253
+ declare function serializeEvalYaml<T extends EvalConfig | DefinedEvalSuite>(definition: T): string;
254
+
255
+ type GraderCommand = string | readonly string[];
256
+ interface GraderHelperOptions {
257
+ readonly metric?: string;
258
+ readonly weight?: number;
259
+ readonly required?: boolean;
260
+ readonly minScore?: number;
261
+ readonly negate?: boolean;
262
+ readonly transform?: string;
263
+ }
264
+ interface GraderCommonConfig {
265
+ readonly metric?: string;
266
+ readonly weight?: number;
267
+ readonly required?: boolean;
268
+ readonly minScore?: number;
269
+ readonly negate?: boolean;
270
+ readonly transform?: string;
271
+ }
272
+ interface ContainsGraderConfig extends EvalAssertionConfig, GraderCommonConfig {
273
+ readonly type: 'contains';
274
+ readonly value: string;
275
+ }
276
+ interface EqualsGraderConfig extends EvalAssertionConfig, GraderCommonConfig {
277
+ readonly type: 'equals';
278
+ readonly value: string;
279
+ }
280
+ interface RegexGraderOptions extends GraderHelperOptions {
281
+ readonly flags?: string;
282
+ }
283
+ interface RegexGraderConfig extends EvalAssertionConfig, GraderCommonConfig {
284
+ readonly type: 'regex';
285
+ readonly value: string;
286
+ readonly flags?: string;
287
+ }
288
+ interface IsJsonGraderConfig extends EvalAssertionConfig, GraderCommonConfig {
289
+ readonly type: 'is-json';
290
+ }
291
+ type GraderRubricOperator = 'correctness' | 'contradiction';
292
+ interface GraderScoreRange {
293
+ readonly scoreRange: readonly [number, number];
294
+ readonly outcome: string;
295
+ }
296
+ interface GraderRubric {
297
+ readonly id?: string;
298
+ readonly outcome?: string;
299
+ readonly criteria?: string;
300
+ readonly operator?: GraderRubricOperator;
301
+ readonly weight?: number;
302
+ readonly required?: boolean;
303
+ readonly minScore?: number;
304
+ readonly scoreRanges?: readonly GraderScoreRange[];
305
+ }
306
+ type GraderRubricCriterion = string | GraderRubric;
307
+ interface LlmRubricGraderConfig extends EvalAssertionConfig, GraderCommonConfig {
308
+ readonly type: 'llm-rubric';
309
+ readonly value?: unknown;
310
+ readonly prompt?: string | GraderPromptScriptConfig;
311
+ readonly provider?: string;
312
+ readonly config?: Readonly<Record<string, unknown>>;
313
+ readonly maxSteps?: number;
314
+ readonly temperature?: number;
315
+ }
316
+ interface GraderPromptScriptConfig {
317
+ readonly command: readonly string[];
318
+ readonly config?: Readonly<Record<string, unknown>>;
319
+ }
320
+ interface ScriptGraderProviderOptions {
321
+ readonly maxCalls?: number;
322
+ }
323
+ interface ScriptGraderOptions extends GraderHelperOptions {
324
+ readonly cwd?: string;
325
+ readonly provider?: true | ScriptGraderProviderOptions;
326
+ readonly config?: Readonly<Record<string, unknown>>;
327
+ }
328
+ interface ScriptGraderConfig extends EvalAssertionConfig, GraderCommonConfig {
329
+ readonly type: 'script';
330
+ readonly command: GraderCommand;
331
+ readonly cwd?: string;
332
+ readonly provider?: true | ScriptGraderProviderOptions;
333
+ readonly config?: Readonly<Record<string, unknown>>;
334
+ }
335
+ /** @deprecated Use ScriptGraderProviderOptions. */
336
+ type ScriptGraderTargetOptions = ScriptGraderProviderOptions;
337
+ /** @deprecated Use ScriptGraderProviderOptions. */
338
+ type CodeGraderTargetOptions = ScriptGraderProviderOptions;
339
+ /** @deprecated Use ScriptGraderOptions. */
340
+ type CodeGraderOptions = ScriptGraderOptions;
341
+ /** @deprecated Use ScriptGraderConfig with type: 'script'. */
342
+ type CodeGraderConfig = ScriptGraderConfig;
343
+ type GraderHelperConfig = ContainsGraderConfig | EqualsGraderConfig | RegexGraderConfig | IsJsonGraderConfig | LlmRubricGraderConfig | ScriptGraderConfig;
344
+ declare function containsGrader(value: string, options?: GraderHelperOptions): ContainsGraderConfig;
345
+ declare function equalsGrader(value: string, options?: GraderHelperOptions): EqualsGraderConfig;
346
+ declare function exactGrader(value: string, options?: GraderHelperOptions): EqualsGraderConfig;
347
+ declare function regexGrader(pattern: string | RegExp, options?: RegexGraderOptions): RegexGraderConfig;
348
+ declare function isJsonGrader(options?: GraderHelperOptions): IsJsonGraderConfig;
349
+ declare function jsonGrader(options?: GraderHelperOptions): IsJsonGraderConfig;
350
+ declare function llmRubricGrader(valueOrCriteria?: string | readonly GraderRubricCriterion[] | Readonly<Record<string, unknown>>, options?: GraderHelperOptions & {
351
+ readonly prompt?: string | GraderPromptScriptConfig;
352
+ readonly provider?: string;
353
+ readonly config?: Readonly<Record<string, unknown>>;
354
+ readonly maxSteps?: number;
355
+ readonly temperature?: number;
356
+ }): LlmRubricGraderConfig;
357
+ /** @deprecated Use scriptGrader. */
358
+ declare function codeGrader(command: GraderCommand, options?: ScriptGraderOptions): ScriptGraderConfig;
359
+ declare function scriptGrader(command: GraderCommand, options?: ScriptGraderOptions): ScriptGraderConfig;
360
+ declare const graders: Readonly<{
361
+ contains: typeof containsGrader;
362
+ equals: typeof equalsGrader;
363
+ exact: typeof exactGrader;
364
+ regex: typeof regexGrader;
365
+ isJson: typeof isJsonGrader;
366
+ json: typeof jsonGrader;
367
+ llmRubric: typeof llmRubricGrader;
368
+ codeGrader: typeof codeGrader;
369
+ script: typeof scriptGrader;
370
+ scriptGrader: typeof scriptGrader;
371
+ }>;
372
+ type GraderCatalog = typeof graders;
373
+
374
+ /**
375
+ * Runtime-only client for invoking configured providers from script-grader scripts
376
+ * through AgentV's provider proxy.
377
+ *
378
+ * Environment variables (set automatically by AgentV when provider proxy access is enabled):
379
+ * - AGENTV_PROVIDER_PROXY_URL: The URL of the local proxy server
380
+ * - AGENTV_PROVIDER_PROXY_TOKEN: Bearer token for authentication
381
+ */
382
+
383
+ /**
384
+ * Request to invoke the provider
385
+ */
386
+ interface ProviderInvokeRequest {
387
+ readonly question: string;
388
+ readonly systemPrompt?: string;
389
+ readonly evalCaseId?: string;
390
+ readonly attempt?: number;
391
+ /** Optional provider override - use a different provider for this invocation */
392
+ readonly provider?: string;
393
+ }
394
+ /**
395
+ * Response from a provider invocation
396
+ */
397
+ interface ProviderInvokeResponse {
398
+ readonly output: readonly unknown[];
399
+ readonly rawText?: string;
400
+ readonly tokenUsage?: TokenUsage;
401
+ }
402
+ /**
403
+ * Information about the provider proxy configuration
404
+ */
405
+ interface ProviderInfo {
406
+ /** Label of the default provider being used */
407
+ readonly providerLabel: string;
408
+ /** Maximum number of calls allowed */
409
+ readonly maxCalls: number;
410
+ /** Current number of calls made */
411
+ readonly callCount: number;
412
+ /** Labels of all available providers */
413
+ readonly availableProviderLabels: readonly string[];
414
+ }
415
+ /**
416
+ * Provider client for making provider invocations
417
+ */
418
+ interface ProviderClient {
419
+ /**
420
+ * Invoke the configured provider with a prompt.
421
+ * @param request - The question and optional system prompt
422
+ * @returns The provider's response with output messages and optional raw text
423
+ */
424
+ invoke(request: ProviderInvokeRequest): Promise<ProviderInvokeResponse>;
425
+ /**
426
+ * Invoke the provider with multiple requests in sequence.
427
+ * Each request counts toward the max_calls limit.
428
+ * @param requests - Array of provider requests
429
+ * @returns Array of provider responses
430
+ */
431
+ invokeBatch(requests: readonly ProviderInvokeRequest[]): Promise<readonly ProviderInvokeResponse[]>;
432
+ /**
433
+ * Get information about the provider proxy configuration.
434
+ * Returns the default provider label, max calls, current call count, and available providers.
435
+ */
436
+ getInfo(): Promise<ProviderInfo>;
437
+ }
438
+ /**
439
+ * Error thrown when provider proxy is not available
440
+ */
441
+ declare class ProviderNotAvailableError extends Error {
442
+ constructor(message: string);
443
+ }
444
+ /**
445
+ * Error thrown when provider invocation fails
446
+ */
447
+ declare class ProviderInvocationError extends Error {
448
+ readonly statusCode?: number;
449
+ constructor(message: string, statusCode?: number);
450
+ }
451
+ /**
452
+ * Create a provider client from environment variables.
453
+ *
454
+ * This function reads the proxy URL and token from environment variables
455
+ * that are automatically set by AgentV when provider access is enabled on a
456
+ * `script` evaluator.
457
+ *
458
+ * @returns A provider client if environment variables are set, otherwise undefined
459
+ * @throws ProviderNotAvailableError if token is missing when URL is present
460
+ *
461
+ * @example
462
+ * ```typescript
463
+ * import { createProviderClient, defineScriptGrader } from 'agentv';
464
+ *
465
+ * export default defineScriptGrader(async ({ input, criteria, output }) => {
466
+ * const provider = createProviderClient();
467
+ * const question = input
468
+ * .filter((message) => message.role === 'user')
469
+ * .map((message) => typeof message.content === 'string' ? message.content : '')
470
+ * .join('\n');
471
+ *
472
+ * if (!provider) {
473
+ * return { pass: false, score: 0.5, reason: 'Provider proxy not available' };
474
+ * }
475
+ *
476
+ * const response = await provider.invoke({
477
+ * question: `Is this answer correct? Question: ${question}, Expected: ${criteria}, Answer: ${output ?? ''}`,
478
+ * systemPrompt: 'You are an expert grader. Respond with JSON: { "correct": true/false }'
479
+ * });
480
+ *
481
+ * const result = JSON.parse(response.rawText ?? '{}');
482
+ * return {
483
+ * pass: result.correct === true,
484
+ * score: result.correct === true ? 1.0 : 0.0,
485
+ * reason: result.correct === true ? 'Provider judged the answer correct' : 'Provider judged the answer incorrect',
486
+ * };
487
+ * });
488
+ * ```
489
+ */
490
+ declare function createProviderClient(): ProviderClient | undefined;
491
+
492
+ interface WorkspaceCheck {
493
+ readonly text: string;
494
+ readonly pass: boolean;
495
+ readonly score?: number;
496
+ readonly reason: string;
497
+ readonly evidence?: string;
498
+ }
499
+ /** @deprecated Use WorkspaceCheck. */
500
+ type WorkspaceAssertion = WorkspaceCheck;
501
+ type Awaitable<T> = T | Promise<T>;
502
+ type WorkspaceGraderReturn = ScriptGraderResult | WorkspaceCheck | readonly Awaitable<WorkspaceCheck>[];
503
+ interface WorkspaceFileAssertionOptions {
504
+ readonly text?: string;
505
+ }
506
+ interface WorkspaceFile {
507
+ readonly path: string;
508
+ readonly absolutePath?: string;
509
+ readText(): Promise<string>;
510
+ exists(options?: WorkspaceFileAssertionOptions): Promise<WorkspaceCheck>;
511
+ contains(expected: string, options?: WorkspaceFileAssertionOptions): Promise<WorkspaceCheck>;
512
+ notContains(expected: string, options?: WorkspaceFileAssertionOptions): Promise<WorkspaceCheck>;
513
+ matches(pattern: RegExp, options?: WorkspaceFileAssertionOptions): Promise<WorkspaceCheck>;
514
+ notMatches(pattern: RegExp, options?: WorkspaceFileAssertionOptions): Promise<WorkspaceCheck>;
515
+ }
516
+ interface Workspace {
517
+ readonly path?: string;
518
+ file(relativePath: string): WorkspaceFile;
519
+ readText(relativePath: string): Promise<string>;
520
+ }
521
+ type WorkspaceGraderContext = ScriptGraderInput & {
522
+ readonly workspace: Workspace;
523
+ };
524
+ type WorkspaceGraderHandler = (context: WorkspaceGraderContext) => WorkspaceGraderReturn | Promise<WorkspaceGraderReturn>;
525
+ declare function createWorkspace(input: ScriptGraderInput): Workspace;
526
+ declare function normalizeWorkspaceGraderResult(result: WorkspaceGraderReturn): Promise<ScriptGraderResult>;
527
+ declare function runWorkspaceGrader(handler: WorkspaceGraderHandler, input: ScriptGraderInput): Promise<ScriptGraderResult>;
528
+ declare function defineWorkspaceGrader(handler: WorkspaceGraderHandler): void;
529
+
530
+ /**
531
+ * Context provided to assertion handlers.
532
+ */
533
+ type AssertionContext = ScriptGraderInput;
534
+ /**
535
+ * Known built-in assertion types. Custom types are extensible via string.
536
+ *
537
+ * Use in EVAL.yaml `assert` blocks:
538
+ * ```yaml
539
+ * assert:
540
+ * - type: contains
541
+ * value: "Paris"
542
+ * ```
543
+ *
544
+ * Custom types registered via `.agentv/assertions/` or `defineAssertion()`
545
+ * are also valid — the `string & {}` escape hatch provides autocomplete
546
+ * for known types while accepting any string.
547
+ */
548
+ type AssertionType = 'llm-rubric' | 'agent-rubric' | 'script' | 'assert-set'
549
+ /** @deprecated Authored eval YAML rejects this compatibility-only runtime type. */
550
+ | 'tool-trajectory' | 'skill-used' | 'not-skill-used' | 'trajectory:tool-used' | 'trajectory:tool-args-match' | 'trajectory:tool-sequence' | 'trajectory:step-count' | 'trajectory:goal-success' | 'field-accuracy' | 'latency' | 'cost' | 'token-usage' | 'execution-metrics' | 'contains' | 'contains-any' | 'contains-all' | 'icontains' | 'icontains-any' | 'icontains-all' | 'starts-with' | 'ends-with' | 'equals' | 'regex' | 'is-json' | 'javascript' | 'python' | 'webhook' | 'similar' | (string & {});
551
+ /**
552
+ * Check returned from an assertion handler.
553
+ */
554
+ interface AssertionCheck {
555
+ readonly id?: string;
556
+ readonly text: string;
557
+ readonly pass: boolean;
558
+ readonly score?: number;
559
+ readonly reason: string;
560
+ readonly evidence?: string;
561
+ }
562
+ /**
563
+ * Result returned from an assertion handler.
564
+ *
565
+ * @example Pass with score
566
+ * ```ts
567
+ * { pass: true, score: 1, reason: 'Output contains expected keywords' }
568
+ * ```
569
+ *
570
+ * @example Fail with checks
571
+ * ```ts
572
+ * { pass: false, score: 0.3, reason: 'Missing required header', checks: [
573
+ * { text: 'Header present', pass: false, reason: 'No header found' },
574
+ * ] }
575
+ * ```
576
+ *
577
+ * @example Granular score (0-1)
578
+ * ```ts
579
+ * { score: 0.75, reason: 'Two of three checks passed', checks: [
580
+ * { text: 'Format correct', pass: true, reason: 'Matches expected format' },
581
+ * { text: 'Content relevant', pass: true, reason: 'Addresses the request' },
582
+ * { text: 'Citation present', pass: false, reason: 'Missing citation' },
583
+ * ] }
584
+ * ```
585
+ */
586
+ interface AssertionScore {
587
+ /** Explicit pass/fail. If omitted, derived from score (>= 0.5 = pass). */
588
+ readonly pass?: boolean;
589
+ /** Numeric score between 0 and 1. Defaults to 1 if pass=true, 0 if pass=false. */
590
+ readonly score?: number;
591
+ /** Explanation for the aggregate pass/fail decision. */
592
+ readonly reason?: string;
593
+ /** Per-check verdicts with optional score and evidence. */
594
+ readonly checks?: readonly AssertionCheck[];
595
+ /** Optional structured details for domain-specific metrics. */
596
+ readonly details?: Record<string, unknown>;
597
+ }
598
+ /**
599
+ * Handler function type for assertions.
600
+ */
601
+ type AssertionHandler = (ctx: AssertionContext) => AssertionScore | Promise<AssertionScore>;
602
+
603
+ /**
604
+ * Handler function type for prompt templates.
605
+ * Returns the prompt string to use for evaluation.
606
+ */
607
+ type PromptTemplateHandler = (input: ScriptGraderInput) => string | Promise<string>;
608
+
609
+ /**
610
+ * AgentV Evaluation SDK
611
+ *
612
+ * Build custom assertions, script graders, and eval authoring helpers for AI agent outputs.
613
+ *
614
+ * @example Custom assertion (simplest way to add evaluation logic)
615
+ * ```typescript
616
+ * #!/usr/bin/env bun
617
+ * import { defineAssertion } from 'agentv';
618
+ *
619
+ * export default defineAssertion(({ output, criteria }) => {
620
+ * const answer = output ?? '';
621
+ * return {
622
+ * pass: answer.includes('hello'),
623
+ * score: answer.includes('hello') ? 1 : 0,
624
+ * reason: answer.includes('hello') ? 'Greeting found' : 'Greeting missing',
625
+ * };
626
+ * }));
627
+ * ```
628
+ *
629
+ * @example script grader (full control)
630
+ * ```typescript
631
+ * #!/usr/bin/env bun
632
+ * import { defineScriptGrader } from 'agentv';
633
+ *
634
+ * export default defineScriptGrader(({ output, traceSummary }) => {
635
+ * return {
636
+ * score: (output ?? '').length > 0 && (traceSummary?.eventCount ?? 0) <= 5 ? 1.0 : 0.5,
637
+ * pass: (output ?? '').length > 0 && (traceSummary?.eventCount ?? 0) <= 5,
638
+ * reason: 'Checks answer text and trace size',
639
+ * checks: [
640
+ * { text: 'Answer is not empty', pass: (output ?? '').length > 0, reason: 'Output text is present' },
641
+ * { text: 'Efficient tool usage', pass: (traceSummary?.eventCount ?? 0) <= 5, reason: 'Trace event count is within limit' },
642
+ * ],
643
+ * };
644
+ * }));
645
+ * ```
646
+ *
647
+ * @example Vitest workspace verifier adapter (custom wrapper form)
648
+ * ```typescript
649
+ * #!/usr/bin/env bun
650
+ * import { defineVitestWorkspaceGrader } from 'agentv';
651
+ *
652
+ * export default defineVitestWorkspaceGrader({
653
+ * testFile: 'graders/welcome-banner.test.ts',
654
+ * copyTestFilesToWorkspace: true,
655
+ * });
656
+ * ```
657
+ *
658
+ * @example Workspace grader (small file checks)
659
+ * ```typescript
660
+ * #!/usr/bin/env bun
661
+ * import { defineWorkspaceGrader } from 'agentv';
662
+ *
663
+ * export default defineWorkspaceGrader(async ({ workspace }) => [
664
+ * await workspace.file('app/page.tsx').contains('Status: All systems ready'),
665
+ * await workspace.file('app/page.tsx').contains('Open dashboard'),
666
+ * await workspace.file('app/page.tsx').matches(/href=["']\/dashboard["']/),
667
+ * await workspace.file('app/page.tsx').notMatches(/TODO/i),
668
+ * ]);
669
+ * ```
670
+ *
671
+ * @packageDocumentation
672
+ */
673
+
674
+ /**
675
+ * Define a script grader with automatic stdin/stdout handling.
676
+ *
677
+ * This function:
678
+ * 1. Reads JSON from stdin (snake_case format)
679
+ * 2. Converts to camelCase and validates with Zod
680
+ * 3. Calls your handler with typed input
681
+ * 4. Validates the result and outputs JSON to stdout
682
+ * 5. Handles errors gracefully with proper exit codes
683
+ *
684
+ * @param handler - Function that evaluates the input and returns a result
685
+ *
686
+ * @example
687
+ * ```typescript
688
+ * import { defineScriptGrader } from 'agentv';
689
+ *
690
+ * export default defineScriptGrader(({ trace }) => {
691
+ * if (!trace) {
692
+ * return { pass: false, score: 0.5, reason: 'No trace available' };
693
+ * }
694
+ *
695
+ * const efficient = trace.eventCount <= 10;
696
+ * return {
697
+ * pass: efficient,
698
+ * score: efficient ? 1.0 : 0.5,
699
+ * reason: efficient ? 'Efficient execution' : 'Too many tool calls',
700
+ * checks: [{ text: 'Trace event count within limit', pass: efficient, reason: `${trace.eventCount} events observed` }],
701
+ * };
702
+ * });
703
+ * ```
704
+ *
705
+ * @example With typed config
706
+ * ```typescript
707
+ * import { defineScriptGrader, z } from 'agentv';
708
+ *
709
+ * const ConfigSchema = z.object({
710
+ * maxToolCalls: z.number().default(10),
711
+ * });
712
+ *
713
+ * export default defineScriptGrader(({ trace, config }) => {
714
+ * const { maxToolCalls } = ConfigSchema.parse(config ?? {});
715
+ * // Use maxToolCalls...
716
+ * });
717
+ * ```
718
+ */
719
+ declare function defineScriptGrader(handler: ScriptGraderHandler): void;
720
+ /** @deprecated Use defineScriptGrader. */
721
+ declare function defineCodeGrader(handler: ScriptGraderHandler): void;
722
+ /**
723
+ * Define a prompt template with automatic stdin/stdout handling.
724
+ *
725
+ * This function:
726
+ * 1. Reads JSON from stdin (snake_case format)
727
+ * 2. Converts to camelCase and validates with Zod
728
+ * 3. Calls your handler with typed input
729
+ * 4. Outputs the generated prompt string to stdout
730
+ * 5. Handles errors gracefully with proper exit codes
731
+ *
732
+ * @param handler - Function that generates the prompt string from input
733
+ *
734
+ * @example
735
+ * ```typescript
736
+ * import { definePromptTemplate } from 'agentv';
737
+ *
738
+ * export default definePromptTemplate((ctx) => {
739
+ * const question = ctx.input
740
+ * .filter((message) => message.role === 'user')
741
+ * .map((message) => typeof message.content === 'string' ? message.content : '')
742
+ * .join('\n');
743
+ * const answer = ctx.output ?? '';
744
+ * return `Question: ${question}\nAnswer: ${answer}`;
745
+ * });
746
+ * ```
747
+ */
748
+ declare function definePromptTemplate(handler: PromptTemplateHandler): void;
749
+ /**
750
+ * Define a custom assertion with automatic stdin/stdout handling.
751
+ *
752
+ * Assertions are the simplest way to add reusable custom checks. They receive
753
+ * the full evaluation context and return a pass/fail result with optional
754
+ * granular scoring. Place these files in `.agentv/assertions/` and reference
755
+ * them by discovered assertion type name. Use defineScriptGrader for
756
+ * command-backed graders referenced with `type: script`.
757
+ *
758
+ * This function:
759
+ * 1. Reads JSON from stdin (snake_case format)
760
+ * 2. Converts to camelCase and validates with Zod
761
+ * 3. Calls your handler with typed context
762
+ * 4. Normalizes the result (pass→score, clamp, etc.)
763
+ * 5. Outputs JSON to stdout
764
+ * 6. Handles errors gracefully with proper exit codes
765
+ *
766
+ * @param handler - Function that evaluates the context and returns a result
767
+ *
768
+ * @example Simple pass/fail
769
+ * ```typescript
770
+ * import { defineAssertion } from 'agentv';
771
+ *
772
+ * export default defineAssertion(({ output }) => {
773
+ * const text = output ?? '';
774
+ * return {
775
+ * pass: text.toLowerCase().includes('hello'),
776
+ * reason: text.toLowerCase().includes('hello') ? 'Greeting found' : 'Greeting missing',
777
+ * };
778
+ * }));
779
+ * ```
780
+ *
781
+ * @example Granular scoring
782
+ * ```typescript
783
+ * import { defineAssertion } from 'agentv';
784
+ *
785
+ * export default defineAssertion(({ output, traceSummary }) => {
786
+ * const text = output ?? '';
787
+ * const hasContent = text.length > 0 ? 0.5 : 0;
788
+ * const isEfficient = (traceSummary?.eventCount ?? 0) <= 5 ? 0.5 : 0;
789
+ * return {
790
+ * score: hasContent + isEfficient,
791
+ * reason: 'Checks content exists and trace size',
792
+ * checks: [
793
+ * { text: 'Has content', pass: !!hasContent, reason: hasContent ? 'Output is non-empty' : 'Output is empty' },
794
+ * { text: 'Efficient', pass: !!isEfficient, reason: isEfficient ? 'Trace is within limit' : 'Trace exceeds limit' },
795
+ * ],
796
+ * };
797
+ * }));
798
+ * ```
799
+ */
800
+ declare function defineAssertion(handler: AssertionHandler): void;
801
+
802
+ export { type AssertionCheck, type AssertionContext, type AssertionHandler, type AssertionScore, type AssertionType, type CodeGraderConfig, type CodeGraderOptions, type CodeGraderTargetOptions, type ContainsGraderConfig, type DefinedEvalSuite, type EqualsGraderConfig, type EvalAssertionConfig, type EvalConfig, type EvalDefaultTest, type EvalDockerEnvironment, type EvalDockerEnvironmentMount, type EvalDockerEnvironmentResources, type EvalEnvironment, type EvalEnvironmentSetup, type EvalExecution, type EvalHostEnvironment, type EvalLifecycleHook, type EvalLifecycleHooks, type EvalMessage, type EvalMessageContent, type EvalProviderConfig, type EvalProviderEntry, type EvalProviderMap, type EvalProviderRef, type EvalRepeat, type EvalRequires, type EvalTest, type EvalTestOptions, type EvalTurn, type GraderCatalog, type GraderCommand, type GraderCommonConfig, type GraderHelperConfig, type GraderHelperOptions, type GraderPromptScriptConfig, type GraderRubric, type GraderRubricCriterion, type GraderRubricOperator, type GraderScoreRange, type IsJsonGraderConfig, type LlmRubricGraderConfig, type LowerEvalYamlValue, type PromptTemplateHandler, type ProviderClient, type ProviderInfo, ProviderInvocationError, type ProviderInvokeRequest, type ProviderInvokeResponse, ProviderNotAvailableError, type RegexGraderConfig, type RegexGraderOptions, type ScriptGraderConfig, type ScriptGraderOptions, type ScriptGraderProviderOptions, type ScriptGraderTargetOptions, type Workspace, type WorkspaceAssertion, type WorkspaceCheck, type WorkspaceFile, type WorkspaceFileAssertionOptions, type WorkspaceGraderContext, type WorkspaceGraderHandler, type WorkspaceGraderReturn, codeGrader, containsGrader, createProviderClient, createWorkspace, defineAssertion, defineCodeGrader, defineEval, definePromptTemplate, defineScriptGrader, defineWorkspaceGrader, equalsGrader, exactGrader, graders, isJsonGrader, jsonGrader, llmRubricGrader, normalizeWorkspaceGraderResult, regexGrader, runWorkspaceGrader, scriptGrader, serializeEvalYaml, toEvalYamlObject };