agentv 5.3.0-next.1 → 5.3.2-next.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +70 -60
- package/dist/{artifact-writer-JFNIPMKW.js → artifact-writer-KJEOROKQ.js} +5 -5
- package/dist/{chunk-6ZZCDZPD.js → chunk-6262OKXM.js} +21774 -21034
- package/dist/chunk-6262OKXM.js.map +1 -0
- package/dist/chunk-AQ5BIAXF.js +604 -0
- package/dist/chunk-AQ5BIAXF.js.map +1 -0
- package/dist/chunk-BV5VQLI2.js +2 -0
- package/dist/{chunk-V52ATPTT.js → chunk-JGSRUJZQ.js} +186 -32
- package/dist/chunk-JGSRUJZQ.js.map +1 -0
- package/dist/chunk-MMDLXYBX.js +721 -0
- package/dist/chunk-MMDLXYBX.js.map +1 -0
- package/dist/{chunk-FVR4RQFK.js → chunk-QSK3PC44.js} +1131 -580
- package/dist/chunk-QSK3PC44.js.map +1 -0
- package/dist/chunk-TEEXVJWM.js +2 -0
- package/dist/{chunk-BHKQHG26.js → chunk-WIAJ7JAV.js} +186 -64
- package/dist/chunk-WIAJ7JAV.js.map +1 -0
- package/dist/cli.d.ts +1 -0
- package/dist/cli.js +17517 -10
- package/dist/cli.js.map +1 -1
- package/dist/config.d.ts +2 -0
- package/dist/{ts-eval-loader-DQDYRULE-V3377FOL.js → config.js} +7 -7
- package/dist/config.js.map +1 -0
- package/dist/contracts-DsZmZLl8.d.ts +742 -0
- package/dist/contracts.d.ts +2 -0
- package/dist/contracts.js +28 -0
- package/dist/contracts.js.map +1 -0
- package/dist/dashboard/assets/{index-DNgf3qJ2.js → index-BfqOlLOF.js} +1 -1
- package/dist/dashboard/assets/index-Bh8lpGce.css +1 -0
- package/dist/dashboard/assets/index-oPR3ywQb.js +121 -0
- package/dist/dashboard/index.html +2 -2
- package/dist/{dist-6Z7U473R.js → dist-A3SGR7TW.js} +30 -18
- package/dist/dist-A3SGR7TW.js.map +1 -0
- package/dist/index.d.ts +4 -0
- package/dist/index.js +137 -18
- package/dist/{interactive-RUY3OCBI.js → interactive-DL7N2C7K.js} +24 -24
- package/dist/interactive-DL7N2C7K.js.map +1 -0
- package/dist/provider.d.ts +2 -0
- package/dist/provider.js +24 -0
- package/dist/provider.js.map +1 -0
- package/dist/sdk.d.ts +802 -0
- package/dist/sdk.js +145 -0
- package/dist/sdk.js.map +1 -0
- package/dist/skills/agentv-bench/SKILL.md +14 -13
- package/dist/skills/agentv-bench/agents/analyzer.md +1 -1
- package/dist/skills/agentv-bench/agents/executor.md +1 -1
- package/dist/skills/agentv-bench/references/autoresearch.md +9 -9
- package/dist/skills/agentv-bench/references/environment-adaptation.md +4 -4
- package/dist/skills/agentv-bench/references/eval-yaml-spec.md +30 -47
- package/dist/skills/agentv-bench/references/schemas.md +44 -60
- package/dist/skills/agentv-bench/references/subagent-pipeline.md +20 -18
- package/dist/skills/agentv-eval-migrations/SKILL.md +13 -0
- package/dist/skills/agentv-eval-migrations/references/breaking-changes.md +56 -39
- package/dist/skills/agentv-eval-writer/SKILL.md +101 -60
- package/dist/skills/agentv-eval-writer/references/custom-evaluators.md +15 -10
- package/dist/skills/agentv-eval-writer/references/eval.schema.json +4543 -5189
- package/dist/skills/agentv-eval-writer/references/python-helpers.md +2 -2
- package/dist/skills/agentv-eval-writer/references/rubric-evaluator.md +20 -3
- package/dist/templates/.agentv/providers.yaml +42 -0
- package/dist/templates/.env.example +2 -2
- package/dist/ts-eval-loader-3G5GEAEC-6F52JLPP.js +18 -0
- package/dist/ts-eval-loader-3G5GEAEC-6F52JLPP.js.map +1 -0
- package/package.json +29 -4
- package/dist/chunk-6ZZCDZPD.js.map +0 -1
- package/dist/chunk-BHKQHG26.js.map +0 -1
- package/dist/chunk-FVR4RQFK.js.map +0 -1
- package/dist/chunk-T32NL3E6.js +0 -17973
- package/dist/chunk-T32NL3E6.js.map +0 -1
- package/dist/chunk-V52ATPTT.js.map +0 -1
- package/dist/dashboard/assets/index-D_bokML8.css +0 -1
- package/dist/dashboard/assets/index-r_jSJmlw.js +0 -121
- package/dist/interactive-RUY3OCBI.js.map +0 -1
- package/dist/templates/.agentv/targets.yaml +0 -97
- /package/dist/{artifact-writer-JFNIPMKW.js.map → artifact-writer-KJEOROKQ.js.map} +0 -0
- /package/dist/{dist-6Z7U473R.js.map → chunk-BV5VQLI2.js.map} +0 -0
- /package/dist/{ts-eval-loader-DQDYRULE-V3377FOL.js.map → chunk-TEEXVJWM.js.map} +0 -0
|
@@ -20,14 +20,14 @@ Promptfoo parity matrix: https://agentv.dev/docs/reference/promptfoo-parity/
|
|
|
20
20
|
Treat YAML as the canonical portable model. Prefer authoring `.eval.yaml` / `EVAL.yaml` first, then use TypeScript helpers, Python scripts, or executable graders only when they lower to the same fields or when the evaluation logic must actually run code.
|
|
21
21
|
|
|
22
22
|
Eval files define what is tested and how it runs: prompts, datasets, assertions,
|
|
23
|
-
task fixtures, top-level `
|
|
23
|
+
task fixtures, top-level `providers`, and suite run controls. Use field-local file
|
|
24
24
|
refs such as `tests: file://...`, `prompts: file://...`, `default_test:
|
|
25
25
|
file://...`, and `environment: file://...`. String-valued `tests` and string
|
|
26
26
|
entries inside `tests[]` are raw-case refs for direct paths, directories, and
|
|
27
27
|
globs. Run several full eval suites directly with CLI multi-file selection and
|
|
28
28
|
tags. Use scoped `run:` on individual tests only for `threshold`, `repeat`,
|
|
29
|
-
`timeout_seconds`, and legacy `budget_usd`; keep
|
|
30
|
-
`
|
|
29
|
+
`timeout_seconds`, and legacy `budget_usd`; keep provider selection at top-level
|
|
30
|
+
`providers` or CLI `--provider`, put suite budget caps under
|
|
31
31
|
`evaluate_options.budget_usd`, authored concurrency under
|
|
32
32
|
`evaluate_options.max_concurrency`, suite repeat policy under
|
|
33
33
|
`evaluate_options.repeat`, coding-agent testbed setup under `environment`,
|
|
@@ -39,8 +39,8 @@ Use `@agentv/sdk` for TypeScript helper imports. Do not use `@agentv/eval` for n
|
|
|
39
39
|
## Authoring Checklist
|
|
40
40
|
|
|
41
41
|
- Put grading criteria in `assert`, not in test-level `criteria`. Plain assertion strings become an `llm-rubric` grader.
|
|
42
|
-
- Prefer plain assertion strings for semantic checks when the default rubric grader can judge them. Use `type: llm-rubric` for structured criteria, custom prompts, custom grader
|
|
43
|
-
-
|
|
42
|
+
- Prefer plain assertion strings for semantic checks when the default rubric grader can judge them. Use `type: llm-rubric` for structured criteria, custom prompts, custom grader providers, or assertion-level transforms. Use `type: agent-rubric` when the grader itself must be an agent-capable provider that can inspect the workspace. Use `type: script` when grading must execute code.
|
|
43
|
+
- Put reference answers in `tests[].vars.expected_output` or `default_test.vars.expected_output`, and consume them with an explicit assertion such as `type: llm-rubric` with `value: "Matches the reference answer: {{ expected_output }}"`. Do not write criteria, scoring instructions, or "the agent should..." rubric prose as the reference answer.
|
|
44
44
|
- For historical or repo-state evals, materialize the repo through a pinned `environment` setup recipe. Mentioning a SHA only in prompt prose is not enough because the agent needs an actual checkout to inspect.
|
|
45
45
|
|
|
46
46
|
## Evaluation Types
|
|
@@ -61,7 +61,12 @@ agentv convert evals.json
|
|
|
61
61
|
agentv eval evals.json
|
|
62
62
|
```
|
|
63
63
|
|
|
64
|
-
The converter maps
|
|
64
|
+
The converter maps Agent Skills prompt text into AgentV prompt/vars data,
|
|
65
|
+
promotes Agent Skills `expected_output` into explicit `llm-rubric` criteria or
|
|
66
|
+
`vars.expected_output` when it is true reference data, and maps Agent Skills
|
|
67
|
+
`assertions` to AgentV `assert` (`llm-rubric` checks). The generated YAML
|
|
68
|
+
includes TODO comments for AgentV features to add (environment setup, script
|
|
69
|
+
graders, rubrics, required gates).
|
|
65
70
|
|
|
66
71
|
After converting, enhance the YAML with AgentV-specific capabilities shown below.
|
|
67
72
|
|
|
@@ -103,18 +108,22 @@ prompts:
|
|
|
103
108
|
|
|
104
109
|
tests:
|
|
105
110
|
- id: multi-turn-context
|
|
106
|
-
|
|
111
|
+
vars:
|
|
112
|
+
expected_output: "Your name is Alice."
|
|
107
113
|
assert:
|
|
114
|
+
- type: llm-rubric
|
|
115
|
+
value: "Matches the reference answer: {{ expected_output }}"
|
|
108
116
|
- Correctly recalls the user's name from earlier in the conversation
|
|
109
117
|
```
|
|
110
118
|
|
|
111
|
-
**Guidelines:** preserve exact wording in `expected_output`; aim for 5–15 tests per transcript; pick exchanges that test different capabilities.
|
|
119
|
+
**Guidelines:** preserve exact wording in `vars.expected_output`; aim for 5–15 tests per transcript; pick exchanges that test different capabilities.
|
|
112
120
|
|
|
113
121
|
## Quick Start
|
|
114
122
|
|
|
115
123
|
```yaml
|
|
116
124
|
description: Example eval
|
|
117
|
-
|
|
125
|
+
providers:
|
|
126
|
+
- default
|
|
118
127
|
|
|
119
128
|
prompts:
|
|
120
129
|
- "{{ prompt }}"
|
|
@@ -123,8 +132,10 @@ tests:
|
|
|
123
132
|
- id: greeting
|
|
124
133
|
vars:
|
|
125
134
|
prompt: "Say hello"
|
|
126
|
-
|
|
135
|
+
expected_output: "Hello! How can I help you?"
|
|
127
136
|
assert:
|
|
137
|
+
- type: llm-rubric
|
|
138
|
+
value: "Matches the reference answer: {{ expected_output }}"
|
|
128
139
|
- Greeting is friendly and warm
|
|
129
140
|
- Offers to help
|
|
130
141
|
```
|
|
@@ -132,7 +143,7 @@ tests:
|
|
|
132
143
|
## Eval File Structure
|
|
133
144
|
|
|
134
145
|
**Required:** `tests` (array or string raw-case path) or `scenarios`
|
|
135
|
-
**Optional:** `name`, `description`, `
|
|
146
|
+
**Optional:** `name`, `description`, `version`, `author`, `tags`, `license`, `requires`, `providers`, `prompts`, `default_test`, `timeout_seconds`, `evaluate_options`, `threshold`, `suite`, `environment`, `env`, `extensions`, `assert`
|
|
136
147
|
|
|
137
148
|
**Test fields:**
|
|
138
149
|
|
|
@@ -140,24 +151,35 @@ tests:
|
|
|
140
151
|
|-------|----------|-------------|
|
|
141
152
|
| `id` | yes | Unique identifier |
|
|
142
153
|
| `vars` | yes when the prompt needs row data | Prompt-template variables for this row |
|
|
143
|
-
| `expected_output` | no |
|
|
144
|
-
| `assert` | yes | Graders: deterministic checks, `llm-rubric` checks, script graders, or plain string rubric criteria |
|
|
145
|
-
| `execution` | no | Per-case grader/default overrides such as `skip_defaults`;
|
|
154
|
+
| `vars.expected_output` | no | Conventional reference-answer var consumed by explicit graders |
|
|
155
|
+
| `assert` | yes | Graders: deterministic checks, `llm-rubric` / `agent-rubric` checks, script graders, or plain string rubric criteria |
|
|
156
|
+
| `execution` | no | Per-case grader/default overrides such as `skip_defaults`; provider selection belongs in top-level `providers` or CLI `--provider` |
|
|
146
157
|
| `environment` | no | Per-case coding-agent testbed config (overrides suite-level) |
|
|
147
158
|
| `metadata` | no | Arbitrary key-value pairs passed to setup/teardown scripts |
|
|
148
159
|
| `conversation_id` | no | Thread grouping |
|
|
149
160
|
|
|
161
|
+
**Provider declarations:** AgentV accepts Promptfoo-shaped provider entries
|
|
162
|
+
where they map cleanly: strings such as `openai:gpt-4.1-mini`, object form with
|
|
163
|
+
`id`, `label`, `config`, `env`, `prompts`, `transform`, `delay`, and `inputs`,
|
|
164
|
+
and provider maps such as `{ "openai:gpt-4": { label, config } }`. In object
|
|
165
|
+
form, `id` is the backend/spec and `label` is the stable AgentV selection and
|
|
166
|
+
result identity. AgentV-only `runtime`, provider-local `environment`, and
|
|
167
|
+
provider `hooks` must be explicit extensions; direct Promptfoo runs do not
|
|
168
|
+
execute them, and export must lower supported cases or reject unsupported ones
|
|
169
|
+
clearly.
|
|
170
|
+
|
|
150
171
|
## Prompt Templates and Vars
|
|
151
172
|
|
|
152
173
|
Use top-level `prompts` plus `tests[].vars` for the Promptfoo-compatible canonical
|
|
153
174
|
input shape. Shared data defaults belong in `default_test.vars`; per-test
|
|
154
175
|
`vars` override those defaults by key. AgentV renders every prompt with each
|
|
155
176
|
test's merged vars, then expands the run across prompts, targets, tests, and
|
|
156
|
-
repeat
|
|
177
|
+
repeat samples.
|
|
157
178
|
|
|
158
179
|
```yaml
|
|
159
180
|
description: Prompt matrix example
|
|
160
|
-
|
|
181
|
+
providers:
|
|
182
|
+
- default
|
|
161
183
|
|
|
162
184
|
prompts:
|
|
163
185
|
- id: support-chat
|
|
@@ -176,15 +198,19 @@ tests:
|
|
|
176
198
|
- id: password-reset
|
|
177
199
|
vars:
|
|
178
200
|
question: How do I reset my password?
|
|
179
|
-
|
|
201
|
+
expected_output: Password reset guidance
|
|
180
202
|
assert:
|
|
203
|
+
- type: llm-rubric
|
|
204
|
+
value: "Matches the reference answer: {{ expected_output }}"
|
|
181
205
|
- Gives correct password reset guidance
|
|
182
206
|
- id: admin-access
|
|
183
207
|
vars:
|
|
184
208
|
audience: admins
|
|
185
209
|
question: How do I revoke a user's access?
|
|
186
|
-
|
|
210
|
+
expected_output: Access revocation guidance
|
|
187
211
|
assert:
|
|
212
|
+
- type: llm-rubric
|
|
213
|
+
value: "Matches the reference answer: {{ expected_output }}"
|
|
188
214
|
- Gives safe access revocation guidance
|
|
189
215
|
```
|
|
190
216
|
|
|
@@ -204,7 +230,7 @@ then render those vars from the prompt template next to the input.
|
|
|
204
230
|
**Shorthand forms:**
|
|
205
231
|
- Prompt entries can be strings, message arrays, file references, or prompt objects.
|
|
206
232
|
- Put chat/system/user messages in `prompts`, not in `tests[].input`.
|
|
207
|
-
- `expected_output`
|
|
233
|
+
- `vars.expected_output` is a conventional reference-answer variable; explicit assertions decide how to grade it
|
|
208
234
|
- Use these canonical field names on disk; keep the wire format `snake_case`
|
|
209
235
|
|
|
210
236
|
**Message format:** `{role, content}` where role is `system`, `user`, `assistant`, or `tool`
|
|
@@ -356,12 +382,14 @@ tests:
|
|
|
356
382
|
verification guidance was added.
|
|
357
383
|
|
|
358
384
|
Decide what durable repo change should be made and explain why.
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
385
|
+
expected_output: |
|
|
386
|
+
The durable repo change is to update .agents/verification.md with the
|
|
387
|
+
reusable verification workflow lessons. AGENTS.md already routes this
|
|
388
|
+
class of work to .agents/verification.md, so no extra AGENTS.md edit is
|
|
389
|
+
needed unless that routing is missing.
|
|
364
390
|
assert:
|
|
391
|
+
- type: llm-rubric
|
|
392
|
+
value: "Matches the reference answer: {{ expected_output }}"
|
|
365
393
|
- The answer recommends updating .agents/verification.md rather than leaving the learning only in PR comments or private evidence.
|
|
366
394
|
- The answer uses the pinned ./agentv checkout to verify the AGENTS.md routing.
|
|
367
395
|
- The answer preserves the historical commit SHA as context.
|
|
@@ -371,7 +399,7 @@ tests:
|
|
|
371
399
|
|
|
372
400
|
When `assert` is defined, **only the declared graders run**. For
|
|
373
401
|
semantic checks, add plain rubric strings. If you need a custom LLM prompt or
|
|
374
|
-
grader
|
|
402
|
+
grader provider, declare `llm-rubric` explicitly:
|
|
375
403
|
|
|
376
404
|
```yaml
|
|
377
405
|
prompts:
|
|
@@ -387,9 +415,9 @@ tests:
|
|
|
387
415
|
value: "fix"
|
|
388
416
|
```
|
|
389
417
|
|
|
390
|
-
`expected_output` is passive reference data. It is available to graders
|
|
391
|
-
`{{expected_output}}` and the script stdin payload, but it does not
|
|
392
|
-
implicit LLM grading call by itself.
|
|
418
|
+
`vars.expected_output` is passive reference data. It is available to graders
|
|
419
|
+
through `{{ expected_output }}` and the script stdin payload, but it does not
|
|
420
|
+
create an implicit LLM grading call by itself.
|
|
393
421
|
|
|
394
422
|
**Common mistake:** putting rubric prose in `expected_output` instead of an
|
|
395
423
|
assertion:
|
|
@@ -402,7 +430,7 @@ tests:
|
|
|
402
430
|
- id: bad-example
|
|
403
431
|
vars:
|
|
404
432
|
prompt: "What is 2+2?"
|
|
405
|
-
|
|
433
|
+
expected_output: The assistant should explain why the answer is 4. # reference answer var, not a grader
|
|
406
434
|
```
|
|
407
435
|
|
|
408
436
|
Write this as:
|
|
@@ -415,8 +443,10 @@ tests:
|
|
|
415
443
|
- id: good-example
|
|
416
444
|
vars:
|
|
417
445
|
prompt: "What is 2+2?"
|
|
418
|
-
|
|
446
|
+
expected_output: "4"
|
|
419
447
|
assert:
|
|
448
|
+
- type: llm-rubric
|
|
449
|
+
value: "Matches the reference answer: {{ expected_output }}"
|
|
420
450
|
- The answer is 4 and explains the arithmetic briefly
|
|
421
451
|
```
|
|
422
452
|
|
|
@@ -438,7 +468,7 @@ assert:
|
|
|
438
468
|
weight: 5.0
|
|
439
469
|
```
|
|
440
470
|
|
|
441
|
-
If a required grader scores below its threshold, the overall
|
|
471
|
+
If a required grader scores below its threshold, the overall case status is forced to `fail`.
|
|
442
472
|
|
|
443
473
|
## Environment Setup/Teardown
|
|
444
474
|
|
|
@@ -521,11 +551,12 @@ Configure via the `assert` array. Multiple graders produce a weighted average sc
|
|
|
521
551
|
type: script
|
|
522
552
|
command: [uv, run, validate.py]
|
|
523
553
|
cwd: ./scripts # optional working directory
|
|
524
|
-
|
|
554
|
+
provider: {} # optional: enable LLM provider proxy (max_calls: 50)
|
|
525
555
|
```
|
|
526
|
-
Contract: stdin JSON -> stdout JSON `{score,
|
|
556
|
+
Contract: stdin JSON -> stdout JSON `{pass, score, reason, checks?: [{text, pass, score?, reason}]}`
|
|
527
557
|
Raw stdin uses snake_case and includes: `input`, `expected_output`, `output` (final answer string), `messages`, `trace`, `trace_summary`, `token_usage`, `cost_usd`, `duration_ms`, `start_time`, `end_time`, `file_changes`, `workspace_path`, `config`
|
|
528
558
|
SDK handlers receive the same payload in camelCase: `expectedOutput`, `traceSummary`, `tokenUsage`, `costUsd`, `durationMs`, `startTime`, `endTime`, `fileChanges`, `workspacePath`.
|
|
559
|
+
`checks` is an SDK/script convenience shape; public `grading.json` artifacts normalize checks into recursive `component_results`.
|
|
529
560
|
When an environment prepares a workspace directory, `workspace_path` is the absolute path to that directory (also available as `AGENTV_WORKSPACE_PATH` env var). Use this for functional grading (e.g., running `npm test` in the prepared workdir).
|
|
530
561
|
For deterministic workspace checks that fit normal Vitest `expect(...)` tests, prefer a plain verifier file and the built-in adapter:
|
|
531
562
|
```yaml
|
|
@@ -541,7 +572,7 @@ See the Script Graders docs for the full stdin/stdout contract.
|
|
|
541
572
|
- name: quality
|
|
542
573
|
type: llm-rubric
|
|
543
574
|
prompt: ./prompts/eval.md # markdown template or command config
|
|
544
|
-
|
|
575
|
+
provider: grader_gpt_5_mini # optional: override the grader provider for this grader
|
|
545
576
|
model: gpt-5-chat # optional model override
|
|
546
577
|
config: # passed to prompt templates as context.config
|
|
547
578
|
strictness: high
|
|
@@ -549,7 +580,7 @@ See the Script Graders docs for the full stdin/stdout contract.
|
|
|
549
580
|
Variables: `{{criteria}}`, `{{input}}`, `{{expected_output}}`, `{{output}}`, `{{metadata}}`, `{{metadata_json}}`, `{{rubrics}}`, `{{rubrics_json}}`, `{{file_changes}}`, `{{tool_calls}}`
|
|
550
581
|
- Markdown templates: use `{{variable}}` syntax
|
|
551
582
|
- TypeScript templates: use `definePromptTemplate(fn)` from `@agentv/sdk`, receives context object with all variables + `config`
|
|
552
|
-
- Use `
|
|
583
|
+
- Use `provider:` to run different `llm-rubric` graders against different named LLM providers in the same eval (useful for grader panels / ensembles)
|
|
553
584
|
|
|
554
585
|
### assert-set
|
|
555
586
|
```yaml
|
|
@@ -567,7 +598,7 @@ Variables: `{{criteria}}`, `{{input}}`, `{{expected_output}}`, `{{output}}`, `{{
|
|
|
567
598
|
type: llm-rubric
|
|
568
599
|
weight: 0.7
|
|
569
600
|
```
|
|
570
|
-
Use `assert-set` for Promptfoo-aligned assertion grouping. Without `threshold`, the set passes only when every nonzero-weight child assertion passes. With `threshold`, the weighted aggregate score determines the set
|
|
601
|
+
Use `assert-set` for Promptfoo-aligned assertion grouping. Without `threshold`, the set passes only when every nonzero-weight child assertion passes. With `threshold`, the weighted aggregate score determines the set pass/fail result. Parent `config` is inherited by children, and child `config` keys override parent keys. Do not use `type: composite`; AgentV rejects it.
|
|
571
602
|
|
|
572
603
|
### Skill And Trajectory Assertions
|
|
573
604
|
```yaml
|
|
@@ -722,24 +753,24 @@ agentv eval assert <grader-name> --agent-output "..." --agent-input "..."
|
|
|
722
753
|
agentv import claude --session-id <uuid>
|
|
723
754
|
|
|
724
755
|
# Re-run only execution errors from a previous run
|
|
725
|
-
agentv eval <file.yaml> --retry-errors .agentv/results/
|
|
756
|
+
agentv eval <file.yaml> --retry-errors .agentv/results/<run_id>/.internal/index.jsonl
|
|
726
757
|
|
|
727
758
|
# Validate eval file
|
|
728
759
|
agentv validate <file.yaml>
|
|
729
760
|
|
|
730
761
|
# Compare completed runs
|
|
731
762
|
agentv results compare \
|
|
732
|
-
.agentv/results
|
|
733
|
-
.agentv/results
|
|
763
|
+
.agentv/results/<baseline-run-id>/.internal/index.jsonl \
|
|
764
|
+
.agentv/results/<candidate-run-id>/.internal/index.jsonl
|
|
734
765
|
agentv results combine \
|
|
735
|
-
.agentv/results
|
|
736
|
-
.agentv/results
|
|
737
|
-
.agentv/results
|
|
738
|
-
--output .agentv/results/
|
|
739
|
-
agentv results compare .agentv/results/
|
|
766
|
+
.agentv/results/<baseline-run-id> \
|
|
767
|
+
.agentv/results/<candidate-run-id> \
|
|
768
|
+
.agentv/results/<third-target-run-id> \
|
|
769
|
+
--output .agentv/results/combined
|
|
770
|
+
agentv results compare .agentv/results/combined/.internal/index.jsonl
|
|
740
771
|
agentv results compare \
|
|
741
|
-
.agentv/results
|
|
742
|
-
.agentv/results
|
|
772
|
+
.agentv/results/<baseline-run-id>/.internal/index.jsonl \
|
|
773
|
+
.agentv/results/<candidate-run-id>/.internal/index.jsonl \
|
|
743
774
|
--json
|
|
744
775
|
|
|
745
776
|
# Author assertions directly in the eval file
|
|
@@ -755,32 +786,36 @@ Use `@agentv/sdk` as the public lightweight SDK package for TypeScript/JavaScrip
|
|
|
755
786
|
|
|
756
787
|
### YAML-aligned eval authoring
|
|
757
788
|
```typescript
|
|
758
|
-
|
|
789
|
+
// evals/helper-suite.eval.ts
|
|
790
|
+
import { graders, type EvalConfig } from '@agentv/sdk';
|
|
759
791
|
|
|
760
|
-
|
|
792
|
+
const config: EvalConfig = {
|
|
761
793
|
name: 'helper-suite',
|
|
762
794
|
target: 'default',
|
|
763
|
-
//
|
|
795
|
+
// TypeScript config loading lowers this to evaluate_options.repeat.
|
|
764
796
|
repeat: {
|
|
765
797
|
count: 3,
|
|
766
798
|
strategy: 'pass_any',
|
|
767
799
|
earlyExit: false,
|
|
768
800
|
},
|
|
769
801
|
threshold: 0.8,
|
|
802
|
+
prompts: ['{{ task }}'],
|
|
770
803
|
tests: [
|
|
771
804
|
{
|
|
772
805
|
id: 'json-answer',
|
|
773
|
-
|
|
806
|
+
vars: { task: 'Return a JSON answer with a status field.' },
|
|
774
807
|
assert: [
|
|
775
808
|
graders.json({ name: 'valid-json', required: true }),
|
|
776
809
|
graders.regex(/"status"\s*:/, { name: 'status-key' }),
|
|
777
810
|
],
|
|
778
811
|
},
|
|
779
812
|
],
|
|
780
|
-
}
|
|
813
|
+
};
|
|
814
|
+
|
|
815
|
+
export default config;
|
|
781
816
|
```
|
|
782
817
|
|
|
783
|
-
The `graders` catalog returns ordinary `assert` entries such as `type: is-json`, `type: regex`, `type: llm-rubric`, and `type: script`. `defineEval()` lowers camelCase
|
|
818
|
+
The `graders` catalog returns ordinary `assert` entries such as `type: is-json`, `type: regex`, `type: llm-rubric`, and `type: script`. Explicit `*.eval.ts` and `*.eval.mts` files should default-export an `EvalConfig`; `defineEval(config)` is only an optional thin helper over that same shape. TypeScript config loading lowers camelCase fields such as `expectedOutput`, `inputFiles`, and `maxSteps` to canonical snake_case YAML/runtime keys.
|
|
784
819
|
|
|
785
820
|
If adapting Braintrust `scores` or DeepEval metrics, write small AgentV helper factories that return `graders.*` configs:
|
|
786
821
|
|
|
@@ -790,7 +825,7 @@ import { graders } from '@agentv/sdk';
|
|
|
790
825
|
export function ragFaithfulness() {
|
|
791
826
|
return graders.llmRubric(undefined, {
|
|
792
827
|
name: 'rag-faithfulness',
|
|
793
|
-
|
|
828
|
+
provider: 'grader-provider',
|
|
794
829
|
prompt: 'Grade whether the answer is supported by the retrieved context.',
|
|
795
830
|
});
|
|
796
831
|
}
|
|
@@ -805,9 +840,11 @@ import { defineAssertion } from '@agentv/sdk';
|
|
|
805
840
|
|
|
806
841
|
export default defineAssertion(({ output, trace }) => {
|
|
807
842
|
const finalOutput = output ?? '';
|
|
843
|
+
const pass = finalOutput.length > 0 && (trace?.eventCount ?? 0) <= 10;
|
|
808
844
|
return {
|
|
809
|
-
pass
|
|
810
|
-
|
|
845
|
+
pass,
|
|
846
|
+
score: pass ? 1 : 0,
|
|
847
|
+
reason: 'Checks content exists and is efficient',
|
|
811
848
|
};
|
|
812
849
|
});
|
|
813
850
|
```
|
|
@@ -823,17 +860,21 @@ import { defineScriptGrader } from '@agentv/sdk';
|
|
|
823
860
|
|
|
824
861
|
export default defineScriptGrader(({ output, trace }) => {
|
|
825
862
|
const finalOutput = output ?? '';
|
|
863
|
+
const hasOutput = finalOutput.length > 0;
|
|
864
|
+
const efficient = (trace?.eventCount ?? 0) <= 5;
|
|
826
865
|
return {
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
866
|
+
pass: hasOutput && efficient,
|
|
867
|
+
score: hasOutput && efficient ? 1.0 : 0.5,
|
|
868
|
+
reason: 'Checks content exists and tool usage is bounded',
|
|
869
|
+
checks: [
|
|
870
|
+
{ text: 'Output is not empty', pass: hasOutput, reason: hasOutput ? 'Output text is present' : 'Output is empty' },
|
|
871
|
+
{ text: 'Efficient tool usage', pass: efficient, reason: efficient ? 'Trace event count is within limit' : 'Trace event count is too high' },
|
|
831
872
|
],
|
|
832
873
|
};
|
|
833
874
|
});
|
|
834
875
|
```
|
|
835
876
|
|
|
836
|
-
Use `defineScriptGrader()` when the custom component is a command-backed grader with explicit score control,
|
|
877
|
+
Use `defineScriptGrader()` when the custom component is a command-backed grader with explicit score control, check arrays, workspace commands, or LLM calls through a grader provider. `defineScriptGrader()` scripts are referenced in YAML with `type: script` and `command: [bun, run, grader.ts]`. Plain Vitest workspace verifier files can use `command: [agentv, eval, graders/check.test.ts]`.
|
|
837
878
|
|
|
838
879
|
### Convention-Based Discovery
|
|
839
880
|
|
|
@@ -56,18 +56,20 @@ Promptfoo normally calls eval custom logic assertions and uses fixed assertion t
|
|
|
56
56
|
|
|
57
57
|
```typescript
|
|
58
58
|
import {
|
|
59
|
-
|
|
59
|
+
createProviderClient,
|
|
60
60
|
defineScriptGrader,
|
|
61
61
|
defineEval,
|
|
62
|
+
type EvalConfig,
|
|
62
63
|
definePromptTemplate,
|
|
63
64
|
graders,
|
|
64
65
|
} from '@agentv/sdk';
|
|
65
66
|
```
|
|
66
67
|
|
|
67
68
|
- `defineScriptGrader(fn)` - Wraps evaluation function with stdin/stdout handling
|
|
68
|
-
- `
|
|
69
|
+
- `EvalConfig` - Public TypeScript eval authoring type for default-exported `*.eval.ts` and `*.eval.mts` files
|
|
70
|
+
- `defineEval(definition)` - Optional thin helper over the same `EvalConfig` shape
|
|
69
71
|
- `graders` - Helper catalog that returns ordinary AgentV `assert` entries
|
|
70
|
-
- `
|
|
72
|
+
- `createProviderClient()` - Returns LLM proxy client (when `provider: {}` configured)
|
|
71
73
|
- `.invoke({question, systemPrompt})` - Single LLM call
|
|
72
74
|
- `.invokeBatch(requests)` - Batch LLM calls
|
|
73
75
|
- `definePromptTemplate(fn)` - Wraps prompt generation function
|
|
@@ -81,29 +83,32 @@ For Python, the repo-local helper example in `examples/features/sdk-python/` kee
|
|
|
81
83
|
Use helper factories for reusable Braintrust/DeepEval-inspired checks, but keep the result as AgentV `assert` entries:
|
|
82
84
|
|
|
83
85
|
```typescript
|
|
84
|
-
import {
|
|
86
|
+
import { graders, type EvalConfig } from '@agentv/sdk';
|
|
85
87
|
|
|
86
88
|
function ragFaithfulness() {
|
|
87
89
|
return graders.llmRubric(undefined, {
|
|
88
90
|
name: 'rag-faithfulness',
|
|
89
|
-
|
|
91
|
+
provider: 'grader-provider',
|
|
90
92
|
prompt: 'Grade whether the answer is supported by the retrieved context.',
|
|
91
93
|
});
|
|
92
94
|
}
|
|
93
95
|
|
|
94
|
-
|
|
96
|
+
const config: EvalConfig = {
|
|
95
97
|
name: 'rag-suite',
|
|
98
|
+
prompts: ['{{ task }}'],
|
|
96
99
|
tests: [
|
|
97
100
|
{
|
|
98
101
|
id: 'grounded-answer',
|
|
99
|
-
|
|
102
|
+
vars: { task: 'Answer using the retrieved context.' },
|
|
100
103
|
assert: [
|
|
101
104
|
graders.contains('source', { name: 'mentions-source' }),
|
|
102
105
|
ragFaithfulness(),
|
|
103
106
|
],
|
|
104
107
|
},
|
|
105
108
|
],
|
|
106
|
-
}
|
|
109
|
+
};
|
|
110
|
+
|
|
111
|
+
export default config;
|
|
107
112
|
```
|
|
108
113
|
|
|
109
114
|
The helper lowers to ordinary YAML:
|
|
@@ -115,7 +120,7 @@ assert:
|
|
|
115
120
|
value: source
|
|
116
121
|
- name: rag-faithfulness
|
|
117
122
|
type: llm-rubric
|
|
118
|
-
|
|
123
|
+
provider: grader-provider
|
|
119
124
|
prompt: Grade whether the answer is supported by the retrieved context.
|
|
120
125
|
```
|
|
121
126
|
|
|
@@ -176,6 +181,6 @@ Derived from test fields (users never author these directly):
|
|
|
176
181
|
| `input` | Full resolved input array (JSON) |
|
|
177
182
|
| `expected_output` | Full resolved expected array (JSON) |
|
|
178
183
|
| `output` | Final answer / scored result string |
|
|
179
|
-
| `messages` | Transcript messages from
|
|
184
|
+
| `messages` | Transcript messages from provider execution |
|
|
180
185
|
|
|
181
186
|
Markdown templates use `{{variable}}` syntax. TypeScript templates receive context object.
|