agentv 5.3.0-next.1 → 5.3.2-next.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/README.md +70 -60
  2. package/dist/{artifact-writer-JFNIPMKW.js → artifact-writer-KJEOROKQ.js} +5 -5
  3. package/dist/{chunk-6ZZCDZPD.js → chunk-6262OKXM.js} +21774 -21034
  4. package/dist/chunk-6262OKXM.js.map +1 -0
  5. package/dist/chunk-AQ5BIAXF.js +604 -0
  6. package/dist/chunk-AQ5BIAXF.js.map +1 -0
  7. package/dist/chunk-BV5VQLI2.js +2 -0
  8. package/dist/{chunk-V52ATPTT.js → chunk-JGSRUJZQ.js} +186 -32
  9. package/dist/chunk-JGSRUJZQ.js.map +1 -0
  10. package/dist/chunk-MMDLXYBX.js +721 -0
  11. package/dist/chunk-MMDLXYBX.js.map +1 -0
  12. package/dist/{chunk-FVR4RQFK.js → chunk-QSK3PC44.js} +1131 -580
  13. package/dist/chunk-QSK3PC44.js.map +1 -0
  14. package/dist/chunk-TEEXVJWM.js +2 -0
  15. package/dist/{chunk-BHKQHG26.js → chunk-WIAJ7JAV.js} +186 -64
  16. package/dist/chunk-WIAJ7JAV.js.map +1 -0
  17. package/dist/cli.d.ts +1 -0
  18. package/dist/cli.js +17517 -10
  19. package/dist/cli.js.map +1 -1
  20. package/dist/config.d.ts +2 -0
  21. package/dist/{ts-eval-loader-DQDYRULE-V3377FOL.js → config.js} +7 -7
  22. package/dist/config.js.map +1 -0
  23. package/dist/contracts-DsZmZLl8.d.ts +742 -0
  24. package/dist/contracts.d.ts +2 -0
  25. package/dist/contracts.js +28 -0
  26. package/dist/contracts.js.map +1 -0
  27. package/dist/dashboard/assets/{index-DNgf3qJ2.js → index-BfqOlLOF.js} +1 -1
  28. package/dist/dashboard/assets/index-Bh8lpGce.css +1 -0
  29. package/dist/dashboard/assets/index-oPR3ywQb.js +121 -0
  30. package/dist/dashboard/index.html +2 -2
  31. package/dist/{dist-6Z7U473R.js → dist-A3SGR7TW.js} +30 -18
  32. package/dist/dist-A3SGR7TW.js.map +1 -0
  33. package/dist/index.d.ts +4 -0
  34. package/dist/index.js +137 -18
  35. package/dist/{interactive-RUY3OCBI.js → interactive-DL7N2C7K.js} +24 -24
  36. package/dist/interactive-DL7N2C7K.js.map +1 -0
  37. package/dist/provider.d.ts +2 -0
  38. package/dist/provider.js +24 -0
  39. package/dist/provider.js.map +1 -0
  40. package/dist/sdk.d.ts +802 -0
  41. package/dist/sdk.js +145 -0
  42. package/dist/sdk.js.map +1 -0
  43. package/dist/skills/agentv-bench/SKILL.md +14 -13
  44. package/dist/skills/agentv-bench/agents/analyzer.md +1 -1
  45. package/dist/skills/agentv-bench/agents/executor.md +1 -1
  46. package/dist/skills/agentv-bench/references/autoresearch.md +9 -9
  47. package/dist/skills/agentv-bench/references/environment-adaptation.md +4 -4
  48. package/dist/skills/agentv-bench/references/eval-yaml-spec.md +30 -47
  49. package/dist/skills/agentv-bench/references/schemas.md +44 -60
  50. package/dist/skills/agentv-bench/references/subagent-pipeline.md +20 -18
  51. package/dist/skills/agentv-eval-migrations/SKILL.md +13 -0
  52. package/dist/skills/agentv-eval-migrations/references/breaking-changes.md +56 -39
  53. package/dist/skills/agentv-eval-writer/SKILL.md +101 -60
  54. package/dist/skills/agentv-eval-writer/references/custom-evaluators.md +15 -10
  55. package/dist/skills/agentv-eval-writer/references/eval.schema.json +4543 -5189
  56. package/dist/skills/agentv-eval-writer/references/python-helpers.md +2 -2
  57. package/dist/skills/agentv-eval-writer/references/rubric-evaluator.md +20 -3
  58. package/dist/templates/.agentv/providers.yaml +42 -0
  59. package/dist/templates/.env.example +2 -2
  60. package/dist/ts-eval-loader-3G5GEAEC-6F52JLPP.js +18 -0
  61. package/dist/ts-eval-loader-3G5GEAEC-6F52JLPP.js.map +1 -0
  62. package/package.json +29 -4
  63. package/dist/chunk-6ZZCDZPD.js.map +0 -1
  64. package/dist/chunk-BHKQHG26.js.map +0 -1
  65. package/dist/chunk-FVR4RQFK.js.map +0 -1
  66. package/dist/chunk-T32NL3E6.js +0 -17973
  67. package/dist/chunk-T32NL3E6.js.map +0 -1
  68. package/dist/chunk-V52ATPTT.js.map +0 -1
  69. package/dist/dashboard/assets/index-D_bokML8.css +0 -1
  70. package/dist/dashboard/assets/index-r_jSJmlw.js +0 -121
  71. package/dist/interactive-RUY3OCBI.js.map +0 -1
  72. package/dist/templates/.agentv/targets.yaml +0 -97
  73. /package/dist/{artifact-writer-JFNIPMKW.js.map → artifact-writer-KJEOROKQ.js.map} +0 -0
  74. /package/dist/{dist-6Z7U473R.js.map → chunk-BV5VQLI2.js.map} +0 -0
  75. /package/dist/{ts-eval-loader-DQDYRULE-V3377FOL.js.map → chunk-TEEXVJWM.js.map} +0 -0
@@ -20,14 +20,14 @@ Promptfoo parity matrix: https://agentv.dev/docs/reference/promptfoo-parity/
20
20
  Treat YAML as the canonical portable model. Prefer authoring `.eval.yaml` / `EVAL.yaml` first, then use TypeScript helpers, Python scripts, or executable graders only when they lower to the same fields or when the evaluation logic must actually run code.
21
21
 
22
22
  Eval files define what is tested and how it runs: prompts, datasets, assertions,
23
- task fixtures, top-level `target`, and suite run controls. Use field-local file
23
+ task fixtures, top-level `providers`, and suite run controls. Use field-local file
24
24
  refs such as `tests: file://...`, `prompts: file://...`, `default_test:
25
25
  file://...`, and `environment: file://...`. String-valued `tests` and string
26
26
  entries inside `tests[]` are raw-case refs for direct paths, directories, and
27
27
  globs. Run several full eval suites directly with CLI multi-file selection and
28
28
  tags. Use scoped `run:` on individual tests only for `threshold`, `repeat`,
29
- `timeout_seconds`, and legacy `budget_usd`; keep target selection at top-level
30
- `target` or CLI `--target`, put suite budget caps under
29
+ `timeout_seconds`, and legacy `budget_usd`; keep provider selection at top-level
30
+ `providers` or CLI `--provider`, put suite budget caps under
31
31
  `evaluate_options.budget_usd`, authored concurrency under
32
32
  `evaluate_options.max_concurrency`, suite repeat policy under
33
33
  `evaluate_options.repeat`, coding-agent testbed setup under `environment`,
@@ -39,8 +39,8 @@ Use `@agentv/sdk` for TypeScript helper imports. Do not use `@agentv/eval` for n
39
39
  ## Authoring Checklist
40
40
 
41
41
  - Put grading criteria in `assert`, not in test-level `criteria`. Plain assertion strings become an `llm-rubric` grader.
42
- - Prefer plain assertion strings for semantic checks when the default rubric grader can judge them. Use `type: llm-rubric` for structured criteria, custom prompts, custom grader targets, or assertion-level transforms, and `type: script` when grading must execute code.
43
- - Write `expected_output` as a golden/reference answer the target could have produced. Do not write criteria, scoring instructions, or "the agent should..." rubric prose there.
42
+ - Prefer plain assertion strings for semantic checks when the default rubric grader can judge them. Use `type: llm-rubric` for structured criteria, custom prompts, custom grader providers, or assertion-level transforms. Use `type: agent-rubric` when the grader itself must be an agent-capable provider that can inspect the workspace. Use `type: script` when grading must execute code.
43
+ - Put reference answers in `tests[].vars.expected_output` or `default_test.vars.expected_output`, and consume them with an explicit assertion such as `type: llm-rubric` with `value: "Matches the reference answer: {{ expected_output }}"`. Do not write criteria, scoring instructions, or "the agent should..." rubric prose as the reference answer.
44
44
  - For historical or repo-state evals, materialize the repo through a pinned `environment` setup recipe. Mentioning a SHA only in prompt prose is not enough because the agent needs an actual checkout to inspect.
45
45
 
46
46
  ## Evaluation Types
@@ -61,7 +61,12 @@ agentv convert evals.json
61
61
  agentv eval evals.json
62
62
  ```
63
63
 
64
- The converter maps `prompt` → `input`, `expected_output` → `expected_output`, and Agent Skills `assertions` → AgentV `assert` (`llm-rubric` checks), and resolves `files[]` paths. The generated YAML includes TODO comments for AgentV features to add (environment setup, script graders, rubrics, required gates).
64
+ The converter maps Agent Skills prompt text into AgentV prompt/vars data,
65
+ promotes Agent Skills `expected_output` into explicit `llm-rubric` criteria or
66
+ `vars.expected_output` when it is true reference data, and maps Agent Skills
67
+ `assertions` to AgentV `assert` (`llm-rubric` checks). The generated YAML
68
+ includes TODO comments for AgentV features to add (environment setup, script
69
+ graders, rubrics, required gates).
65
70
 
66
71
  After converting, enhance the YAML with AgentV-specific capabilities shown below.
67
72
 
@@ -103,18 +108,22 @@ prompts:
103
108
 
104
109
  tests:
105
110
  - id: multi-turn-context
106
- expected_output: "Your name is Alice."
111
+ vars:
112
+ expected_output: "Your name is Alice."
107
113
  assert:
114
+ - type: llm-rubric
115
+ value: "Matches the reference answer: {{ expected_output }}"
108
116
  - Correctly recalls the user's name from earlier in the conversation
109
117
  ```
110
118
 
111
- **Guidelines:** preserve exact wording in `expected_output`; aim for 5–15 tests per transcript; pick exchanges that test different capabilities.
119
+ **Guidelines:** preserve exact wording in `vars.expected_output`; aim for 5–15 tests per transcript; pick exchanges that test different capabilities.
112
120
 
113
121
  ## Quick Start
114
122
 
115
123
  ```yaml
116
124
  description: Example eval
117
- target: default
125
+ providers:
126
+ - default
118
127
 
119
128
  prompts:
120
129
  - "{{ prompt }}"
@@ -123,8 +132,10 @@ tests:
123
132
  - id: greeting
124
133
  vars:
125
134
  prompt: "Say hello"
126
- expected_output: "Hello! How can I help you?"
135
+ expected_output: "Hello! How can I help you?"
127
136
  assert:
137
+ - type: llm-rubric
138
+ value: "Matches the reference answer: {{ expected_output }}"
128
139
  - Greeting is friendly and warm
129
140
  - Offers to help
130
141
  ```
@@ -132,7 +143,7 @@ tests:
132
143
  ## Eval File Structure
133
144
 
134
145
  **Required:** `tests` (array or string raw-case path) or `scenarios`
135
- **Optional:** `name`, `description`, `experiment`, `version`, `author`, `tags`, `license`, `requires`, `target`, `targets`, `prompts`, `default_test`, `timeout_seconds`, `evaluate_options`, `threshold`, `suite`, `environment`, `env`, `extensions`, `assert`
146
+ **Optional:** `name`, `description`, `version`, `author`, `tags`, `license`, `requires`, `providers`, `prompts`, `default_test`, `timeout_seconds`, `evaluate_options`, `threshold`, `suite`, `environment`, `env`, `extensions`, `assert`
136
147
 
137
148
  **Test fields:**
138
149
 
@@ -140,24 +151,35 @@ tests:
140
151
  |-------|----------|-------------|
141
152
  | `id` | yes | Unique identifier |
142
153
  | `vars` | yes when the prompt needs row data | Prompt-template variables for this row |
143
- | `expected_output` | no | Gold-standard reference answer (string shorthand or full message array) |
144
- | `assert` | yes | Graders: deterministic checks, `llm-rubric` checks, script graders, or plain string rubric criteria |
145
- | `execution` | no | Per-case grader/default overrides such as `skip_defaults`; target selection belongs in top-level `target` or CLI `--target` |
154
+ | `vars.expected_output` | no | Conventional reference-answer var consumed by explicit graders |
155
+ | `assert` | yes | Graders: deterministic checks, `llm-rubric` / `agent-rubric` checks, script graders, or plain string rubric criteria |
156
+ | `execution` | no | Per-case grader/default overrides such as `skip_defaults`; provider selection belongs in top-level `providers` or CLI `--provider` |
146
157
  | `environment` | no | Per-case coding-agent testbed config (overrides suite-level) |
147
158
  | `metadata` | no | Arbitrary key-value pairs passed to setup/teardown scripts |
148
159
  | `conversation_id` | no | Thread grouping |
149
160
 
161
+ **Provider declarations:** AgentV accepts Promptfoo-shaped provider entries
162
+ where they map cleanly: strings such as `openai:gpt-4.1-mini`, object form with
163
+ `id`, `label`, `config`, `env`, `prompts`, `transform`, `delay`, and `inputs`,
164
+ and provider maps such as `{ "openai:gpt-4": { label, config } }`. In object
165
+ form, `id` is the backend/spec and `label` is the stable AgentV selection and
166
+ result identity. AgentV-only `runtime`, provider-local `environment`, and
167
+ provider `hooks` must be explicit extensions; direct Promptfoo runs do not
168
+ execute them, and export must lower supported cases or reject unsupported ones
169
+ clearly.
170
+
150
171
  ## Prompt Templates and Vars
151
172
 
152
173
  Use top-level `prompts` plus `tests[].vars` for the Promptfoo-compatible canonical
153
174
  input shape. Shared data defaults belong in `default_test.vars`; per-test
154
175
  `vars` override those defaults by key. AgentV renders every prompt with each
155
176
  test's merged vars, then expands the run across prompts, targets, tests, and
156
- repeat attempts.
177
+ repeat samples.
157
178
 
158
179
  ```yaml
159
180
  description: Prompt matrix example
160
- target: default
181
+ providers:
182
+ - default
161
183
 
162
184
  prompts:
163
185
  - id: support-chat
@@ -176,15 +198,19 @@ tests:
176
198
  - id: password-reset
177
199
  vars:
178
200
  question: How do I reset my password?
179
- expected_output: Password reset guidance
201
+ expected_output: Password reset guidance
180
202
  assert:
203
+ - type: llm-rubric
204
+ value: "Matches the reference answer: {{ expected_output }}"
181
205
  - Gives correct password reset guidance
182
206
  - id: admin-access
183
207
  vars:
184
208
  audience: admins
185
209
  question: How do I revoke a user's access?
186
- expected_output: Access revocation guidance
210
+ expected_output: Access revocation guidance
187
211
  assert:
212
+ - type: llm-rubric
213
+ value: "Matches the reference answer: {{ expected_output }}"
188
214
  - Gives safe access revocation guidance
189
215
  ```
190
216
 
@@ -204,7 +230,7 @@ then render those vars from the prompt template next to the input.
204
230
  **Shorthand forms:**
205
231
  - Prompt entries can be strings, message arrays, file references, or prompt objects.
206
232
  - Put chat/system/user messages in `prompts`, not in `tests[].input`.
207
- - `expected_output` (string/object) expands to `[{role: "assistant", content: ...}]`
233
+ - `vars.expected_output` is a conventional reference-answer variable; explicit assertions decide how to grade it
208
234
  - Use these canonical field names on disk; keep the wire format `snake_case`
209
235
 
210
236
  **Message format:** `{role, content}` where role is `system`, `user`, `assistant`, or `tool`
@@ -356,12 +382,14 @@ tests:
356
382
  verification guidance was added.
357
383
 
358
384
  Decide what durable repo change should be made and explain why.
359
- expected_output: |
360
- The durable repo change is to update .agents/verification.md with the
361
- reusable verification workflow lessons. AGENTS.md already routes this
362
- class of work to .agents/verification.md, so no extra AGENTS.md edit is
363
- needed unless that routing is missing.
385
+ expected_output: |
386
+ The durable repo change is to update .agents/verification.md with the
387
+ reusable verification workflow lessons. AGENTS.md already routes this
388
+ class of work to .agents/verification.md, so no extra AGENTS.md edit is
389
+ needed unless that routing is missing.
364
390
  assert:
391
+ - type: llm-rubric
392
+ value: "Matches the reference answer: {{ expected_output }}"
365
393
  - The answer recommends updating .agents/verification.md rather than leaving the learning only in PR comments or private evidence.
366
394
  - The answer uses the pinned ./agentv checkout to verify the AGENTS.md routing.
367
395
  - The answer preserves the historical commit SHA as context.
@@ -371,7 +399,7 @@ tests:
371
399
 
372
400
  When `assert` is defined, **only the declared graders run**. For
373
401
  semantic checks, add plain rubric strings. If you need a custom LLM prompt or
374
- grader target, declare `llm-rubric` explicitly:
402
+ grader provider, declare `llm-rubric` explicitly:
375
403
 
376
404
  ```yaml
377
405
  prompts:
@@ -387,9 +415,9 @@ tests:
387
415
  value: "fix"
388
416
  ```
389
417
 
390
- `expected_output` is passive reference data. It is available to graders through
391
- `{{expected_output}}` and the script stdin payload, but it does not create an
392
- implicit LLM grading call by itself.
418
+ `vars.expected_output` is passive reference data. It is available to graders
419
+ through `{{ expected_output }}` and the script stdin payload, but it does not
420
+ create an implicit LLM grading call by itself.
393
421
 
394
422
  **Common mistake:** putting rubric prose in `expected_output` instead of an
395
423
  assertion:
@@ -402,7 +430,7 @@ tests:
402
430
  - id: bad-example
403
431
  vars:
404
432
  prompt: "What is 2+2?"
405
- expected_output: The assistant should explain why the answer is 4. # reference answer field, not a grader
433
+ expected_output: The assistant should explain why the answer is 4. # reference answer var, not a grader
406
434
  ```
407
435
 
408
436
  Write this as:
@@ -415,8 +443,10 @@ tests:
415
443
  - id: good-example
416
444
  vars:
417
445
  prompt: "What is 2+2?"
418
- expected_output: "4"
446
+ expected_output: "4"
419
447
  assert:
448
+ - type: llm-rubric
449
+ value: "Matches the reference answer: {{ expected_output }}"
420
450
  - The answer is 4 and explains the arithmetic briefly
421
451
  ```
422
452
 
@@ -438,7 +468,7 @@ assert:
438
468
  weight: 5.0
439
469
  ```
440
470
 
441
- If a required grader scores below its threshold, the overall verdict is forced to `fail`.
471
+ If a required grader scores below its threshold, the overall case status is forced to `fail`.
442
472
 
443
473
  ## Environment Setup/Teardown
444
474
 
@@ -521,11 +551,12 @@ Configure via the `assert` array. Multiple graders produce a weighted average sc
521
551
  type: script
522
552
  command: [uv, run, validate.py]
523
553
  cwd: ./scripts # optional working directory
524
- target: {} # optional: enable LLM target proxy (max_calls: 50)
554
+ provider: {} # optional: enable LLM provider proxy (max_calls: 50)
525
555
  ```
526
- Contract: stdin JSON -> stdout JSON `{score, assertions: [{text, passed, evidence?}], reasoning}`
556
+ Contract: stdin JSON -> stdout JSON `{pass, score, reason, checks?: [{text, pass, score?, reason}]}`
527
557
  Raw stdin uses snake_case and includes: `input`, `expected_output`, `output` (final answer string), `messages`, `trace`, `trace_summary`, `token_usage`, `cost_usd`, `duration_ms`, `start_time`, `end_time`, `file_changes`, `workspace_path`, `config`
528
558
  SDK handlers receive the same payload in camelCase: `expectedOutput`, `traceSummary`, `tokenUsage`, `costUsd`, `durationMs`, `startTime`, `endTime`, `fileChanges`, `workspacePath`.
559
+ `checks` is an SDK/script convenience shape; public `grading.json` artifacts normalize checks into recursive `component_results`.
529
560
  When an environment prepares a workspace directory, `workspace_path` is the absolute path to that directory (also available as `AGENTV_WORKSPACE_PATH` env var). Use this for functional grading (e.g., running `npm test` in the prepared workdir).
530
561
  For deterministic workspace checks that fit normal Vitest `expect(...)` tests, prefer a plain verifier file and the built-in adapter:
531
562
  ```yaml
@@ -541,7 +572,7 @@ See the Script Graders docs for the full stdin/stdout contract.
541
572
  - name: quality
542
573
  type: llm-rubric
543
574
  prompt: ./prompts/eval.md # markdown template or command config
544
- target: grader_gpt_5_mini # optional: override the grader target for this grader
575
+ provider: grader_gpt_5_mini # optional: override the grader provider for this grader
545
576
  model: gpt-5-chat # optional model override
546
577
  config: # passed to prompt templates as context.config
547
578
  strictness: high
@@ -549,7 +580,7 @@ See the Script Graders docs for the full stdin/stdout contract.
549
580
  Variables: `{{criteria}}`, `{{input}}`, `{{expected_output}}`, `{{output}}`, `{{metadata}}`, `{{metadata_json}}`, `{{rubrics}}`, `{{rubrics_json}}`, `{{file_changes}}`, `{{tool_calls}}`
550
581
  - Markdown templates: use `{{variable}}` syntax
551
582
  - TypeScript templates: use `definePromptTemplate(fn)` from `@agentv/sdk`, receives context object with all variables + `config`
552
- - Use `target:` to run different `llm-rubric` graders against different named LLM targets in the same eval (useful for grader panels / ensembles)
583
+ - Use `provider:` to run different `llm-rubric` graders against different named LLM providers in the same eval (useful for grader panels / ensembles)
553
584
 
554
585
  ### assert-set
555
586
  ```yaml
@@ -567,7 +598,7 @@ Variables: `{{criteria}}`, `{{input}}`, `{{expected_output}}`, `{{output}}`, `{{
567
598
  type: llm-rubric
568
599
  weight: 0.7
569
600
  ```
570
- Use `assert-set` for Promptfoo-aligned assertion grouping. Without `threshold`, the set passes only when every nonzero-weight child assertion passes. With `threshold`, the weighted aggregate score determines the set verdict. Parent `config` is inherited by children, and child `config` keys override parent keys. Do not use `type: composite`; AgentV rejects it.
601
+ Use `assert-set` for Promptfoo-aligned assertion grouping. Without `threshold`, the set passes only when every nonzero-weight child assertion passes. With `threshold`, the weighted aggregate score determines the set pass/fail result. Parent `config` is inherited by children, and child `config` keys override parent keys. Do not use `type: composite`; AgentV rejects it.
571
602
 
572
603
  ### Skill And Trajectory Assertions
573
604
  ```yaml
@@ -722,24 +753,24 @@ agentv eval assert <grader-name> --agent-output "..." --agent-input "..."
722
753
  agentv import claude --session-id <uuid>
723
754
 
724
755
  # Re-run only execution errors from a previous run
725
- agentv eval <file.yaml> --retry-errors .agentv/results/default/<timestamp>/index.jsonl
756
+ agentv eval <file.yaml> --retry-errors .agentv/results/<run_id>/.internal/index.jsonl
726
757
 
727
758
  # Validate eval file
728
759
  agentv validate <file.yaml>
729
760
 
730
761
  # Compare completed runs
731
762
  agentv results compare \
732
- .agentv/results/default/<baseline-timestamp>/index.jsonl \
733
- .agentv/results/default/<candidate-timestamp>/index.jsonl
763
+ .agentv/results/<baseline-run-id>/.internal/index.jsonl \
764
+ .agentv/results/<candidate-run-id>/.internal/index.jsonl
734
765
  agentv results combine \
735
- .agentv/results/default/<baseline-timestamp> \
736
- .agentv/results/default/<candidate-timestamp> \
737
- .agentv/results/default/<third-target-timestamp> \
738
- --output .agentv/results/default/combined
739
- agentv results compare .agentv/results/default/combined/index.jsonl
766
+ .agentv/results/<baseline-run-id> \
767
+ .agentv/results/<candidate-run-id> \
768
+ .agentv/results/<third-target-run-id> \
769
+ --output .agentv/results/combined
770
+ agentv results compare .agentv/results/combined/.internal/index.jsonl
740
771
  agentv results compare \
741
- .agentv/results/default/<baseline-timestamp>/index.jsonl \
742
- .agentv/results/default/<candidate-timestamp>/index.jsonl \
772
+ .agentv/results/<baseline-run-id>/.internal/index.jsonl \
773
+ .agentv/results/<candidate-run-id>/.internal/index.jsonl \
743
774
  --json
744
775
 
745
776
  # Author assertions directly in the eval file
@@ -755,32 +786,36 @@ Use `@agentv/sdk` as the public lightweight SDK package for TypeScript/JavaScrip
755
786
 
756
787
  ### YAML-aligned eval authoring
757
788
  ```typescript
758
- import { defineEval, graders } from '@agentv/sdk';
789
+ // evals/helper-suite.eval.ts
790
+ import { graders, type EvalConfig } from '@agentv/sdk';
759
791
 
760
- export default defineEval({
792
+ const config: EvalConfig = {
761
793
  name: 'helper-suite',
762
794
  target: 'default',
763
- // The SDK helper lowers this to evaluate_options.repeat in generated YAML.
795
+ // TypeScript config loading lowers this to evaluate_options.repeat.
764
796
  repeat: {
765
797
  count: 3,
766
798
  strategy: 'pass_any',
767
799
  earlyExit: false,
768
800
  },
769
801
  threshold: 0.8,
802
+ prompts: ['{{ task }}'],
770
803
  tests: [
771
804
  {
772
805
  id: 'json-answer',
773
- input: 'Return a JSON answer with a status field.',
806
+ vars: { task: 'Return a JSON answer with a status field.' },
774
807
  assert: [
775
808
  graders.json({ name: 'valid-json', required: true }),
776
809
  graders.regex(/"status"\s*:/, { name: 'status-key' }),
777
810
  ],
778
811
  },
779
812
  ],
780
- });
813
+ };
814
+
815
+ export default config;
781
816
  ```
782
817
 
783
- The `graders` catalog returns ordinary `assert` entries such as `type: is-json`, `type: regex`, `type: llm-rubric`, and `type: script`. `defineEval()` lowers camelCase TypeScript fields such as `expectedOutput`, `inputFiles`, and `maxSteps` to canonical snake_case YAML/runtime keys.
818
+ The `graders` catalog returns ordinary `assert` entries such as `type: is-json`, `type: regex`, `type: llm-rubric`, and `type: script`. Explicit `*.eval.ts` and `*.eval.mts` files should default-export an `EvalConfig`; `defineEval(config)` is only an optional thin helper over that same shape. TypeScript config loading lowers camelCase fields such as `expectedOutput`, `inputFiles`, and `maxSteps` to canonical snake_case YAML/runtime keys.
784
819
 
785
820
  If adapting Braintrust `scores` or DeepEval metrics, write small AgentV helper factories that return `graders.*` configs:
786
821
 
@@ -790,7 +825,7 @@ import { graders } from '@agentv/sdk';
790
825
  export function ragFaithfulness() {
791
826
  return graders.llmRubric(undefined, {
792
827
  name: 'rag-faithfulness',
793
- target: 'grader-target',
828
+ provider: 'grader-provider',
794
829
  prompt: 'Grade whether the answer is supported by the retrieved context.',
795
830
  });
796
831
  }
@@ -805,9 +840,11 @@ import { defineAssertion } from '@agentv/sdk';
805
840
 
806
841
  export default defineAssertion(({ output, trace }) => {
807
842
  const finalOutput = output ?? '';
843
+ const pass = finalOutput.length > 0 && (trace?.eventCount ?? 0) <= 10;
808
844
  return {
809
- pass: finalOutput.length > 0 && (trace?.eventCount ?? 0) <= 10,
810
- reasoning: 'Checks content exists and is efficient',
845
+ pass,
846
+ score: pass ? 1 : 0,
847
+ reason: 'Checks content exists and is efficient',
811
848
  };
812
849
  });
813
850
  ```
@@ -823,17 +860,21 @@ import { defineScriptGrader } from '@agentv/sdk';
823
860
 
824
861
  export default defineScriptGrader(({ output, trace }) => {
825
862
  const finalOutput = output ?? '';
863
+ const hasOutput = finalOutput.length > 0;
864
+ const efficient = (trace?.eventCount ?? 0) <= 5;
826
865
  return {
827
- score: finalOutput.length > 0 && (trace?.eventCount ?? 0) <= 5 ? 1.0 : 0.5,
828
- assert: [
829
- { text: 'Output is not empty', passed: finalOutput.length > 0 },
830
- { text: 'Efficient tool usage', passed: (trace?.eventCount ?? 0) <= 5 },
866
+ pass: hasOutput && efficient,
867
+ score: hasOutput && efficient ? 1.0 : 0.5,
868
+ reason: 'Checks content exists and tool usage is bounded',
869
+ checks: [
870
+ { text: 'Output is not empty', pass: hasOutput, reason: hasOutput ? 'Output text is present' : 'Output is empty' },
871
+ { text: 'Efficient tool usage', pass: efficient, reason: efficient ? 'Trace event count is within limit' : 'Trace event count is too high' },
831
872
  ],
832
873
  };
833
874
  });
834
875
  ```
835
876
 
836
- Use `defineScriptGrader()` when the custom component is a command-backed grader with explicit score control, custom assertion-result arrays, workspace commands, or LLM calls through a grader target. `defineScriptGrader()` scripts are referenced in YAML with `type: script` and `command: [bun, run, grader.ts]`. Plain Vitest workspace verifier files can use `command: [agentv, eval, graders/check.test.ts]`.
877
+ Use `defineScriptGrader()` when the custom component is a command-backed grader with explicit score control, check arrays, workspace commands, or LLM calls through a grader provider. `defineScriptGrader()` scripts are referenced in YAML with `type: script` and `command: [bun, run, grader.ts]`. Plain Vitest workspace verifier files can use `command: [agentv, eval, graders/check.test.ts]`.
837
878
 
838
879
  ### Convention-Based Discovery
839
880
 
@@ -56,18 +56,20 @@ Promptfoo normally calls eval custom logic assertions and uses fixed assertion t
56
56
 
57
57
  ```typescript
58
58
  import {
59
- createTargetClient,
59
+ createProviderClient,
60
60
  defineScriptGrader,
61
61
  defineEval,
62
+ type EvalConfig,
62
63
  definePromptTemplate,
63
64
  graders,
64
65
  } from '@agentv/sdk';
65
66
  ```
66
67
 
67
68
  - `defineScriptGrader(fn)` - Wraps evaluation function with stdin/stdout handling
68
- - `defineEval(definition)` - Defines a YAML-aligned `.eval.ts` suite
69
+ - `EvalConfig` - Public TypeScript eval authoring type for default-exported `*.eval.ts` and `*.eval.mts` files
70
+ - `defineEval(definition)` - Optional thin helper over the same `EvalConfig` shape
69
71
  - `graders` - Helper catalog that returns ordinary AgentV `assert` entries
70
- - `createTargetClient()` - Returns LLM proxy client (when `target: {}` configured)
72
+ - `createProviderClient()` - Returns LLM proxy client (when `provider: {}` configured)
71
73
  - `.invoke({question, systemPrompt})` - Single LLM call
72
74
  - `.invokeBatch(requests)` - Batch LLM calls
73
75
  - `definePromptTemplate(fn)` - Wraps prompt generation function
@@ -81,29 +83,32 @@ For Python, the repo-local helper example in `examples/features/sdk-python/` kee
81
83
  Use helper factories for reusable Braintrust/DeepEval-inspired checks, but keep the result as AgentV `assert` entries:
82
84
 
83
85
  ```typescript
84
- import { defineEval, graders } from '@agentv/sdk';
86
+ import { graders, type EvalConfig } from '@agentv/sdk';
85
87
 
86
88
  function ragFaithfulness() {
87
89
  return graders.llmRubric(undefined, {
88
90
  name: 'rag-faithfulness',
89
- target: 'grader-target',
91
+ provider: 'grader-provider',
90
92
  prompt: 'Grade whether the answer is supported by the retrieved context.',
91
93
  });
92
94
  }
93
95
 
94
- export default defineEval({
96
+ const config: EvalConfig = {
95
97
  name: 'rag-suite',
98
+ prompts: ['{{ task }}'],
96
99
  tests: [
97
100
  {
98
101
  id: 'grounded-answer',
99
- input: 'Answer using the retrieved context.',
102
+ vars: { task: 'Answer using the retrieved context.' },
100
103
  assert: [
101
104
  graders.contains('source', { name: 'mentions-source' }),
102
105
  ragFaithfulness(),
103
106
  ],
104
107
  },
105
108
  ],
106
- });
109
+ };
110
+
111
+ export default config;
107
112
  ```
108
113
 
109
114
  The helper lowers to ordinary YAML:
@@ -115,7 +120,7 @@ assert:
115
120
  value: source
116
121
  - name: rag-faithfulness
117
122
  type: llm-rubric
118
- target: grader-target
123
+ provider: grader-provider
119
124
  prompt: Grade whether the answer is supported by the retrieved context.
120
125
  ```
121
126
 
@@ -176,6 +181,6 @@ Derived from test fields (users never author these directly):
176
181
  | `input` | Full resolved input array (JSON) |
177
182
  | `expected_output` | Full resolved expected array (JSON) |
178
183
  | `output` | Final answer / scored result string |
179
- | `messages` | Transcript messages from target execution |
184
+ | `messages` | Transcript messages from provider execution |
180
185
 
181
186
  Markdown templates use `{{variable}}` syntax. TypeScript templates receive context object.