agentv 5.3.1-next.1 → 5.3.2-next.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +52 -44
- package/dist/{artifact-writer-7NBCOAYC.js → artifact-writer-KJEOROKQ.js} +5 -5
- package/dist/{chunk-ELCJ23K4.js → chunk-6262OKXM.js} +2856 -2404
- package/dist/chunk-6262OKXM.js.map +1 -0
- package/dist/chunk-AQ5BIAXF.js +604 -0
- package/dist/chunk-AQ5BIAXF.js.map +1 -0
- package/dist/chunk-BV5VQLI2.js +2 -0
- package/dist/{chunk-LXBI3SPX.js → chunk-JGSRUJZQ.js} +46 -20
- package/dist/chunk-JGSRUJZQ.js.map +1 -0
- package/dist/chunk-MMDLXYBX.js +721 -0
- package/dist/chunk-MMDLXYBX.js.map +1 -0
- package/dist/{chunk-LKGARI3W.js → chunk-QSK3PC44.js} +812 -570
- package/dist/chunk-QSK3PC44.js.map +1 -0
- package/dist/chunk-TEEXVJWM.js +2 -0
- package/dist/{chunk-RKE7SSET.js → chunk-WIAJ7JAV.js} +186 -64
- package/dist/chunk-WIAJ7JAV.js.map +1 -0
- package/dist/cli.d.ts +1 -0
- package/dist/cli.js +17517 -10
- package/dist/cli.js.map +1 -1
- package/dist/config.d.ts +2 -0
- package/dist/config.js +14 -0
- package/dist/config.js.map +1 -0
- package/dist/contracts-DsZmZLl8.d.ts +742 -0
- package/dist/contracts.d.ts +2 -0
- package/dist/contracts.js +28 -0
- package/dist/contracts.js.map +1 -0
- package/dist/dashboard/assets/{index-CbEMiJSb.js → index-BfqOlLOF.js} +1 -1
- package/dist/dashboard/assets/index-Bh8lpGce.css +1 -0
- package/dist/dashboard/assets/index-oPR3ywQb.js +121 -0
- package/dist/dashboard/index.html +2 -2
- package/dist/{dist-NMXMI5SK.js → dist-A3SGR7TW.js} +16 -16
- package/dist/dist-A3SGR7TW.js.map +1 -0
- package/dist/index.d.ts +4 -0
- package/dist/index.js +137 -18
- package/dist/{interactive-BN527UV3.js → interactive-DL7N2C7K.js} +24 -24
- package/dist/interactive-DL7N2C7K.js.map +1 -0
- package/dist/provider.d.ts +2 -0
- package/dist/provider.js +24 -0
- package/dist/provider.js.map +1 -0
- package/dist/sdk.d.ts +802 -0
- package/dist/sdk.js +145 -0
- package/dist/sdk.js.map +1 -0
- package/dist/skills/agentv-eval-migrations/references/breaking-changes.md +17 -18
- package/dist/skills/agentv-eval-writer/SKILL.md +27 -15
- package/dist/skills/agentv-eval-writer/references/custom-evaluators.md +5 -5
- package/dist/skills/agentv-eval-writer/references/eval.schema.json +5402 -4639
- package/dist/skills/agentv-eval-writer/references/python-helpers.md +2 -2
- package/dist/skills/agentv-eval-writer/references/rubric-evaluator.md +19 -2
- package/dist/templates/.agentv/providers.yaml +42 -0
- package/dist/templates/.env.example +2 -2
- package/dist/{ts-eval-loader-2RFVZHCT-7CZ3DCDD.js → ts-eval-loader-3G5GEAEC-6F52JLPP.js} +3 -3
- package/dist/ts-eval-loader-3G5GEAEC-6F52JLPP.js.map +1 -0
- package/package.json +29 -4
- package/dist/chunk-ASIGJIOJ.js +0 -17993
- package/dist/chunk-ASIGJIOJ.js.map +0 -1
- package/dist/chunk-ELCJ23K4.js.map +0 -1
- package/dist/chunk-LKGARI3W.js.map +0 -1
- package/dist/chunk-LXBI3SPX.js.map +0 -1
- package/dist/chunk-RKE7SSET.js.map +0 -1
- package/dist/dashboard/assets/index-DTA6-l7q.js +0 -121
- package/dist/dashboard/assets/index-D_bokML8.css +0 -1
- package/dist/interactive-BN527UV3.js.map +0 -1
- package/dist/templates/.agentv/targets.yaml +0 -97
- /package/dist/{artifact-writer-7NBCOAYC.js.map → artifact-writer-KJEOROKQ.js.map} +0 -0
- /package/dist/{dist-NMXMI5SK.js.map → chunk-BV5VQLI2.js.map} +0 -0
- /package/dist/{ts-eval-loader-2RFVZHCT-7CZ3DCDD.js.map → chunk-TEEXVJWM.js.map} +0 -0
package/dist/sdk.js
ADDED
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
import { createRequire } from 'node:module'; const require = createRequire(import.meta.url);
|
|
2
|
+
import {
|
|
3
|
+
ProviderInvocationError,
|
|
4
|
+
ProviderNotAvailableError,
|
|
5
|
+
codeGrader,
|
|
6
|
+
containsGrader,
|
|
7
|
+
createProviderClient,
|
|
8
|
+
createWorkspace,
|
|
9
|
+
defineAssertion,
|
|
10
|
+
defineCodeGrader,
|
|
11
|
+
defineEval,
|
|
12
|
+
definePromptTemplate,
|
|
13
|
+
defineScriptGrader,
|
|
14
|
+
defineWorkspaceGrader,
|
|
15
|
+
equalsGrader,
|
|
16
|
+
exactGrader,
|
|
17
|
+
graders,
|
|
18
|
+
isJsonGrader,
|
|
19
|
+
jsonGrader,
|
|
20
|
+
llmRubricGrader,
|
|
21
|
+
normalizeWorkspaceGraderResult,
|
|
22
|
+
regexGrader,
|
|
23
|
+
runWorkspaceGrader,
|
|
24
|
+
scriptGrader,
|
|
25
|
+
serializeEvalYaml,
|
|
26
|
+
toEvalYamlObject
|
|
27
|
+
} from "./chunk-MMDLXYBX.js";
|
|
28
|
+
import {
|
|
29
|
+
CodeGraderInputSchema,
|
|
30
|
+
CodeGraderResultSchema,
|
|
31
|
+
ContentFileSchema,
|
|
32
|
+
ContentImageSchema,
|
|
33
|
+
ContentSchema,
|
|
34
|
+
ContentTextSchema,
|
|
35
|
+
MessageSchema,
|
|
36
|
+
PromptTemplateInputSchema,
|
|
37
|
+
ScriptGraderCheckSchema,
|
|
38
|
+
ScriptGraderInputSchema,
|
|
39
|
+
ScriptGraderResultSchema,
|
|
40
|
+
TRACE_EVENT_TYPES,
|
|
41
|
+
TRACE_REDACTION_LEVELS,
|
|
42
|
+
TRACE_SOURCE_KINDS,
|
|
43
|
+
TRACE_TOOL_STATUSES,
|
|
44
|
+
TokenUsageSchema,
|
|
45
|
+
ToolCallSchema,
|
|
46
|
+
TraceArtifactSchema,
|
|
47
|
+
TraceBranchSchema,
|
|
48
|
+
TraceErrorSchema,
|
|
49
|
+
TraceEventSchema,
|
|
50
|
+
TraceMessageSchema,
|
|
51
|
+
TraceModelSchema,
|
|
52
|
+
TraceRawEvidenceSchema,
|
|
53
|
+
TraceRedactionStateSchema,
|
|
54
|
+
TraceSchema,
|
|
55
|
+
TraceSessionSchema,
|
|
56
|
+
TraceSourceRefSchema,
|
|
57
|
+
TraceSourceSchema,
|
|
58
|
+
TraceSummarySchema,
|
|
59
|
+
TraceToolSchema,
|
|
60
|
+
defineVitestWorkspaceGrader,
|
|
61
|
+
runCodeGrader,
|
|
62
|
+
runScriptGrader,
|
|
63
|
+
runVitestWorkspaceGrader,
|
|
64
|
+
vitestReportToCodeGraderResult,
|
|
65
|
+
vitestReportToScriptGraderResult
|
|
66
|
+
} from "./chunk-AQ5BIAXF.js";
|
|
67
|
+
import "./chunk-TEEXVJWM.js";
|
|
68
|
+
import {
|
|
69
|
+
defineConfig
|
|
70
|
+
} from "./chunk-JGSRUJZQ.js";
|
|
71
|
+
import {
|
|
72
|
+
evaluate,
|
|
73
|
+
external_exports
|
|
74
|
+
} from "./chunk-6262OKXM.js";
|
|
75
|
+
import "./chunk-M7BUKBAF.js";
|
|
76
|
+
import "./chunk-7BGERE6L.js";
|
|
77
|
+
import "./chunk-PEUTJS7B.js";
|
|
78
|
+
import "./chunk-QMRVH5ZP.js";
|
|
79
|
+
export {
|
|
80
|
+
CodeGraderInputSchema,
|
|
81
|
+
CodeGraderResultSchema,
|
|
82
|
+
ContentFileSchema,
|
|
83
|
+
ContentImageSchema,
|
|
84
|
+
ContentSchema,
|
|
85
|
+
ContentTextSchema,
|
|
86
|
+
MessageSchema,
|
|
87
|
+
PromptTemplateInputSchema,
|
|
88
|
+
ProviderInvocationError,
|
|
89
|
+
ProviderNotAvailableError,
|
|
90
|
+
ScriptGraderCheckSchema,
|
|
91
|
+
ScriptGraderInputSchema,
|
|
92
|
+
ScriptGraderResultSchema,
|
|
93
|
+
TRACE_EVENT_TYPES,
|
|
94
|
+
TRACE_REDACTION_LEVELS,
|
|
95
|
+
TRACE_SOURCE_KINDS,
|
|
96
|
+
TRACE_TOOL_STATUSES,
|
|
97
|
+
TokenUsageSchema,
|
|
98
|
+
ToolCallSchema,
|
|
99
|
+
TraceArtifactSchema,
|
|
100
|
+
TraceBranchSchema,
|
|
101
|
+
TraceErrorSchema,
|
|
102
|
+
TraceEventSchema,
|
|
103
|
+
TraceMessageSchema,
|
|
104
|
+
TraceModelSchema,
|
|
105
|
+
TraceRawEvidenceSchema,
|
|
106
|
+
TraceRedactionStateSchema,
|
|
107
|
+
TraceSchema,
|
|
108
|
+
TraceSessionSchema,
|
|
109
|
+
TraceSourceRefSchema,
|
|
110
|
+
TraceSourceSchema,
|
|
111
|
+
TraceSummarySchema,
|
|
112
|
+
TraceToolSchema,
|
|
113
|
+
codeGrader,
|
|
114
|
+
containsGrader,
|
|
115
|
+
createProviderClient,
|
|
116
|
+
createWorkspace,
|
|
117
|
+
defineAssertion,
|
|
118
|
+
defineCodeGrader,
|
|
119
|
+
defineConfig,
|
|
120
|
+
defineEval,
|
|
121
|
+
definePromptTemplate,
|
|
122
|
+
defineScriptGrader,
|
|
123
|
+
defineVitestWorkspaceGrader,
|
|
124
|
+
defineWorkspaceGrader,
|
|
125
|
+
equalsGrader,
|
|
126
|
+
evaluate,
|
|
127
|
+
exactGrader,
|
|
128
|
+
graders,
|
|
129
|
+
isJsonGrader,
|
|
130
|
+
jsonGrader,
|
|
131
|
+
llmRubricGrader,
|
|
132
|
+
normalizeWorkspaceGraderResult,
|
|
133
|
+
regexGrader,
|
|
134
|
+
runCodeGrader,
|
|
135
|
+
runScriptGrader,
|
|
136
|
+
runVitestWorkspaceGrader,
|
|
137
|
+
runWorkspaceGrader,
|
|
138
|
+
scriptGrader,
|
|
139
|
+
serializeEvalYaml,
|
|
140
|
+
toEvalYamlObject,
|
|
141
|
+
vitestReportToCodeGraderResult,
|
|
142
|
+
vitestReportToScriptGraderResult,
|
|
143
|
+
external_exports as z
|
|
144
|
+
};
|
|
145
|
+
//# sourceMappingURL=sdk.js.map
|
package/dist/sdk.js.map
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":[],"sourcesContent":[],"mappings":"","names":[]}
|
|
@@ -74,7 +74,7 @@ v4.42.4 docs and schema used `assertions` for suite-level and per-test graders:
|
|
|
74
74
|
```yaml
|
|
75
75
|
assertions:
|
|
76
76
|
- name: correctness
|
|
77
|
-
type: llm-
|
|
77
|
+
type: llm-rubric
|
|
78
78
|
prompt: ./graders/correctness.md
|
|
79
79
|
|
|
80
80
|
tests:
|
|
@@ -539,11 +539,13 @@ experiment:
|
|
|
539
539
|
|
|
540
540
|
### Current Shape
|
|
541
541
|
|
|
542
|
-
Current schema
|
|
543
|
-
|
|
542
|
+
Current schema rejects top-level `experiment` in authored eval YAML. Use the
|
|
543
|
+
promptfoo-shaped `tags.experiment` key when the label belongs in the eval file,
|
|
544
|
+
or pass CLI `--experiment` when the label is run-time context.
|
|
544
545
|
|
|
545
546
|
```yaml
|
|
546
|
-
|
|
547
|
+
tags:
|
|
548
|
+
experiment: with-skills
|
|
547
549
|
target: codex
|
|
548
550
|
timeout_seconds: 600
|
|
549
551
|
evaluate_options:
|
|
@@ -552,14 +554,6 @@ evaluate_options:
|
|
|
552
554
|
strategy: pass_any
|
|
553
555
|
```
|
|
554
556
|
|
|
555
|
-
or:
|
|
556
|
-
|
|
557
|
-
```yaml
|
|
558
|
-
tags:
|
|
559
|
-
experiment: with-skills
|
|
560
|
-
target: codex
|
|
561
|
-
```
|
|
562
|
-
|
|
563
557
|
### Migration Steps
|
|
564
558
|
|
|
565
559
|
- If `experiment` is an object, move runtime fields out:
|
|
@@ -568,7 +562,8 @@ target: codex
|
|
|
568
562
|
- budget -> `evaluate_options.budget_usd`
|
|
569
563
|
- timeout -> top-level `timeout_seconds`
|
|
570
564
|
- threshold -> top-level `threshold`
|
|
571
|
-
-
|
|
565
|
+
- Move string experiment labels to `tags.experiment`, or supply them at run time
|
|
566
|
+
with CLI `--experiment`.
|
|
572
567
|
|
|
573
568
|
### Verification
|
|
574
569
|
|
|
@@ -577,6 +572,8 @@ bun apps/cli/src/cli.ts validate path/to/eval.eval.yaml
|
|
|
577
572
|
rg -n "^experiment:" path/to/evals
|
|
578
573
|
```
|
|
579
574
|
|
|
575
|
+
Any match under authored eval YAML should be removed or migrated.
|
|
576
|
+
|
|
580
577
|
### Compatibility Notes
|
|
581
578
|
|
|
582
579
|
Do not describe a v4.42.4 eval as if `experiment:` was already the main runtime
|
|
@@ -895,8 +892,10 @@ targets:
|
|
|
895
892
|
- Remove `use_target`; current authored target definitions must resolve to
|
|
896
893
|
concrete provider objects.
|
|
897
894
|
- Keep supported AgentV target extensions such as `grader_target`,
|
|
898
|
-
`fallback_targets`,
|
|
899
|
-
|
|
895
|
+
`fallback_targets`, and `workers` as top-level fields on target objects.
|
|
896
|
+
- Remove runner-level request batching from migrated configs. CLI providers are
|
|
897
|
+
invoked once per eval case; throughput batching belongs inside provider
|
|
898
|
+
adapters or CLIs without eval YAML batch configuration.
|
|
900
899
|
|
|
901
900
|
### Verification
|
|
902
901
|
|
|
@@ -961,7 +960,7 @@ assert:
|
|
|
961
960
|
- `type: g-eval` -> `type: llm-rubric`.
|
|
962
961
|
- `type: code-grader`, `code-judge`, `code_grader`, or `code_judge` ->
|
|
963
962
|
`type: script`.
|
|
964
|
-
- `type: llm_judge` or `llm_grader` -> `type: llm-
|
|
963
|
+
- `type: llm_judge` or `llm_grader` -> `type: llm-rubric`.
|
|
965
964
|
- Convert multi-word snake_case deterministic types to kebab-case:
|
|
966
965
|
`is_json` -> `is-json`, `contains_all` -> `contains-all`,
|
|
967
966
|
`starts_with` -> `starts-with`, and so on.
|
|
@@ -990,9 +989,9 @@ v4.42.4 LLM grader docs allowed both:
|
|
|
990
989
|
|
|
991
990
|
```yaml
|
|
992
991
|
assertions:
|
|
993
|
-
- type: llm-
|
|
992
|
+
- type: llm-rubric
|
|
994
993
|
prompt: ./graders/correctness.md
|
|
995
|
-
- type: llm-
|
|
994
|
+
- type: llm-rubric
|
|
996
995
|
prompt: file://graders/correctness.md
|
|
997
996
|
```
|
|
998
997
|
|
|
@@ -20,14 +20,14 @@ Promptfoo parity matrix: https://agentv.dev/docs/reference/promptfoo-parity/
|
|
|
20
20
|
Treat YAML as the canonical portable model. Prefer authoring `.eval.yaml` / `EVAL.yaml` first, then use TypeScript helpers, Python scripts, or executable graders only when they lower to the same fields or when the evaluation logic must actually run code.
|
|
21
21
|
|
|
22
22
|
Eval files define what is tested and how it runs: prompts, datasets, assertions,
|
|
23
|
-
task fixtures, top-level `
|
|
23
|
+
task fixtures, top-level `providers`, and suite run controls. Use field-local file
|
|
24
24
|
refs such as `tests: file://...`, `prompts: file://...`, `default_test:
|
|
25
25
|
file://...`, and `environment: file://...`. String-valued `tests` and string
|
|
26
26
|
entries inside `tests[]` are raw-case refs for direct paths, directories, and
|
|
27
27
|
globs. Run several full eval suites directly with CLI multi-file selection and
|
|
28
28
|
tags. Use scoped `run:` on individual tests only for `threshold`, `repeat`,
|
|
29
|
-
`timeout_seconds`, and legacy `budget_usd`; keep
|
|
30
|
-
`
|
|
29
|
+
`timeout_seconds`, and legacy `budget_usd`; keep provider selection at top-level
|
|
30
|
+
`providers` or CLI `--provider`, put suite budget caps under
|
|
31
31
|
`evaluate_options.budget_usd`, authored concurrency under
|
|
32
32
|
`evaluate_options.max_concurrency`, suite repeat policy under
|
|
33
33
|
`evaluate_options.repeat`, coding-agent testbed setup under `environment`,
|
|
@@ -39,7 +39,7 @@ Use `@agentv/sdk` for TypeScript helper imports. Do not use `@agentv/eval` for n
|
|
|
39
39
|
## Authoring Checklist
|
|
40
40
|
|
|
41
41
|
- Put grading criteria in `assert`, not in test-level `criteria`. Plain assertion strings become an `llm-rubric` grader.
|
|
42
|
-
- Prefer plain assertion strings for semantic checks when the default rubric grader can judge them. Use `type: llm-rubric` for structured criteria, custom prompts, custom grader
|
|
42
|
+
- Prefer plain assertion strings for semantic checks when the default rubric grader can judge them. Use `type: llm-rubric` for structured criteria, custom prompts, custom grader providers, or assertion-level transforms. Use `type: agent-rubric` when the grader itself must be an agent-capable provider that can inspect the workspace. Use `type: script` when grading must execute code.
|
|
43
43
|
- Put reference answers in `tests[].vars.expected_output` or `default_test.vars.expected_output`, and consume them with an explicit assertion such as `type: llm-rubric` with `value: "Matches the reference answer: {{ expected_output }}"`. Do not write criteria, scoring instructions, or "the agent should..." rubric prose as the reference answer.
|
|
44
44
|
- For historical or repo-state evals, materialize the repo through a pinned `environment` setup recipe. Mentioning a SHA only in prompt prose is not enough because the agent needs an actual checkout to inspect.
|
|
45
45
|
|
|
@@ -122,7 +122,8 @@ tests:
|
|
|
122
122
|
|
|
123
123
|
```yaml
|
|
124
124
|
description: Example eval
|
|
125
|
-
|
|
125
|
+
providers:
|
|
126
|
+
- default
|
|
126
127
|
|
|
127
128
|
prompts:
|
|
128
129
|
- "{{ prompt }}"
|
|
@@ -142,7 +143,7 @@ tests:
|
|
|
142
143
|
## Eval File Structure
|
|
143
144
|
|
|
144
145
|
**Required:** `tests` (array or string raw-case path) or `scenarios`
|
|
145
|
-
**Optional:** `name`, `description`, `
|
|
146
|
+
**Optional:** `name`, `description`, `version`, `author`, `tags`, `license`, `requires`, `providers`, `prompts`, `default_test`, `timeout_seconds`, `evaluate_options`, `threshold`, `suite`, `environment`, `env`, `extensions`, `assert`
|
|
146
147
|
|
|
147
148
|
**Test fields:**
|
|
148
149
|
|
|
@@ -151,12 +152,22 @@ tests:
|
|
|
151
152
|
| `id` | yes | Unique identifier |
|
|
152
153
|
| `vars` | yes when the prompt needs row data | Prompt-template variables for this row |
|
|
153
154
|
| `vars.expected_output` | no | Conventional reference-answer var consumed by explicit graders |
|
|
154
|
-
| `assert` | yes | Graders: deterministic checks, `llm-rubric` checks, script graders, or plain string rubric criteria |
|
|
155
|
-
| `execution` | no | Per-case grader/default overrides such as `skip_defaults`;
|
|
155
|
+
| `assert` | yes | Graders: deterministic checks, `llm-rubric` / `agent-rubric` checks, script graders, or plain string rubric criteria |
|
|
156
|
+
| `execution` | no | Per-case grader/default overrides such as `skip_defaults`; provider selection belongs in top-level `providers` or CLI `--provider` |
|
|
156
157
|
| `environment` | no | Per-case coding-agent testbed config (overrides suite-level) |
|
|
157
158
|
| `metadata` | no | Arbitrary key-value pairs passed to setup/teardown scripts |
|
|
158
159
|
| `conversation_id` | no | Thread grouping |
|
|
159
160
|
|
|
161
|
+
**Provider declarations:** AgentV accepts Promptfoo-shaped provider entries
|
|
162
|
+
where they map cleanly: strings such as `openai:gpt-4.1-mini`, object form with
|
|
163
|
+
`id`, `label`, `config`, `env`, `prompts`, `transform`, `delay`, and `inputs`,
|
|
164
|
+
and provider maps such as `{ "openai:gpt-4": { label, config } }`. In object
|
|
165
|
+
form, `id` is the backend/spec and `label` is the stable AgentV selection and
|
|
166
|
+
result identity. AgentV-only `runtime`, provider-local `environment`, and
|
|
167
|
+
provider `hooks` must be explicit extensions; direct Promptfoo runs do not
|
|
168
|
+
execute them, and export must lower supported cases or reject unsupported ones
|
|
169
|
+
clearly.
|
|
170
|
+
|
|
160
171
|
## Prompt Templates and Vars
|
|
161
172
|
|
|
162
173
|
Use top-level `prompts` plus `tests[].vars` for the Promptfoo-compatible canonical
|
|
@@ -167,7 +178,8 @@ repeat samples.
|
|
|
167
178
|
|
|
168
179
|
```yaml
|
|
169
180
|
description: Prompt matrix example
|
|
170
|
-
|
|
181
|
+
providers:
|
|
182
|
+
- default
|
|
171
183
|
|
|
172
184
|
prompts:
|
|
173
185
|
- id: support-chat
|
|
@@ -387,7 +399,7 @@ tests:
|
|
|
387
399
|
|
|
388
400
|
When `assert` is defined, **only the declared graders run**. For
|
|
389
401
|
semantic checks, add plain rubric strings. If you need a custom LLM prompt or
|
|
390
|
-
grader
|
|
402
|
+
grader provider, declare `llm-rubric` explicitly:
|
|
391
403
|
|
|
392
404
|
```yaml
|
|
393
405
|
prompts:
|
|
@@ -539,7 +551,7 @@ Configure via the `assert` array. Multiple graders produce a weighted average sc
|
|
|
539
551
|
type: script
|
|
540
552
|
command: [uv, run, validate.py]
|
|
541
553
|
cwd: ./scripts # optional working directory
|
|
542
|
-
|
|
554
|
+
provider: {} # optional: enable LLM provider proxy (max_calls: 50)
|
|
543
555
|
```
|
|
544
556
|
Contract: stdin JSON -> stdout JSON `{pass, score, reason, checks?: [{text, pass, score?, reason}]}`
|
|
545
557
|
Raw stdin uses snake_case and includes: `input`, `expected_output`, `output` (final answer string), `messages`, `trace`, `trace_summary`, `token_usage`, `cost_usd`, `duration_ms`, `start_time`, `end_time`, `file_changes`, `workspace_path`, `config`
|
|
@@ -560,7 +572,7 @@ See the Script Graders docs for the full stdin/stdout contract.
|
|
|
560
572
|
- name: quality
|
|
561
573
|
type: llm-rubric
|
|
562
574
|
prompt: ./prompts/eval.md # markdown template or command config
|
|
563
|
-
|
|
575
|
+
provider: grader_gpt_5_mini # optional: override the grader provider for this grader
|
|
564
576
|
model: gpt-5-chat # optional model override
|
|
565
577
|
config: # passed to prompt templates as context.config
|
|
566
578
|
strictness: high
|
|
@@ -568,7 +580,7 @@ See the Script Graders docs for the full stdin/stdout contract.
|
|
|
568
580
|
Variables: `{{criteria}}`, `{{input}}`, `{{expected_output}}`, `{{output}}`, `{{metadata}}`, `{{metadata_json}}`, `{{rubrics}}`, `{{rubrics_json}}`, `{{file_changes}}`, `{{tool_calls}}`
|
|
569
581
|
- Markdown templates: use `{{variable}}` syntax
|
|
570
582
|
- TypeScript templates: use `definePromptTemplate(fn)` from `@agentv/sdk`, receives context object with all variables + `config`
|
|
571
|
-
- Use `
|
|
583
|
+
- Use `provider:` to run different `llm-rubric` graders against different named LLM providers in the same eval (useful for grader panels / ensembles)
|
|
572
584
|
|
|
573
585
|
### assert-set
|
|
574
586
|
```yaml
|
|
@@ -813,7 +825,7 @@ import { graders } from '@agentv/sdk';
|
|
|
813
825
|
export function ragFaithfulness() {
|
|
814
826
|
return graders.llmRubric(undefined, {
|
|
815
827
|
name: 'rag-faithfulness',
|
|
816
|
-
|
|
828
|
+
provider: 'grader-provider',
|
|
817
829
|
prompt: 'Grade whether the answer is supported by the retrieved context.',
|
|
818
830
|
});
|
|
819
831
|
}
|
|
@@ -862,7 +874,7 @@ export default defineScriptGrader(({ output, trace }) => {
|
|
|
862
874
|
});
|
|
863
875
|
```
|
|
864
876
|
|
|
865
|
-
Use `defineScriptGrader()` when the custom component is a command-backed grader with explicit score control, check arrays, workspace commands, or LLM calls through a grader
|
|
877
|
+
Use `defineScriptGrader()` when the custom component is a command-backed grader with explicit score control, check arrays, workspace commands, or LLM calls through a grader provider. `defineScriptGrader()` scripts are referenced in YAML with `type: script` and `command: [bun, run, grader.ts]`. Plain Vitest workspace verifier files can use `command: [agentv, eval, graders/check.test.ts]`.
|
|
866
878
|
|
|
867
879
|
### Convention-Based Discovery
|
|
868
880
|
|
|
@@ -56,7 +56,7 @@ Promptfoo normally calls eval custom logic assertions and uses fixed assertion t
|
|
|
56
56
|
|
|
57
57
|
```typescript
|
|
58
58
|
import {
|
|
59
|
-
|
|
59
|
+
createProviderClient,
|
|
60
60
|
defineScriptGrader,
|
|
61
61
|
defineEval,
|
|
62
62
|
type EvalConfig,
|
|
@@ -69,7 +69,7 @@ import {
|
|
|
69
69
|
- `EvalConfig` - Public TypeScript eval authoring type for default-exported `*.eval.ts` and `*.eval.mts` files
|
|
70
70
|
- `defineEval(definition)` - Optional thin helper over the same `EvalConfig` shape
|
|
71
71
|
- `graders` - Helper catalog that returns ordinary AgentV `assert` entries
|
|
72
|
-
- `
|
|
72
|
+
- `createProviderClient()` - Returns LLM proxy client (when `provider: {}` configured)
|
|
73
73
|
- `.invoke({question, systemPrompt})` - Single LLM call
|
|
74
74
|
- `.invokeBatch(requests)` - Batch LLM calls
|
|
75
75
|
- `definePromptTemplate(fn)` - Wraps prompt generation function
|
|
@@ -88,7 +88,7 @@ import { graders, type EvalConfig } from '@agentv/sdk';
|
|
|
88
88
|
function ragFaithfulness() {
|
|
89
89
|
return graders.llmRubric(undefined, {
|
|
90
90
|
name: 'rag-faithfulness',
|
|
91
|
-
|
|
91
|
+
provider: 'grader-provider',
|
|
92
92
|
prompt: 'Grade whether the answer is supported by the retrieved context.',
|
|
93
93
|
});
|
|
94
94
|
}
|
|
@@ -120,7 +120,7 @@ assert:
|
|
|
120
120
|
value: source
|
|
121
121
|
- name: rag-faithfulness
|
|
122
122
|
type: llm-rubric
|
|
123
|
-
|
|
123
|
+
provider: grader-provider
|
|
124
124
|
prompt: Grade whether the answer is supported by the retrieved context.
|
|
125
125
|
```
|
|
126
126
|
|
|
@@ -181,6 +181,6 @@ Derived from test fields (users never author these directly):
|
|
|
181
181
|
| `input` | Full resolved input array (JSON) |
|
|
182
182
|
| `expected_output` | Full resolved expected array (JSON) |
|
|
183
183
|
| `output` | Final answer / scored result string |
|
|
184
|
-
| `messages` | Transcript messages from
|
|
184
|
+
| `messages` | Transcript messages from provider execution |
|
|
185
185
|
|
|
186
186
|
Markdown templates use `{{variable}}` syntax. TypeScript templates receive context object.
|