agentv 5.3.1-next.1 → 5.3.2-next.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +52 -44
- package/dist/{artifact-writer-7NBCOAYC.js → artifact-writer-KJEOROKQ.js} +5 -5
- package/dist/{chunk-ELCJ23K4.js → chunk-6262OKXM.js} +2856 -2404
- package/dist/chunk-6262OKXM.js.map +1 -0
- package/dist/chunk-AQ5BIAXF.js +604 -0
- package/dist/chunk-AQ5BIAXF.js.map +1 -0
- package/dist/chunk-BV5VQLI2.js +2 -0
- package/dist/{chunk-LXBI3SPX.js → chunk-JGSRUJZQ.js} +46 -20
- package/dist/chunk-JGSRUJZQ.js.map +1 -0
- package/dist/chunk-MMDLXYBX.js +721 -0
- package/dist/chunk-MMDLXYBX.js.map +1 -0
- package/dist/{chunk-LKGARI3W.js → chunk-QSK3PC44.js} +812 -570
- package/dist/chunk-QSK3PC44.js.map +1 -0
- package/dist/chunk-TEEXVJWM.js +2 -0
- package/dist/{chunk-RKE7SSET.js → chunk-WIAJ7JAV.js} +186 -64
- package/dist/chunk-WIAJ7JAV.js.map +1 -0
- package/dist/cli.d.ts +1 -0
- package/dist/cli.js +17517 -10
- package/dist/cli.js.map +1 -1
- package/dist/config.d.ts +2 -0
- package/dist/config.js +14 -0
- package/dist/config.js.map +1 -0
- package/dist/contracts-DsZmZLl8.d.ts +742 -0
- package/dist/contracts.d.ts +2 -0
- package/dist/contracts.js +28 -0
- package/dist/contracts.js.map +1 -0
- package/dist/dashboard/assets/{index-CbEMiJSb.js → index-BfqOlLOF.js} +1 -1
- package/dist/dashboard/assets/index-Bh8lpGce.css +1 -0
- package/dist/dashboard/assets/index-oPR3ywQb.js +121 -0
- package/dist/dashboard/index.html +2 -2
- package/dist/{dist-NMXMI5SK.js → dist-A3SGR7TW.js} +16 -16
- package/dist/dist-A3SGR7TW.js.map +1 -0
- package/dist/index.d.ts +4 -0
- package/dist/index.js +137 -18
- package/dist/{interactive-BN527UV3.js → interactive-DL7N2C7K.js} +24 -24
- package/dist/interactive-DL7N2C7K.js.map +1 -0
- package/dist/provider.d.ts +2 -0
- package/dist/provider.js +24 -0
- package/dist/provider.js.map +1 -0
- package/dist/sdk.d.ts +802 -0
- package/dist/sdk.js +145 -0
- package/dist/sdk.js.map +1 -0
- package/dist/skills/agentv-eval-migrations/references/breaking-changes.md +17 -18
- package/dist/skills/agentv-eval-writer/SKILL.md +27 -15
- package/dist/skills/agentv-eval-writer/references/custom-evaluators.md +5 -5
- package/dist/skills/agentv-eval-writer/references/eval.schema.json +5402 -4639
- package/dist/skills/agentv-eval-writer/references/python-helpers.md +2 -2
- package/dist/skills/agentv-eval-writer/references/rubric-evaluator.md +19 -2
- package/dist/templates/.agentv/providers.yaml +42 -0
- package/dist/templates/.env.example +2 -2
- package/dist/{ts-eval-loader-2RFVZHCT-7CZ3DCDD.js → ts-eval-loader-3G5GEAEC-6F52JLPP.js} +3 -3
- package/dist/ts-eval-loader-3G5GEAEC-6F52JLPP.js.map +1 -0
- package/package.json +29 -4
- package/dist/chunk-ASIGJIOJ.js +0 -17993
- package/dist/chunk-ASIGJIOJ.js.map +0 -1
- package/dist/chunk-ELCJ23K4.js.map +0 -1
- package/dist/chunk-LKGARI3W.js.map +0 -1
- package/dist/chunk-LXBI3SPX.js.map +0 -1
- package/dist/chunk-RKE7SSET.js.map +0 -1
- package/dist/dashboard/assets/index-DTA6-l7q.js +0 -121
- package/dist/dashboard/assets/index-D_bokML8.css +0 -1
- package/dist/interactive-BN527UV3.js.map +0 -1
- package/dist/templates/.agentv/targets.yaml +0 -97
- /package/dist/{artifact-writer-7NBCOAYC.js.map → artifact-writer-KJEOROKQ.js.map} +0 -0
- /package/dist/{dist-NMXMI5SK.js.map → chunk-BV5VQLI2.js.map} +0 -0
- /package/dist/{ts-eval-loader-2RFVZHCT-7CZ3DCDD.js.map → chunk-TEEXVJWM.js.map} +0 -0
package/README.md
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
# AgentV
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
> **Deprecated:** AgentV has been replaced by [oh-my-promptfoo](https://github.com/allagentsdev/oh-my-promptfoo) and [Promptfoo](https://github.com/promptfoo/promptfoo). Use oh-my-promptfoo for workspace setup, and migrate your YAML configs to `promptfooconfig.yaml` for Promptfoo.
|
|
4
|
+
|
|
5
|
+
Test AI providers on real repo tasks and measure what actually works.
|
|
4
6
|
|
|
5
7
|
## Why?
|
|
6
8
|
|
|
@@ -10,18 +12,18 @@ Test AI targets on real repo tasks and measure what actually works.
|
|
|
10
12
|
- **Version-controlled** — evals, judges, and results all live in Git
|
|
11
13
|
- **Hybrid graders** — deterministic code checks + LLM-based subjective scoring
|
|
12
14
|
- **CI/CD native** — exit codes, JSONL output, threshold flags for pipeline gating
|
|
13
|
-
- **Any
|
|
15
|
+
- **Any provider** — run against agents, model providers, gateways, replay providers, CLI wrappers, transcript providers, and future app or service wrappers
|
|
14
16
|
|
|
15
17
|
## Core Concepts
|
|
16
18
|
|
|
17
19
|
- **Eval suite / tests** are the task corpus: the prompts, cases, datasets, and reusable field-local files you want to evaluate.
|
|
18
20
|
- **Category** is derived from where the eval lives, such as folder path and file name. Use paths to organize the corpus instead of repeating category labels in every eval.
|
|
19
21
|
- **Environment / fixtures / graders** are task-owned context: host or Docker setup, repos, setup scripts, files, fixtures, deterministic checks, and LLM grading prompts.
|
|
20
|
-
- **
|
|
21
|
-
- **Tags** are run/result grouping labels. `tags.experiment` is the default experiment namespace, such as `with-skills` or `without-skills`; keep suite/category and
|
|
22
|
-
- **Evaluate options** configure eval run behavior such as `max_concurrency`, repeat
|
|
22
|
+
- **Provider** is the configured system under test: an agent, model provider, gateway, replay provider, CLI wrapper, transcript provider, or future app/service wrapper. Each provider entry uses `id` for the backend/spec and optional `label` for the stable AgentV selection and result identity.
|
|
23
|
+
- **Tags** are run/result grouping labels. `tags.experiment` is the default experiment namespace, such as `with-skills` or `without-skills`; keep suite/category and provider/model names out of that tag.
|
|
24
|
+
- **Evaluate options** configure eval run behavior such as `max_concurrency`, repeat sample count, and budgets.
|
|
23
25
|
- **Default test** configures inherited per-test defaults such as score `threshold`.
|
|
24
|
-
- **Run** is one concrete execution of a tagged eval against a resolved
|
|
26
|
+
- **Run** is one concrete execution of a tagged eval against a resolved provider that writes portable artifacts for readers such as Dashboard, compare, and trend.
|
|
25
27
|
|
|
26
28
|
## Quick start
|
|
27
29
|
|
|
@@ -31,12 +33,12 @@ npm install -g agentv
|
|
|
31
33
|
agentv init
|
|
32
34
|
```
|
|
33
35
|
|
|
34
|
-
**2. Configure
|
|
36
|
+
**2. Configure providers and graders** in `.agentv/providers.yaml` — point to the system under test and the reusable grader. Provider `id` names the backend/spec; `label` is the stable selection name used by evals and CLI flags:
|
|
35
37
|
|
|
36
38
|
```yaml
|
|
37
|
-
|
|
38
|
-
- id:
|
|
39
|
-
|
|
39
|
+
providers:
|
|
40
|
+
- id: openai
|
|
41
|
+
label: local-openai
|
|
40
42
|
runtime: host
|
|
41
43
|
config:
|
|
42
44
|
api_format: chat
|
|
@@ -44,9 +46,9 @@ targets:
|
|
|
44
46
|
api_key: "{{ env.LOCAL_OPENAI_PROXY_API_KEY }}"
|
|
45
47
|
model: "{{ env.LOCAL_OPENAI_PROXY_MODEL }}"
|
|
46
48
|
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
49
|
+
- id: openai
|
|
50
|
+
label: local-openai-grader
|
|
51
|
+
runtime: host
|
|
50
52
|
config:
|
|
51
53
|
api_format: chat
|
|
52
54
|
base_url: "{{ env.LOCAL_OPENAI_PROXY_BASE_URL }}"
|
|
@@ -54,7 +56,7 @@ graders:
|
|
|
54
56
|
model: "{{ env.LOCAL_OPENAI_PROXY_MODEL }}"
|
|
55
57
|
|
|
56
58
|
defaults:
|
|
57
|
-
|
|
59
|
+
provider: local-openai
|
|
58
60
|
grader: local-openai-grader
|
|
59
61
|
```
|
|
60
62
|
|
|
@@ -82,7 +84,8 @@ options:
|
|
|
82
84
|
description: Code generation quality
|
|
83
85
|
tags:
|
|
84
86
|
experiment: with-skills
|
|
85
|
-
|
|
87
|
+
providers:
|
|
88
|
+
- local-openai
|
|
86
89
|
evaluate_options:
|
|
87
90
|
max_concurrency: 2
|
|
88
91
|
|
|
@@ -112,29 +115,27 @@ tests:
|
|
|
112
115
|
Plain assertion strings are short-form rubric criteria: AgentV groups them into
|
|
113
116
|
`llm-rubric` and writes grader detail to `grading.json.component_results` for
|
|
114
117
|
the Dashboard. Use explicit `type: llm-rubric` when you need weights, required
|
|
115
|
-
flags, `score_ranges`, a custom grader prompt, a grader
|
|
118
|
+
flags, `score_ranges`, a custom grader prompt, a grader provider, or output
|
|
116
119
|
transforms; use string `value` for free-form rubric checks. Executable graders
|
|
117
120
|
use `type: script`.
|
|
118
121
|
|
|
119
|
-
The
|
|
122
|
+
The provider can be an eval-local object when this eval needs provider settings of its own:
|
|
120
123
|
|
|
121
124
|
```yaml
|
|
122
|
-
description: Code generation quality with eval-local
|
|
125
|
+
description: Code generation quality with eval-local provider settings
|
|
123
126
|
tags:
|
|
124
127
|
experiment: with-skills
|
|
125
|
-
|
|
126
|
-
id:
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
128
|
+
providers:
|
|
129
|
+
- id: openai
|
|
130
|
+
label: local-mini
|
|
131
|
+
runtime: host
|
|
132
|
+
config:
|
|
133
|
+
api_format: chat
|
|
134
|
+
base_url: "{{ env.LOCAL_OPENAI_PROXY_BASE_URL }}"
|
|
135
|
+
api_key: "{{ env.LOCAL_OPENAI_PROXY_API_KEY }}"
|
|
136
|
+
model: gpt-5.4-mini
|
|
134
137
|
evaluate_options:
|
|
135
|
-
repeat:
|
|
136
|
-
count: 2
|
|
137
|
-
strategy: pass_any
|
|
138
|
+
repeat: 2
|
|
138
139
|
|
|
139
140
|
default_test:
|
|
140
141
|
threshold: 0.85
|
|
@@ -148,7 +149,7 @@ tests:
|
|
|
148
149
|
input: Write FizzBuzz in Python
|
|
149
150
|
```
|
|
150
151
|
|
|
151
|
-
`
|
|
152
|
+
`providers: [local-openai]` resolves the configured provider label from `.agentv/providers.yaml` and uses its backend, model, hooks, and provider settings. The object form above defines a full eval-local provider and must include enough provider configuration to run. AgentV records the resolved provider information in run artifacts so results can be audited and replayed. The `tags.experiment` label stays `with-skills` because the condition is unchanged; the model/provider variation belongs to the resolved provider metadata.
|
|
152
153
|
|
|
153
154
|
Use `default_test.threshold` for the inherited per-test pass cutoff. `default_test` can also point at a shared file:
|
|
154
155
|
|
|
@@ -179,7 +180,7 @@ agentv results compare .agentv/results/<baseline-run-id>/.internal/index.jsonl .
|
|
|
179
180
|
|
|
180
181
|
## Results
|
|
181
182
|
|
|
182
|
-
Each run writes a portable bundle directly under `.agentv/results/<run_id>/`. In this example, `tags.experiment: with-skills` names the condition being measured and `
|
|
183
|
+
Each run writes a portable bundle directly under `.agentv/results/<run_id>/`. In this example, `tags.experiment: with-skills` names the condition being measured and `providers: [local-openai]` selects the system under test from `.agentv/providers.yaml`; both are recorded as metadata, not path segments. The `.internal/index.jsonl` file is the portable row index used by scripts, CI, and `agentv results compare`; per-case sidecars include the resolved eval and provider configuration used for the run.
|
|
183
184
|
|
|
184
185
|
```bash
|
|
185
186
|
agentv eval evals/my-eval.eval.yaml
|
|
@@ -192,11 +193,11 @@ Run bundle layout:
|
|
|
192
193
|
.agentv/results/
|
|
193
194
|
├── 2026-06-30T08-30-00-000Z/ # <run_id> — one committed run bundle
|
|
194
195
|
│ ├── summary.json # run rollup: metadata, pass rate, counts, cost
|
|
195
|
-
│ ├── fizzbuzz--a1b2c3d4/ # <result_dir> for one test/
|
|
196
|
+
│ ├── fizzbuzz--a1b2c3d4/ # <result_dir> for one test/provider row
|
|
196
197
|
│ │ ├── summary.json # optional per-case rollup across samples
|
|
197
198
|
│ │ ├── test/ # generated test bundle: frozen inputs for reproducibility
|
|
198
199
|
│ │ │ ├── EVAL.yaml # resolved eval spec
|
|
199
|
-
│ │ │ ├──
|
|
200
|
+
│ │ │ ├── providers.yaml # resolved provider config
|
|
200
201
|
│ │ │ └── graders/ # grader files used
|
|
201
202
|
│ │ └── sample-1/ # one materialized sample
|
|
202
203
|
│ │ ├── result.json # compact sample manifest
|
|
@@ -216,7 +217,7 @@ Run bundle layout:
|
|
|
216
217
|
Use `evaluate()` when your application owns the run:
|
|
217
218
|
|
|
218
219
|
```typescript
|
|
219
|
-
import { evaluate } from '
|
|
220
|
+
import { evaluate } from 'agentv';
|
|
220
221
|
|
|
221
222
|
const { results, summary } = await evaluate({
|
|
222
223
|
experiment: 'with-skills',
|
|
@@ -243,20 +244,27 @@ console.log(`${summary.passed}/${summary.total} passed`);
|
|
|
243
244
|
Use `*.eval.ts` when you want AgentV to run a TypeScript eval config:
|
|
244
245
|
|
|
245
246
|
```typescript
|
|
246
|
-
import type { EvalConfig } from '
|
|
247
|
+
import type { EvalConfig } from 'agentv';
|
|
247
248
|
|
|
248
249
|
const config: EvalConfig = {
|
|
249
250
|
description: 'Code generation quality',
|
|
250
251
|
tags: { experiment: 'with-skills' },
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
252
|
+
providers: [
|
|
253
|
+
{
|
|
254
|
+
id: 'copilot-sdk',
|
|
255
|
+
label: 'copilot',
|
|
256
|
+
config: { model: 'claude-sonnet-4.6' },
|
|
257
|
+
},
|
|
258
|
+
{
|
|
259
|
+
id: 'openai:gpt-5-mini',
|
|
260
|
+
label: 'grader',
|
|
261
|
+
},
|
|
262
|
+
],
|
|
263
|
+
defaults: {
|
|
264
|
+
provider: 'copilot',
|
|
265
|
+
grader: 'grader',
|
|
259
266
|
},
|
|
267
|
+
repeat: 3,
|
|
260
268
|
threshold: 0.8,
|
|
261
269
|
prompts: ['{{ input }}'],
|
|
262
270
|
environment: {
|
|
@@ -4,8 +4,8 @@ import {
|
|
|
4
4
|
buildResultIndexArtifact,
|
|
5
5
|
writeArtifactsFromResults,
|
|
6
6
|
writePerTestArtifacts
|
|
7
|
-
} from "./chunk-
|
|
8
|
-
import "./chunk-
|
|
7
|
+
} from "./chunk-WIAJ7JAV.js";
|
|
8
|
+
import "./chunk-JGSRUJZQ.js";
|
|
9
9
|
import {
|
|
10
10
|
RESULT_INDEX_FILENAME,
|
|
11
11
|
RUN_CONFIG_FILENAME,
|
|
@@ -23,10 +23,10 @@ import {
|
|
|
23
23
|
readRunConfigArtifact,
|
|
24
24
|
writeArtifacts,
|
|
25
25
|
writeInitialRunSummaryArtifact
|
|
26
|
-
} from "./chunk-
|
|
26
|
+
} from "./chunk-6262OKXM.js";
|
|
27
|
+
import "./chunk-M7BUKBAF.js";
|
|
27
28
|
import "./chunk-7BGERE6L.js";
|
|
28
29
|
import "./chunk-PEUTJS7B.js";
|
|
29
|
-
import "./chunk-M7BUKBAF.js";
|
|
30
30
|
import "./chunk-QMRVH5ZP.js";
|
|
31
31
|
export {
|
|
32
32
|
RESULT_INDEX_FILENAME,
|
|
@@ -50,4 +50,4 @@ export {
|
|
|
50
50
|
writeInitialRunSummaryArtifact,
|
|
51
51
|
writePerTestArtifacts
|
|
52
52
|
};
|
|
53
|
-
//# sourceMappingURL=artifact-writer-
|
|
53
|
+
//# sourceMappingURL=artifact-writer-KJEOROKQ.js.map
|