agentv 5.3.1-next.1 → 5.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. package/README.md +52 -44
  2. package/dist/{artifact-writer-7NBCOAYC.js → artifact-writer-KJEOROKQ.js} +5 -5
  3. package/dist/{chunk-ELCJ23K4.js → chunk-6262OKXM.js} +2856 -2404
  4. package/dist/chunk-6262OKXM.js.map +1 -0
  5. package/dist/{chunk-LKGARI3W.js → chunk-6RFRY7X2.js} +812 -570
  6. package/dist/chunk-6RFRY7X2.js.map +1 -0
  7. package/dist/chunk-AQ5BIAXF.js +604 -0
  8. package/dist/chunk-AQ5BIAXF.js.map +1 -0
  9. package/dist/chunk-BV5VQLI2.js +2 -0
  10. package/dist/{chunk-LXBI3SPX.js → chunk-JGSRUJZQ.js} +46 -20
  11. package/dist/chunk-JGSRUJZQ.js.map +1 -0
  12. package/dist/chunk-MMDLXYBX.js +721 -0
  13. package/dist/chunk-MMDLXYBX.js.map +1 -0
  14. package/dist/chunk-TEEXVJWM.js +2 -0
  15. package/dist/{chunk-RKE7SSET.js → chunk-WIAJ7JAV.js} +186 -64
  16. package/dist/chunk-WIAJ7JAV.js.map +1 -0
  17. package/dist/cli.d.ts +1 -0
  18. package/dist/cli.js +17517 -10
  19. package/dist/cli.js.map +1 -1
  20. package/dist/config.d.ts +2 -0
  21. package/dist/config.js +14 -0
  22. package/dist/config.js.map +1 -0
  23. package/dist/contracts-DsZmZLl8.d.ts +742 -0
  24. package/dist/contracts.d.ts +2 -0
  25. package/dist/contracts.js +28 -0
  26. package/dist/contracts.js.map +1 -0
  27. package/dist/dashboard/assets/{index-CbEMiJSb.js → index-BfqOlLOF.js} +1 -1
  28. package/dist/dashboard/assets/index-Bh8lpGce.css +1 -0
  29. package/dist/dashboard/assets/index-oPR3ywQb.js +121 -0
  30. package/dist/dashboard/index.html +2 -2
  31. package/dist/{dist-NMXMI5SK.js → dist-A3SGR7TW.js} +16 -16
  32. package/dist/dist-A3SGR7TW.js.map +1 -0
  33. package/dist/index.d.ts +4 -0
  34. package/dist/index.js +137 -18
  35. package/dist/{interactive-BN527UV3.js → interactive-WPQJDJKZ.js} +24 -24
  36. package/dist/interactive-WPQJDJKZ.js.map +1 -0
  37. package/dist/provider.d.ts +2 -0
  38. package/dist/provider.js +24 -0
  39. package/dist/provider.js.map +1 -0
  40. package/dist/sdk.d.ts +802 -0
  41. package/dist/sdk.js +145 -0
  42. package/dist/sdk.js.map +1 -0
  43. package/dist/skills/agentv-eval-migrations/references/breaking-changes.md +17 -18
  44. package/dist/skills/agentv-eval-writer/SKILL.md +27 -15
  45. package/dist/skills/agentv-eval-writer/references/custom-evaluators.md +5 -5
  46. package/dist/skills/agentv-eval-writer/references/eval.schema.json +5402 -4639
  47. package/dist/skills/agentv-eval-writer/references/python-helpers.md +2 -2
  48. package/dist/skills/agentv-eval-writer/references/rubric-evaluator.md +19 -2
  49. package/dist/templates/.agentv/providers.yaml +42 -0
  50. package/dist/templates/.env.example +2 -2
  51. package/dist/{ts-eval-loader-2RFVZHCT-7CZ3DCDD.js → ts-eval-loader-3G5GEAEC-6F52JLPP.js} +3 -3
  52. package/dist/ts-eval-loader-3G5GEAEC-6F52JLPP.js.map +1 -0
  53. package/package.json +29 -4
  54. package/dist/chunk-ASIGJIOJ.js +0 -17993
  55. package/dist/chunk-ASIGJIOJ.js.map +0 -1
  56. package/dist/chunk-ELCJ23K4.js.map +0 -1
  57. package/dist/chunk-LKGARI3W.js.map +0 -1
  58. package/dist/chunk-LXBI3SPX.js.map +0 -1
  59. package/dist/chunk-RKE7SSET.js.map +0 -1
  60. package/dist/dashboard/assets/index-DTA6-l7q.js +0 -121
  61. package/dist/dashboard/assets/index-D_bokML8.css +0 -1
  62. package/dist/interactive-BN527UV3.js.map +0 -1
  63. package/dist/templates/.agentv/targets.yaml +0 -97
  64. /package/dist/{artifact-writer-7NBCOAYC.js.map → artifact-writer-KJEOROKQ.js.map} +0 -0
  65. /package/dist/{dist-NMXMI5SK.js.map → chunk-BV5VQLI2.js.map} +0 -0
  66. /package/dist/{ts-eval-loader-2RFVZHCT-7CZ3DCDD.js.map → chunk-TEEXVJWM.js.map} +0 -0
package/README.md CHANGED
@@ -1,6 +1,8 @@
1
1
  # AgentV
2
2
 
3
- Test AI targets on real repo tasks and measure what actually works.
3
+ > **Deprecated:** AgentV has been replaced by [oh-my-promptfoo](https://github.com/allagentsdev/oh-my-promptfoo) and [Promptfoo](https://github.com/promptfoo/promptfoo). Use oh-my-promptfoo for workspace setup, and migrate your YAML configs to `promptfooconfig.yaml` for Promptfoo.
4
+
5
+ Test AI providers on real repo tasks and measure what actually works.
4
6
 
5
7
  ## Why?
6
8
 
@@ -10,18 +12,18 @@ Test AI targets on real repo tasks and measure what actually works.
10
12
  - **Version-controlled** — evals, judges, and results all live in Git
11
13
  - **Hybrid graders** — deterministic code checks + LLM-based subjective scoring
12
14
  - **CI/CD native** — exit codes, JSONL output, threshold flags for pipeline gating
13
- - **Any target** — run against agents, model providers, gateways, replay targets, CLI wrappers, transcript providers, and future app or service wrappers
15
+ - **Any provider** — run against agents, model providers, gateways, replay providers, CLI wrappers, transcript providers, and future app or service wrappers
14
16
 
15
17
  ## Core Concepts
16
18
 
17
19
  - **Eval suite / tests** are the task corpus: the prompts, cases, datasets, and reusable field-local files you want to evaluate.
18
20
  - **Category** is derived from where the eval lives, such as folder path and file name. Use paths to organize the corpus instead of repeating category labels in every eval.
19
21
  - **Environment / fixtures / graders** are task-owned context: host or Docker setup, repos, setup scripts, files, fixtures, deterministic checks, and LLM grading prompts.
20
- - **Target** is the system under test: an agent, provider, gateway, replay target, CLI wrapper, transcript provider, or future app/service wrapper. Each eval selects one `target` by configured target `id` or with an eval-local target object.
21
- - **Tags** are run/result grouping labels. `tags.experiment` is the default experiment namespace, such as `with-skills` or `without-skills`; keep suite/category and target/model names out of that tag.
22
- - **Evaluate options** configure eval run behavior such as `max_concurrency`, repeat policy, and budgets.
22
+ - **Provider** is the configured system under test: an agent, model provider, gateway, replay provider, CLI wrapper, transcript provider, or future app/service wrapper. Each provider entry uses `id` for the backend/spec and optional `label` for the stable AgentV selection and result identity.
23
+ - **Tags** are run/result grouping labels. `tags.experiment` is the default experiment namespace, such as `with-skills` or `without-skills`; keep suite/category and provider/model names out of that tag.
24
+ - **Evaluate options** configure eval run behavior such as `max_concurrency`, repeat sample count, and budgets.
23
25
  - **Default test** configures inherited per-test defaults such as score `threshold`.
24
- - **Run** is one concrete execution of a tagged eval against a resolved target that writes portable artifacts for readers such as Dashboard, compare, and trend.
26
+ - **Run** is one concrete execution of a tagged eval against a resolved provider that writes portable artifacts for readers such as Dashboard, compare, and trend.
25
27
 
26
28
  ## Quick start
27
29
 
@@ -31,12 +33,12 @@ npm install -g agentv
31
33
  agentv init
32
34
  ```
33
35
 
34
- **2. Configure targets and graders** in `.agentv/config.yaml` — point to the system under test and the reusable grader. Provider settings live under `config`, and target `id` is the selection name used by evals and CLI flags:
36
+ **2. Configure providers and graders** in `.agentv/providers.yaml` — point to the system under test and the reusable grader. Provider `id` names the backend/spec; `label` is the stable selection name used by evals and CLI flags:
35
37
 
36
38
  ```yaml
37
- targets:
38
- - id: local-openai
39
- provider: openai
39
+ providers:
40
+ - id: openai
41
+ label: local-openai
40
42
  runtime: host
41
43
  config:
42
44
  api_format: chat
@@ -44,9 +46,9 @@ targets:
44
46
  api_key: "{{ env.LOCAL_OPENAI_PROXY_API_KEY }}"
45
47
  model: "{{ env.LOCAL_OPENAI_PROXY_MODEL }}"
46
48
 
47
- graders:
48
- - id: local-openai-grader
49
- provider: openai
49
+ - id: openai
50
+ label: local-openai-grader
51
+ runtime: host
50
52
  config:
51
53
  api_format: chat
52
54
  base_url: "{{ env.LOCAL_OPENAI_PROXY_BASE_URL }}"
@@ -54,7 +56,7 @@ graders:
54
56
  model: "{{ env.LOCAL_OPENAI_PROXY_MODEL }}"
55
57
 
56
58
  defaults:
57
- target: local-openai
59
+ provider: local-openai
58
60
  grader: local-openai-grader
59
61
  ```
60
62
 
@@ -82,7 +84,8 @@ options:
82
84
  description: Code generation quality
83
85
  tags:
84
86
  experiment: with-skills
85
- target: local-openai
87
+ providers:
88
+ - local-openai
86
89
  evaluate_options:
87
90
  max_concurrency: 2
88
91
 
@@ -112,29 +115,27 @@ tests:
112
115
  Plain assertion strings are short-form rubric criteria: AgentV groups them into
113
116
  `llm-rubric` and writes grader detail to `grading.json.component_results` for
114
117
  the Dashboard. Use explicit `type: llm-rubric` when you need weights, required
115
- flags, `score_ranges`, a custom grader prompt, a grader target, or output
118
+ flags, `score_ranges`, a custom grader prompt, a grader provider, or output
116
119
  transforms; use string `value` for free-form rubric checks. Executable graders
117
120
  use `type: script`.
118
121
 
119
- The target can be an eval-local object when this eval needs target settings of its own:
122
+ The provider can be an eval-local object when this eval needs provider settings of its own:
120
123
 
121
124
  ```yaml
122
- description: Code generation quality with eval-local target settings
125
+ description: Code generation quality with eval-local provider settings
123
126
  tags:
124
127
  experiment: with-skills
125
- target:
126
- id: local-mini
127
- provider: openai
128
- runtime: host
129
- config:
130
- api_format: chat
131
- base_url: "{{ env.LOCAL_OPENAI_PROXY_BASE_URL }}"
132
- api_key: "{{ env.LOCAL_OPENAI_PROXY_API_KEY }}"
133
- model: gpt-5.4-mini
128
+ providers:
129
+ - id: openai
130
+ label: local-mini
131
+ runtime: host
132
+ config:
133
+ api_format: chat
134
+ base_url: "{{ env.LOCAL_OPENAI_PROXY_BASE_URL }}"
135
+ api_key: "{{ env.LOCAL_OPENAI_PROXY_API_KEY }}"
136
+ model: gpt-5.4-mini
134
137
  evaluate_options:
135
- repeat:
136
- count: 2
137
- strategy: pass_any
138
+ repeat: 2
138
139
 
139
140
  default_test:
140
141
  threshold: 0.85
@@ -148,7 +149,7 @@ tests:
148
149
  input: Write FizzBuzz in Python
149
150
  ```
150
151
 
151
- `target: local-openai` resolves the configured target id from `.agentv/config.yaml` and uses its provider, model, hooks, and provider settings. The object form above defines a full eval-local target and must include enough provider configuration to run. AgentV records the resolved target information in run artifacts so results can be audited and replayed. The `tags.experiment` label stays `with-skills` because the condition is unchanged; the model/provider variation belongs to the resolved target metadata.
152
+ `providers: [local-openai]` resolves the configured provider label from `.agentv/providers.yaml` and uses its backend, model, hooks, and provider settings. The object form above defines a full eval-local provider and must include enough provider configuration to run. AgentV records the resolved provider information in run artifacts so results can be audited and replayed. The `tags.experiment` label stays `with-skills` because the condition is unchanged; the model/provider variation belongs to the resolved provider metadata.
152
153
 
153
154
  Use `default_test.threshold` for the inherited per-test pass cutoff. `default_test` can also point at a shared file:
154
155
 
@@ -179,7 +180,7 @@ agentv results compare .agentv/results/<baseline-run-id>/.internal/index.jsonl .
179
180
 
180
181
  ## Results
181
182
 
182
- Each run writes a portable bundle directly under `.agentv/results/<run_id>/`. In this example, `tags.experiment: with-skills` names the condition being measured and `target: local-openai` selects the system under test from `.agentv/config.yaml`; both are recorded as metadata, not path segments. The `.internal/index.jsonl` file is the portable row index used by scripts, CI, and `agentv results compare`; per-case sidecars include the resolved eval and target configuration used for the run.
183
+ Each run writes a portable bundle directly under `.agentv/results/<run_id>/`. In this example, `tags.experiment: with-skills` names the condition being measured and `providers: [local-openai]` selects the system under test from `.agentv/providers.yaml`; both are recorded as metadata, not path segments. The `.internal/index.jsonl` file is the portable row index used by scripts, CI, and `agentv results compare`; per-case sidecars include the resolved eval and provider configuration used for the run.
183
184
 
184
185
  ```bash
185
186
  agentv eval evals/my-eval.eval.yaml
@@ -192,11 +193,11 @@ Run bundle layout:
192
193
  .agentv/results/
193
194
  ├── 2026-06-30T08-30-00-000Z/ # <run_id> — one committed run bundle
194
195
  │ ├── summary.json # run rollup: metadata, pass rate, counts, cost
195
- │ ├── fizzbuzz--a1b2c3d4/ # <result_dir> for one test/target row
196
+ │ ├── fizzbuzz--a1b2c3d4/ # <result_dir> for one test/provider row
196
197
  │ │ ├── summary.json # optional per-case rollup across samples
197
198
  │ │ ├── test/ # generated test bundle: frozen inputs for reproducibility
198
199
  │ │ │ ├── EVAL.yaml # resolved eval spec
199
- │ │ │ ├── targets.yaml # resolved target config
200
+ │ │ │ ├── providers.yaml # resolved provider config
200
201
  │ │ │ └── graders/ # grader files used
201
202
  │ │ └── sample-1/ # one materialized sample
202
203
  │ │ ├── result.json # compact sample manifest
@@ -216,7 +217,7 @@ Run bundle layout:
216
217
  Use `evaluate()` when your application owns the run:
217
218
 
218
219
  ```typescript
219
- import { evaluate } from '@agentv/sdk';
220
+ import { evaluate } from 'agentv';
220
221
 
221
222
  const { results, summary } = await evaluate({
222
223
  experiment: 'with-skills',
@@ -243,20 +244,27 @@ console.log(`${summary.passed}/${summary.total} passed`);
243
244
  Use `*.eval.ts` when you want AgentV to run a TypeScript eval config:
244
245
 
245
246
  ```typescript
246
- import type { EvalConfig } from '@agentv/sdk';
247
+ import type { EvalConfig } from 'agentv';
247
248
 
248
249
  const config: EvalConfig = {
249
250
  description: 'Code generation quality',
250
251
  tags: { experiment: 'with-skills' },
251
- target: {
252
- extends: 'copilot-sdk',
253
- model: 'claude-sonnet-4.6',
254
- },
255
- repeat: {
256
- count: 3,
257
- strategy: 'pass_any',
258
- earlyExit: false,
252
+ providers: [
253
+ {
254
+ id: 'copilot-sdk',
255
+ label: 'copilot',
256
+ config: { model: 'claude-sonnet-4.6' },
257
+ },
258
+ {
259
+ id: 'openai:gpt-5-mini',
260
+ label: 'grader',
261
+ },
262
+ ],
263
+ defaults: {
264
+ provider: 'copilot',
265
+ grader: 'grader',
259
266
  },
267
+ repeat: 3,
260
268
  threshold: 0.8,
261
269
  prompts: ['{{ input }}'],
262
270
  environment: {
@@ -4,8 +4,8 @@ import {
4
4
  buildResultIndexArtifact,
5
5
  writeArtifactsFromResults,
6
6
  writePerTestArtifacts
7
- } from "./chunk-RKE7SSET.js";
8
- import "./chunk-LXBI3SPX.js";
7
+ } from "./chunk-WIAJ7JAV.js";
8
+ import "./chunk-JGSRUJZQ.js";
9
9
  import {
10
10
  RESULT_INDEX_FILENAME,
11
11
  RUN_CONFIG_FILENAME,
@@ -23,10 +23,10 @@ import {
23
23
  readRunConfigArtifact,
24
24
  writeArtifacts,
25
25
  writeInitialRunSummaryArtifact
26
- } from "./chunk-ELCJ23K4.js";
26
+ } from "./chunk-6262OKXM.js";
27
+ import "./chunk-M7BUKBAF.js";
27
28
  import "./chunk-7BGERE6L.js";
28
29
  import "./chunk-PEUTJS7B.js";
29
- import "./chunk-M7BUKBAF.js";
30
30
  import "./chunk-QMRVH5ZP.js";
31
31
  export {
32
32
  RESULT_INDEX_FILENAME,
@@ -50,4 +50,4 @@ export {
50
50
  writeInitialRunSummaryArtifact,
51
51
  writePerTestArtifacts
52
52
  };
53
- //# sourceMappingURL=artifact-writer-7NBCOAYC.js.map
53
+ //# sourceMappingURL=artifact-writer-KJEOROKQ.js.map