@tangle-network/agent-eval 0.79.0 → 0.81.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/README.md +101 -169
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/adapters/langchain.d.ts +2 -2
  5. package/dist/adapters/otel.d.ts +4 -4
  6. package/dist/{agent-profile-aSEaJ9Pl.d.ts → agent-profile-D0PBIWlV.d.ts} +14 -2
  7. package/dist/analyst/index.d.ts +10 -10
  8. package/dist/analyst/index.js +3 -3
  9. package/dist/{analyst-t7zZS3TV.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
  10. package/dist/belief-state/index.d.ts +524 -0
  11. package/dist/belief-state/index.js +1862 -0
  12. package/dist/belief-state/index.js.map +1 -0
  13. package/dist/benchmarks/index.d.ts +2 -2
  14. package/dist/calibration-Cpr3WaX3.d.ts +101 -0
  15. package/dist/campaign/index.d.ts +40 -120
  16. package/dist/campaign/index.js +129 -238
  17. package/dist/campaign/index.js.map +1 -1
  18. package/dist/chunk-4DIJWVUT.js +131 -0
  19. package/dist/chunk-4DIJWVUT.js.map +1 -0
  20. package/dist/{chunk-RPLZ4OIB.js → chunk-BABOZOSN.js} +7 -4
  21. package/dist/{chunk-RPLZ4OIB.js.map → chunk-BABOZOSN.js.map} +1 -1
  22. package/dist/{chunk-IHDHUN2X.js → chunk-CVVHBFGN.js} +3 -3
  23. package/dist/chunk-CVVHBFGN.js.map +1 -0
  24. package/dist/{chunk-B26KI423.js → chunk-FZWAFVAA.js} +2 -2
  25. package/dist/{chunk-ITBRCT73.js → chunk-IDVBLYCY.js} +2 -10
  26. package/dist/chunk-IDVBLYCY.js.map +1 -0
  27. package/dist/{chunk-5LVWPNS5.js → chunk-L5G7OUKD.js} +4 -4
  28. package/dist/{chunk-5LVWPNS5.js.map → chunk-L5G7OUKD.js.map} +1 -1
  29. package/dist/chunk-NPCTHQIO.js +91 -0
  30. package/dist/chunk-NPCTHQIO.js.map +1 -0
  31. package/dist/{chunk-GXHLRXDI.js → chunk-OTYQPHPL.js} +4 -4
  32. package/dist/{chunk-6REHLN5J.js → chunk-QS3RBQPI.js} +2 -2
  33. package/dist/{chunk-GWGO2K6Y.js → chunk-RBNA5AZT.js} +2 -2
  34. package/dist/chunk-S42AWHMP.js +697 -0
  35. package/dist/chunk-S42AWHMP.js.map +1 -0
  36. package/dist/chunk-VI2UW6B6.js +162 -0
  37. package/dist/chunk-VI2UW6B6.js.map +1 -0
  38. package/dist/{chunk-CF67I6QY.js → chunk-VIDQF3F5.js} +2 -2
  39. package/dist/{chunk-XXNIODOM.js → chunk-WJL2NJXN.js} +3 -3
  40. package/dist/{chunk-LB2UOI5F.js → chunk-YGYXHNAQ.js} +47 -160
  41. package/dist/chunk-YGYXHNAQ.js.map +1 -0
  42. package/dist/{chunk-KX6F6NCG.js → chunk-Z7VFTS2J.js} +2 -2
  43. package/dist/{chunk-ZPSKPT3V.js → chunk-ZZ2HOPME.js} +5 -2
  44. package/dist/chunk-ZZ2HOPME.js.map +1 -0
  45. package/dist/cli.js +2 -2
  46. package/dist/code-agent-session-BRXmavYv.d.ts +80 -0
  47. package/dist/contract/index.d.ts +132 -18
  48. package/dist/contract/index.js +139 -6
  49. package/dist/contract/index.js.map +1 -1
  50. package/dist/{control-CehLtoET.d.ts → control-GeE8OhpN.d.ts} +1 -1
  51. package/dist/control.d.ts +2 -2
  52. package/dist/governance/index.d.ts +1 -1
  53. package/dist/hosted/index.d.ts +4 -4
  54. package/dist/{index-B1RKber3.d.ts → index-DE3RXAXD.d.ts} +1 -1
  55. package/dist/index.d.ts +79 -288
  56. package/dist/index.js +87 -410
  57. package/dist/index.js.map +1 -1
  58. package/dist/{insight-report-dlpEzQDi.d.ts → insight-report-3ADTfClO.d.ts} +1 -1
  59. package/dist/{kind-factory-DqV2t1Xk.d.ts → kind-factory-CVecZZG_.d.ts} +2 -2
  60. package/dist/{llm-client-DbjLfz-K.d.ts → llm-client-CuUg2Mn3.d.ts} +1 -1
  61. package/dist/meta-eval/index.d.ts +6 -99
  62. package/dist/meta-eval/index.js +7 -76
  63. package/dist/meta-eval/index.js.map +1 -1
  64. package/dist/off-policy-DiwuKKg7.d.ts +132 -0
  65. package/dist/openapi.json +1 -1
  66. package/dist/{outcome-store-D6KWmYvj.d.ts → outcome-store-rnXLEqSn.d.ts} +1 -1
  67. package/dist/pipelines/index.js +2 -2
  68. package/dist/{provenance-CEAJI9rm.d.ts → provenance-B9Q4886D.d.ts} +4 -4
  69. package/dist/{registry-BmEuU94S.d.ts → registry-DrEQ3Luj.d.ts} +2 -2
  70. package/dist/{release-report-CXXZlR8g.d.ts → release-report-hlNtD12q.d.ts} +2 -2
  71. package/dist/reporting.d.ts +6 -6
  72. package/dist/reporting.js +3 -3
  73. package/dist/{researcher-rInLj9De.d.ts → researcher-BLPHBbNV.d.ts} +3 -3
  74. package/dist/rl.d.ts +11 -141
  75. package/dist/rl.js +10 -124
  76. package/dist/rl.js.map +1 -1
  77. package/dist/{rubric-predictive-validity-CWyWWLBg.d.ts → rubric-predictive-validity-CnEl9Jc8.d.ts} +2 -2
  78. package/dist/{run-campaign-OVEZF24D.js → run-campaign-4Y5V5CN3.js} +3 -3
  79. package/dist/{run-improvement-loop-Bgu4C59E.d.ts → run-improvement-loop-D6PZOoQL.d.ts} +2 -2
  80. package/dist/{run-record-sItO5ftF.d.ts → run-record-De9VarXR.d.ts} +1 -1
  81. package/dist/{semantic-concept-judge-Du4ZVyef.d.ts → semantic-concept-judge-DIEgr_6v.d.ts} +6 -6
  82. package/dist/{statistics-B7yCbi9i.d.ts → statistics-CnC1FMbx.d.ts} +3 -7
  83. package/dist/{store-GmBE2pZZ.d.ts → store-C1YxJDEK.d.ts} +1 -1
  84. package/dist/{summary-report-BTaXq1TS.d.ts → summary-report-Db0dDSWP.d.ts} +1 -1
  85. package/dist/traces.d.ts +5 -5
  86. package/dist/{types-DRvV0zRo.d.ts → types-Cu3u_x59.d.ts} +3 -3
  87. package/dist/{types-QHG0KnkF.d.ts → types-D7lLRYe9.d.ts} +2 -2
  88. package/dist/wire/index.js +2 -2
  89. package/dist/workflow/index.d.ts +7 -7
  90. package/dist/workflow/index.js +1 -1
  91. package/docs/concepts.md +1 -0
  92. package/docs/research/belief-state-agent-eval-roadmap.md +590 -0
  93. package/docs/research/research-roadmap.md +1 -0
  94. package/docs/self-improvement-map.md +111 -0
  95. package/package.json +7 -2
  96. package/dist/chunk-IHDHUN2X.js.map +0 -1
  97. package/dist/chunk-ITBRCT73.js.map +0 -1
  98. package/dist/chunk-LB2UOI5F.js.map +0 -1
  99. package/dist/chunk-ZPSKPT3V.js.map +0 -1
  100. /package/dist/{chunk-B26KI423.js.map → chunk-FZWAFVAA.js.map} +0 -0
  101. /package/dist/{chunk-GXHLRXDI.js.map → chunk-OTYQPHPL.js.map} +0 -0
  102. /package/dist/{chunk-6REHLN5J.js.map → chunk-QS3RBQPI.js.map} +0 -0
  103. /package/dist/{chunk-GWGO2K6Y.js.map → chunk-RBNA5AZT.js.map} +0 -0
  104. /package/dist/{chunk-CF67I6QY.js.map → chunk-VIDQF3F5.js.map} +0 -0
  105. /package/dist/{chunk-XXNIODOM.js.map → chunk-WJL2NJXN.js.map} +0 -0
  106. /package/dist/{chunk-KX6F6NCG.js.map → chunk-Z7VFTS2J.js.map} +0 -0
  107. /package/dist/{run-campaign-OVEZF24D.js.map → run-campaign-4Y5V5CN3.js.map} +0 -0
package/CHANGELOG.md CHANGED
@@ -4,6 +4,20 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
4
4
 
5
5
  ---
6
6
 
7
+ ## [0.81.0] — 2026-06-05 — eval-campaign scaffold prep primitives
8
+
9
+ ### Added
10
+
11
+ - **`aggregateJudgeVerdicts<D>` (root).** Generic judge-ensemble reducer: fan out N uncorrelated judges, mean each rubric dimension over the SURVIVORS, report the inter-rater disagreement spread, sum cost. Replaces the same reduction hand-rolled in legal (`aggregateEnsemble`), creative (`production-loop/judges.ts`), and tax (`judge-ensemble.ts`). Fail-loud: a failed judge (`perDimension: null`) is recorded in `failedJudges`, never folded into a zero; all-failed throws; a failed judge's cost is still summed. Composite reuses `weightedComposite`.
12
+ - **`createTokenRecallChecker` (root).** The deterministic, no-LLM `CorrectnessChecker` — sibling of `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its content is substantive and recalls ≥ `minRecall` of the requirement title's significant tokens. The default completion gate for apps/tests without an LLM judge.
13
+ - **`ErrorCluster` (root + `/analyst`).** The failure-cluster element type is now a named export, so consumers import it instead of deriving `DatasetOverview['error_clusters'][number]`.
14
+
15
+ ### Fixed
16
+
17
+ - **Lint drift + non-executable pre-commit hook.** `.husky/pre-commit` was tracked `100644`, so the hook silently no-op'd and unformatted code reached `main`; marked executable and reformatted the drift.
18
+
19
+ ---
20
+
7
21
  ## [0.72.3] — 2026-06-01 — workflow trace hardening and driver backtests
8
22
 
9
23
  ### Added
package/README.md CHANGED
@@ -1,197 +1,106 @@
1
1
  # `@tangle-network/agent-eval`
2
2
 
3
- **Ship better agent prompts with statistical confidence.** One function call returns a decision packet: lift CI, judge calibration, contamination check, failure clusters, cost-quality Pareto, and a ranked action list. Same shape whether you've got a closed improvement loop or just production logs.
3
+ Evaluate and improve AI agents from the runs they already produce.
4
+
5
+ `agent-eval` turns agent outputs, traces, judge scores, and production feedback into a decision packet: did this change help, what failed, what should ship, and what needs more data?
4
6
 
5
7
  [![npm](https://img.shields.io/npm/v/@tangle-network/agent-eval.svg)](https://www.npmjs.com/package/@tangle-network/agent-eval)
6
8
  [![pypi](https://img.shields.io/pypi/v/agent-eval-rpc.svg)](https://pypi.org/project/agent-eval-rpc/)
7
9
  [![tests](https://github.com/tangle-network/agent-eval/actions/workflows/ci.yml/badge.svg)](https://github.com/tangle-network/agent-eval/actions/workflows/ci.yml)
8
10
  [![license: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](./LICENSE)
9
11
 
10
- > TypeScript first-class, Python (`agent-eval-rpc`) speaks the same wire protocol, hosted-tier-friendly, MIT, self-hostable, no SaaS dependency.
12
+ Use it when you need to:
11
13
 
12
- ---
14
+ - compare a candidate agent/prompt/model against a baseline,
15
+ - turn production traces or human feedback into eval results,
16
+ - run a gated self-improvement loop,
17
+ - explain failures by cluster, cost, judge disagreement, and release risk.
13
18
 
14
- ## Table of contents
15
-
16
- - [What you get back](#what-you-get-back-the-decision-packet)
17
- - [Quick start](#quick-start)
18
- - [Closed loop — `selfImprove()`](#closed-loop--selfimprove)
19
- - [Observed runs — `analyzeRuns()`](#observed-runs--analyzeruns)
20
- - [Existing data — intake adapters](#existing-data--intake-adapters)
21
- - [How it compares](#how-it-compares)
22
- - [Customer journeys](#customer-journeys)
23
- - [Subpath entry points](#subpath-entry-points)
24
- - [Concepts + design](#concepts--design)
25
- - [Hosted tier](#hosted-tier)
26
- - [Install + run](#install--run)
27
- - [Stability + versioning](#stability--versioning)
28
- - [License](#license)
19
+ It is a library, not a SaaS requirement. TypeScript is first-class; Python can call the same wire protocol through `agent-eval-rpc`.
29
20
 
30
21
  ---
31
22
 
32
- ## What you get back: the decision packet
23
+ ## Install
33
24
 
34
- Whether you call `selfImprove()` (closed loop) or `analyzeRuns()` (observed runs), the report has the same shape. Here's a real one, abridged:
35
-
36
- ```jsonc
37
- {
38
- "n": 80, // runs analyzed
39
- "composite": { // distributional summary
40
- "mean": 0.62, "p50": 0.65, "p95": 0.88, "stddev": 0.17,
41
- "histogram": [/* 12 bins */]
42
- },
43
- "lift": { // paired bootstrap
44
- "baselineMean": 0.58, "candidateMean": 0.65,
45
- "delta": 0.07,
46
- "ci95": [0.04, 0.10], // 95% CI on the delta
47
- "pValue": 0.0008, // paired-t
48
- "cohensD": 0.41,
49
- "n": 40,
50
- "mde": 0.06, // min detectable effect at 80% power
51
- "requiredN": 38 // n needed to detect observed delta
52
- },
53
- "judges": { // per-judge calibration
54
- "domain-expert": { "n": 80, "meanScore": 0.64 },
55
- "helpfulness-llm": { "n": 80, "meanScore": 0.61 }
56
- },
57
- "interRater": { // multi-rater agreement
58
- "raters": 3, "jointlyRated": 80, "kappa": 0.71,
59
- "disagreementCases": [/* top 20 ranked by spread */]
60
- },
61
- "costQuality": { // cost-vs-quality
62
- "cost": { "mean": 0.024, "p95": 0.041, /* ... */ },
63
- "pareto": { /* ParetoFigureSpec the dashboard renders */ }
64
- },
65
- "failureClusters": { // when an AnalystRegistry is wired
66
- "totalFailures": 11,
67
- "clusters": [
68
- { "name": "off-topic-drift", "share": 0.45, "exemplars": ["run-12", "run-19"] },
69
- { "name": "over-confidence", "share": 0.27, "exemplars": ["run-3"] },
70
- { "name": "format-mismatch", "share": 0.18, "exemplars": ["run-41"] }
71
- ]
72
- },
73
- "contamination": { "leaks": 0, "holdoutAuditPassed": true },
74
- "outcomeCorrelation": { // when downstream metric supplied
75
- "metric": "engagement_rate", "n": 80,
76
- "pearson": 0.72, "spearman": 0.69,
77
- "rewardModel": { "intercept": 0.04, "slope": 1.93, "r2": 0.52 }
78
- },
79
- "release": {
80
- "status": "pass",
81
- "axes": [
82
- { "name": "quality-lift", "status": "pass" },
83
- { "name": "contamination", "status": "pass" },
84
- { "name": "composite-distribution","status": "pass" }
85
- ]
86
- },
87
- "recommendations": [
88
- { "priority": "critical", "kind": "ship",
89
- "title": "Ship — lift 0.070 (95% CI 0.040..0.100)",
90
- "detail": "Holdout lift exceeds threshold 0.02 with 95% bootstrap confidence (n=40, p=0.0008, d=0.41)." },
91
- { "priority": "high", "kind": "investigate",
92
- "title": "Top failure cluster: off-topic-drift (45% of failures)",
93
- "detail": "11 runs failed. Drill into exemplars run-12 / run-19 to identify the pattern." }
94
- ]
95
- }
25
+ ```sh
26
+ pnpm add @tangle-network/agent-eval
96
27
  ```
97
28
 
98
- The `recommendations` array is the human-readable layer; everything above it is the evidence. Read the recs, act on them, the numbers are the proof.
29
+ Python clients can use the RPC package:
30
+
31
+ ```sh
32
+ pip install agent-eval-rpc
33
+ ```
99
34
 
100
35
  ---
101
36
 
102
37
  ## Quick start
103
38
 
104
- ### Closed loop `selfImprove()`
39
+ ### 1. Analyze runs you already have
105
40
 
106
- You have scenarios, a dispatch, judges, and want the loop to propose better prompts + tell you which to ship.
41
+ Start here if you already have production logs, benchmark rows, human ratings, or agent run records.
107
42
 
108
43
  ```ts
109
- import { selfImprove } from '@tangle-network/agent-eval/contract'
44
+ import { analyzeRuns } from '@tangle-network/agent-eval/contract'
110
45
 
111
- const result = await selfImprove({
112
- scenarios, // your scenario corpus
113
- dispatch: async ({ scenario }) => // your agent — anything that returns an artifact
114
- await myAgent.run(scenario),
115
- judges: [myJudge], // any JudgeConfig — LLM, rule, ensemble
116
- baselineSurface: { systemPrompt: currentPrompt },
46
+ const report = await analyzeRuns({
47
+ runs, // RunRecord[]
48
+ baselineRuns,
117
49
  })
118
50
 
119
- result.gateDecision // 'ship' | 'hold' | 'need_more_work' | ...
120
- result.lift // raw delta on holdout
121
- result.insight // the full decision packet above
51
+ console.log(report.recommendations)
52
+ console.log(report.lift)
53
+ console.log(report.failureClusters)
122
54
  ```
123
55
 
124
- ### Observed runs `analyzeRuns()`
56
+ The output includes score distributions, lift confidence intervals, failure modes, cost-quality tradeoffs, judge agreement, contamination checks, and release recommendations when the input supports them.
57
+
58
+ ### 2. Run a gated improvement loop
125
59
 
126
- You don't have a closed loop yet — you have observed runs (production traces, an approve/reject corpus, a CSV gold set). Same report shape, no agent invocation.
60
+ Use this when you have scenarios, a runnable agent, and judges.
127
61
 
128
62
  ```ts
129
- import { analyzeRuns } from '@tangle-network/agent-eval/contract'
63
+ import { selfImprove } from '@tangle-network/agent-eval/contract'
130
64
 
131
- const report = await analyzeRuns({
132
- runs, // RunRecord[]
133
- outcomeSignal: { // optional closes the loop on real outcomes
134
- metric: 'engagement_rate',
135
- valueByRunId: enrichedFromProd,
136
- },
137
- canaryScenarios, // optional — contamination probe
138
- analyst: myAnalystRegistry, // optional — AI-powered failure clustering
65
+ const result = await selfImprove({
66
+ scenarios,
67
+ dispatch: async ({ scenario }) => myAgent.run(scenario),
68
+ judges: [myJudge],
69
+ baselineSurface: { systemPrompt: currentPrompt },
139
70
  })
140
71
 
141
- report.recommendations // ranked actions
142
- report.failureClusters // grouped failure modes
143
- report.outcomeCorrelation // judge↔outcome correlation + linear reward model
72
+ console.log(result.gateDecision)
73
+ console.log(result.winnerSurface)
74
+ console.log(result.insight.recommendations)
144
75
  ```
145
76
 
146
- ### Existing data intake adapters
77
+ `selfImprove()` evaluates candidates on held-out scenarios before recommending a winner.
147
78
 
148
- You have data already. Don't reshape it — pipe it through an adapter.
79
+ ### 3. Adapt existing data
149
80
 
150
81
  ```ts
151
- import {
152
- fromFeedbackTable,
153
- fromOtelSpans,
154
- analyzeRuns,
155
- } from '@tangle-network/agent-eval/contract'
82
+ import { analyzeRuns, fromFeedbackTable, fromOtelSpans } from '@tangle-network/agent-eval/contract'
156
83
 
157
- // Multi-rater approve/reject (Obsidian tags, Sheets, CSV, Postgres).
158
84
  const { runs, raterScores } = fromFeedbackTable({
159
- ratings: parseYourFeedbackTable(), // Array<{ runId, rater, rating }>
85
+ ratings: parseYourFeedbackTable(),
160
86
  })
161
- await analyzeRuns({ runs, raterScores })
162
87
 
163
- // Production OTel traces group by tangle.runId or traceId.
164
- const runs2 = fromOtelSpans({ spans: yourOtelStream })
165
- await analyzeRuns({ runs: runs2 })
166
- ```
88
+ const traceRuns = fromOtelSpans({ spans: yourOtelSpans })
167
89
 
168
- Both intake adapters preserve every signal in the source — multi-rater scores stay rater-keyed so the report can compute inter-rater agreement and surface the disagreement triage list.
90
+ await analyzeRuns({ runs: [...runs, ...traceRuns], raterScores })
91
+ ```
169
92
 
170
93
  ---
171
94
 
172
- ## How it compares
173
-
174
- | | LangSmith | Braintrust | Phoenix | **agent-eval** |
175
- |---|:---:|:---:|:---:|:---:|
176
- | Closed-loop self-improvement | ✱ human-in-loop | ✱ experiment-driven | — | ✓ autonomous + gated |
177
- | Statistical lift CI (paired bootstrap) | — | partial | — | ✓ |
178
- | Judge calibration + bias detection | — | — | — | ✓ |
179
- | Inter-rater agreement + disagreement triage | — | — | — | ✓ |
180
- | Contamination / canary check | — | — | — | ✓ |
181
- | AI-driven failure clustering | partial | — | partial | ✓ |
182
- | Cost-quality Pareto | — | — | — | ✓ |
183
- | Multi-language clients (TS + Python) | TS only | TS only | TS + Py | ✓ TS + Py |
184
- | Self-hostable / no-SaaS option | — | — | OSS | ✓ MIT, OSS |
185
- | Substrate vs SaaS shape | SaaS | SaaS | OSS server | **library** |
186
- | Hosted tier (optional) | required | required | optional | optional |
95
+ ## Core concepts
187
96
 
188
- Position: agent-eval is the **substrate** (one library, decision-grade output) the others are SaaS *around* the substrate. If you want a closed loop that ships your prompt under statistical confidence, you call agent-eval. If you want a dashboard rendered from your data, you pipe agent-eval into the hosted tier or your own renderer.
97
+ - **RunRecord**: the durable row for one agent run: model, prompt/config hashes, split, cost, tokens, outcome.
98
+ - **Scenario**: one task or case the agent attempts.
99
+ - **Judge**: a scoring function, rule-based or model-based.
100
+ - **InsightReport**: the decision packet returned by `analyzeRuns()` and embedded in `selfImprove()`.
101
+ - **Gate**: the policy that decides `ship`, `hold`, or `need_more_data`.
189
102
 
190
- ---
191
-
192
- ## Customer journeys
193
-
194
- Three runnable examples — each is self-contained, each shows the actual output.
103
+ ## Examples
195
104
 
196
105
  | Journey | Example | Who it's for |
197
106
  |---|---|---|
@@ -207,28 +116,55 @@ Each example: `README.md` + a single `index.ts` runnable via `pnpm tsx`. Prints
207
116
 
208
117
  | Subpath | What it gives you |
209
118
  |---|---|
210
- | `@tangle-network/agent-eval/contract` | **The headline surface.** `selfImprove`, `analyzeRuns`, `runImprovementLoop`, `runCampaign`, `runEval`, `diffRuns`, intake adapters (`fromFeedbackTable`, `fromOtelSpans`), drivers (`gepaDriver`, `evolutionaryDriver`), gates (`defaultProductionGate`, `heldOutGate`, `composeGate`), storage. **New code starts here.** |
211
- | `@tangle-network/agent-eval/hosted` | Hosted-tier wire-format types + `createHostedClient` to ship eval-run events + trace spans to any orchestrator speaking the spec |
212
- | `@tangle-network/agent-eval/adapters/otel` | `createOtelBridge` — forwards OpenTelemetry-shape spans into the hosted-tier ingest |
213
- | `@tangle-network/agent-eval/adapters/langchain` | LangChain runnable `Dispatch` adapter |
214
- | `@tangle-network/agent-eval/adapters/http` | `httpDispatch` + `runDispatchServer` for distributed campaigns across machines |
215
- | `@tangle-network/agent-eval/campaign` | Lower-level campaign primitives (storage, drivers, types) |
216
- | `@tangle-network/agent-eval/multishot` | N-shot persona × shot matrix runner |
217
- | `@tangle-network/agent-eval/control` | Agent control loop primitives (`runAgentControlLoop`, action policy, propose/review) |
218
- | `@tangle-network/agent-eval/traces` | Trace stores, emitters, OTLP-JSONL replay |
219
- | `@tangle-network/agent-eval/reporting` | Release confidence, paired stats, sequential e-values, launch reports |
220
- | `@tangle-network/agent-eval/rl` | RL bridge verifiable rewards, preferences, OPE, PRM, tournaments, contamination, compute curves, auto-research |
221
- | `@tangle-network/agent-eval/matrix` | N-axis cartesian over substrate types |
222
- | `@tangle-network/agent-eval/wire` | HTTP/RPC server + Zod schemas (same protocol the Python client speaks) |
223
- | `@tangle-network/agent-eval/benchmarks` | Benchmark adapter contracts and reference wrappers |
224
-
225
- The root export remains available for backward compatibility; new code should prefer focused subpaths. Anything under `/rl`, `/pipelines`, `/meta-eval`, `/prm`, or `/builder-eval` is **only** reachable via its subpath.
119
+ | `…/contract` | **The headline, frozen surface — new code starts here.** `selfImprove`, `analyzeRuns`, `runEval`, `runCampaign`, `runImprovementLoop`, `diffRuns`; intake adapters (`fromFeedbackTable`, `fromOtelSpans`); drivers (`gepaDriver`, `evolutionaryDriver`); gates (`defaultProductionGate`, `heldOutGate`, `paretoSignificanceGate`, `composeGate`); the deployment-outcome store; storage; and the five core types `Scenario` / `Dispatch` / `JudgeConfig` / `Mutator` / `Gate`. |
120
+ | `…/hosted` | `createHostedClient` / `hostedClientFromEnv` + the wire types to ship eval-run events + trace spans to a hosted orchestrator (ours or your own implementation of the spec) |
121
+ | `…/adapters/otel` | `createOtelBridge` — forwards OpenTelemetry-shape spans into the hosted-tier ingest, no `@opentelemetry/*` dependency |
122
+ | `…/adapters/langchain` | Wrap any LangChain `Runnable` as a `Dispatch` (or `JudgeConfig`), no `@langchain/core` peer dep |
123
+ | `…/adapters/http` | `httpDispatch` + `runDispatchServer` run a campaign's worker on another machine (multi-region, driver-as-a-service) |
124
+ | `…/campaign` | **The measurement + improvement engine** (`@experimental`): `runProfileMatrix`, `compareDrivers`, every driver (`gepaDriver`, `haloDriver`, `skillOptDriver`, `aceDriver`, `memoryCurationDriver`, …), the gates, storage backends, and loop provenance. `/contract` re-exports the stable subset. |
125
+ | `…/rl` | RL bridge from eval artifacts to training signal: verifiable rewards, preferences, OPE, PRM, tournaments, contamination, compute curves, plus the durable corpus + `buildRlDataset` / datasheet bundle |
126
+ | `…/reporting` | Release-decision statistics: `pairedBootstrap`, `benjaminiHochberg`, anytime-valid sequential e-values, `evaluateReleaseConfidence`, and the report renderers |
127
+ | `…/analyst` | The trace-analyst surface: `AnalystRegistry` + `buildDefaultAnalystRegistry` (run the failure-clustering panel), `FindingsStore`, and the LLM chat transports |
128
+ | `…/traces` | Trace stores + emitters, OTLP-JSONL deterministic replay, `analyzeTraces`, and the `traceAnalystOnRunComplete` hook |
129
+ | `…/control` | Agent control loop: `runAgentControlLoop` (observe validate decide → act), action policy, propose/review |
130
+ | `…/matrix` | `runAgentMatrix` — an N-axis cartesian over caller-supplied substrate values, per-axis pass/score/cost/duration |
131
+ | `…/multishot` | N-shot persona × shot matrix runner (`runMultishot` / `runMultishotMatrix`) |
132
+ | `…/wire` | The cross-language HTTP/RPC server + Zod schemas (the source-of-truth protocol the Python client speaks) + the built-in rubric registry |
133
+ | `…/benchmarks` | `BenchmarkAdapter` contract + `deterministicSplit` + the bundled `routing` reference benchmark |
134
+
135
+ **Specialized surfaces** (subpath-only): `…/prm` (process-reward grading + best-of-N), `…/meta-eval` (judge calibration + the deployment-outcome store), `…/pipelines` (trace-diagnostic views: budget breach, failure cluster, stuck loop, …), `…/governance` (EU AI Act / NIST AI RMF / SOC2 reports), `…/knowledge` (knowledge-readiness gating before a run), `…/builder-eval` (code-generator three-layer eval), `…/storyboard` (trace → watchable replay), `…/authenticity` (anti-Goodhart "real or convincing BS" scorer over produced files), `…/workflow` (workflow-trace eval + partner export), `…/telemetry` (Workers-safe telemetry client).
136
+
137
+ The root export remains available for backward compatibility; new code should prefer the focused subpaths above — `/contract` first.
138
+
139
+ ---
140
+
141
+ ## Composition with the stack
142
+
143
+ agent-eval is the bottom of the layering: consumers depend on it, it depends on none of them.
144
+
145
+ ```
146
+ agent-runtime Runs agents (chat turns, one-shot tasks, multi-attempt loops), captures every
147
+ run as a trace, and calls optimizePrompt / runImprovementLoop. Produces the
148
+ RunRecords + traces agent-eval scores. Depends on agent-eval.
149
+
150
+ agent-eval selfImprove, analyzeRuns, runCampaign + drivers (gepaDriver, …), the gates
151
+ (this repo) (heldOutGate, defaultProductionGate, paretoSignificanceGate), the InsightReport
152
+ decision packet, the RL bridge, the wire protocol. Depends on neither consumer.
153
+
154
+ agent-knowledge proposeKnowledgeWrites / applyKnowledgeWriteBlocks. agent-eval's analyst findings
155
+ feed it; the knowledge gate consumes them. Depends on agent-eval.
156
+
157
+ sandbox AgentProfile, Sandbox.create, streamPrompt. The execution surface the runtime's
158
+ loops run on; agent-eval scores what comes back.
159
+ ```
160
+
161
+ The rule: **agent-eval has zero upward dependencies on a consumer.** A concept that makes sense *without* a running agent loop — a verdict, a run record, a scenario, a judge score — is substrate and lives here; a runtime-shaped one (a sandbox profile, a validation context with an abort signal) lives in agent-runtime. When in doubt, lean substrate.
226
162
 
227
163
  ---
228
164
 
229
165
  ## Concepts + design
230
166
 
231
- - [`docs/concepts.md`](./docs/concepts.md) — five types, three top-level functions, the layering rule, the wire protocol contract
167
+ - [`docs/concepts.md`](./docs/concepts.md) — the three top-level functions, the layering rule, and the wire-protocol contract (the five core contract types are documented in the `/contract` barrel itself)
232
168
  - [`docs/insight-report.md`](./docs/insight-report.md) — annotated walkthrough of every section of the decision packet
233
169
  - [`docs/customer-journeys.md`](./docs/customer-journeys.md) — three end-to-end journeys with code + expected output
234
170
  - [`docs/adapters-observability.md`](./docs/adapters-observability.md) — composing agent-eval with LangSmith, Langfuse, Phoenix, OpenLLMetry, TraceAI
@@ -259,13 +195,7 @@ The substrate runs the loop in your process. Only the eval-run events + (optiona
259
195
 
260
196
  ---
261
197
 
262
- ## Install + run
263
-
264
- ```sh
265
- pnpm add @tangle-network/agent-eval
266
- # or, from Python:
267
- pip install agent-eval-rpc
268
- ```
198
+ ## Development
269
199
 
270
200
  Run an example:
271
201
 
@@ -287,7 +217,9 @@ pnpm test
287
217
 
288
218
  ## Stability + versioning
289
219
 
290
- Public exports carry JSDoc stability markers visible in IDE hover + `.d.ts`:
220
+ The `/contract` surface is the **stability contract**: its barrel freezes the API — a `0.x` minor only *adds*; nothing there changes shape or disappears. Depend on `/contract` (and the documented subpaths) rather than the root barrel.
221
+
222
+ In the deeper subpaths, `@stable` / `@experimental` JSDoc markers (visible in IDE hover + `.d.ts`) call out what may still move — most granularly in `/rl` (tagged per export) and `/campaign` (whole barrel `@experimental`, since `/contract` re-exports only its settled subset).
291
223
 
292
224
  | Tag | Meaning |
293
225
  |---|---|
@@ -1,5 +1,5 @@
1
- import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-QHG0KnkF.js';
2
- import '../run-record-sItO5ftF.js';
1
+ import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-D7lLRYe9.js';
2
+ import '../run-record-De9VarXR.js';
3
3
  import '../errors-Dwqw-T_m.js';
4
4
  import '../schema-m0gsnbt3.js';
5
5
 
@@ -1,5 +1,5 @@
1
- import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-QHG0KnkF.js';
2
- import '../run-record-sItO5ftF.js';
1
+ import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-D7lLRYe9.js';
2
+ import '../run-record-De9VarXR.js';
3
3
  import '../errors-Dwqw-T_m.js';
4
4
  import '../schema-m0gsnbt3.js';
5
5
 
@@ -1,10 +1,10 @@
1
1
  import { TraceSpanEvent, HostedClient } from '../hosted/index.js';
2
- import '../types-QHG0KnkF.js';
3
- import '../run-record-sItO5ftF.js';
2
+ import '../types-D7lLRYe9.js';
3
+ import '../run-record-De9VarXR.js';
4
4
  import '../errors-Dwqw-T_m.js';
5
5
  import '../schema-m0gsnbt3.js';
6
- import '../insight-report-dlpEzQDi.js';
7
- import '../summary-report-BTaXq1TS.js';
6
+ import '../insight-report-3ADTfClO.js';
7
+ import '../summary-report-Db0dDSWP.js';
8
8
  import '../failure-cluster-CL7IVgkJ.js';
9
9
  import '../store-CKUAgsJz.js';
10
10
  import '../judge-calibration-DilmB3Ml.js';
@@ -1,5 +1,5 @@
1
1
  import { A as AgentEvalError } from './errors-Dwqw-T_m.js';
2
- import { R as RunRecord } from './run-record-sItO5ftF.js';
2
+ import { R as RunRecord } from './run-record-De9VarXR.js';
3
3
  import { TCloud } from '@tangle-network/tcloud';
4
4
 
5
5
  /**
@@ -258,6 +258,18 @@ declare function parseCorrectnessResponse(raw: string): {
258
258
  * fulfil a requirement — the artifact must BE the deliverable.
259
259
  */
260
260
  declare function createLlmCorrectnessChecker(tc: TCloud, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
261
+ /**
262
+ * Deterministic `CorrectnessChecker` — the no-LLM counterpart to
263
+ * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its
264
+ * content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`
265
+ * of the requirement title's significant tokens. No network — the default gate
266
+ * for apps and tests without an LLM judge. Pass to `verifyCompletion` as the
267
+ * checker.
268
+ */
269
+ declare function createTokenRecallChecker(opts?: {
270
+ minRecall?: number;
271
+ minContentLength?: number;
272
+ }): CorrectnessChecker;
261
273
 
262
274
  /**
263
275
  * Produced-state extraction — normalize a run's runtime event stream into the
@@ -361,4 +373,4 @@ interface AgentProfile {
361
373
  */
362
374
  declare function agentProfileHash(profile: AgentProfile): string;
363
375
 
364
- export { type AgentProfile as A, type BackendIntegrityReport as B, type CompletionRequirement as C, type LlmCorrectnessCheckerOpts as L, type ProducedState as P, type RuntimeEventLike as R, type SatisfiedBy as S, type TaskGold as T, type ValidationContext as V, type CompletionVerdict as a, type CorrectnessChecker as b, type Artifact as c, type ArtifactEventLike as d, type ArtifactValidator as e, BackendIntegrityError as f, type ProducedProposal as g, type ProposalEventLike as h, type RequirementCheck as i, type ToolCallEventLike as j, type ValidationIssue as k, type ValidationResult as l, agentProfileHash as m, assertRealBackend as n, byteLengthRange as o, composeValidators as p, containsAll as q, createLlmCorrectnessChecker as r, extractProducedState as s, jsonHasKeys as t, parseCorrectnessResponse as u, regexMatch as v, summarizeBackendIntegrity as w, verifyCompletion as x };
376
+ export { type AgentProfile as A, type BackendIntegrityReport as B, type CompletionRequirement as C, type LlmCorrectnessCheckerOpts as L, type ProducedState as P, type RuntimeEventLike as R, type SatisfiedBy as S, type TaskGold as T, type ValidationContext as V, type CompletionVerdict as a, type CorrectnessChecker as b, type Artifact as c, type ArtifactEventLike as d, type ArtifactValidator as e, BackendIntegrityError as f, type ProducedProposal as g, type ProposalEventLike as h, type RequirementCheck as i, type ToolCallEventLike as j, type ValidationIssue as k, type ValidationResult as l, agentProfileHash as m, assertRealBackend as n, byteLengthRange as o, composeValidators as p, containsAll as q, createLlmCorrectnessChecker as r, createTokenRecallChecker as s, extractProducedState as t, jsonHasKeys as u, parseCorrectnessResponse as v, regexMatch as w, summarizeBackendIntegrity as x, verifyCompletion as y };
@@ -1,21 +1,21 @@
1
1
  import { AxAIService, AxFunction } from '@ax-llm/ax';
2
2
  import { M as MultiLayerVerifier, V as VerifyOptions, S as Severity } from '../multi-layer-verifier-DlWCXuxL.js';
3
3
  import { c as RunCritic, a as RunTrace } from '../run-critic-BAIjX99r.js';
4
- import { S as SemanticConceptJudgeOptions, a as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-Du4ZVyef.js';
5
- export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, b as DefaultAnalystRegistryOptions, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FindingSubject, g as FindingSubjectKind, h as FindingSubjectStringSchema, i as FindingsDiff, j as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, k as KNOWLEDGE_GAP_KIND_SPEC, l as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, m as SKILL_USAGE_ANALYST, n as SkillUsageAnalyst, o as SkillUsageRecord, p as SkillUsageReport, q as SkillUsageScanConfig, r as buildDefaultAnalystRegistry, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as parseFindingSubject, y as renderFindingSubject } from '../semantic-concept-judge-Du4ZVyef.js';
6
- import { A as AnalyzeTracesOptions } from '../analyst-t7zZS3TV.js';
7
- import { T as TraceAnalysisStore } from '../store-GmBE2pZZ.js';
4
+ import { S as SemanticConceptJudgeOptions, a as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-DIEgr_6v.js';
5
+ export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, b as DefaultAnalystRegistryOptions, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FindingSubject, g as FindingSubjectKind, h as FindingSubjectStringSchema, i as FindingsDiff, j as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, k as KNOWLEDGE_GAP_KIND_SPEC, l as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, m as SKILL_USAGE_ANALYST, n as SkillUsageAnalyst, o as SkillUsageRecord, p as SkillUsageReport, q as SkillUsageScanConfig, r as buildDefaultAnalystRegistry, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as parseFindingSubject, y as renderFindingSubject } from '../semantic-concept-judge-DIEgr_6v.js';
6
+ import { A as AnalyzeTracesOptions } from '../analyst-C8HHvfJp.js';
7
+ import { T as TraceAnalysisStore } from '../store-C1YxJDEK.js';
8
8
  import { b as JudgeFn, a as JudgeInput } from '../types-Croy5h7V.js';
9
- import { A as Analyst, h as AnalystSeverity, c as AnalystFinding } from '../types-DRvV0zRo.js';
10
- export { a as AnalystContext, g as AnalystCost, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, k as ChatCallOpts, C as ChatClient, l as ChatRequest, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, p as CreateChatClientOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from '../types-DRvV0zRo.js';
9
+ import { A as Analyst, h as AnalystSeverity, c as AnalystFinding } from '../types-Cu3u_x59.js';
10
+ export { a as AnalystContext, g as AnalystCost, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, k as ChatCallOpts, C as ChatClient, l as ChatRequest, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, p as CreateChatClientOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from '../types-Cu3u_x59.js';
11
11
  import { TCloud } from '@tangle-network/tcloud';
12
- export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, T as TraceAnalystGolden, c as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-DqV2t1Xk.js';
13
- export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, R as RegistryRunOpts } from '../registry-BmEuU94S.js';
14
- import { L as LlmClientOptions } from '../llm-client-DbjLfz-K.js';
12
+ export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, T as TraceAnalystGolden, c as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-CVecZZG_.js';
13
+ export { A as AnalystHooks, a as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, R as RegistryRunOpts } from '../registry-DrEQ3Luj.js';
14
+ import { L as LlmClientOptions } from '../llm-client-CuUg2Mn3.js';
15
15
  import '../schema-m0gsnbt3.js';
16
16
  import '../store-CKUAgsJz.js';
17
17
  import 'zod';
18
- import '../run-record-sItO5ftF.js';
18
+ import '../run-record-De9VarXR.js';
19
19
  import '../errors-Dwqw-T_m.js';
20
20
  import '../raw-provider-sink-C46HDghv.js';
21
21
 
@@ -14,7 +14,7 @@ import {
14
14
  diffFindings,
15
15
  emitSkillUsageFindings,
16
16
  runSemanticConceptJudge
17
- } from "../chunk-5LVWPNS5.js";
17
+ } from "../chunk-L5G7OUKD.js";
18
18
  import {
19
19
  ANALYST_SEVERITIES,
20
20
  AnalystRegistry,
@@ -41,8 +41,8 @@ import {
41
41
  renderPriorFindings,
42
42
  stripCodeFences,
43
43
  structureFindings
44
- } from "../chunk-CF67I6QY.js";
45
- import "../chunk-IHDHUN2X.js";
44
+ } from "../chunk-VIDQF3F5.js";
45
+ import "../chunk-CVVHBFGN.js";
46
46
  import {
47
47
  analyzeTraces
48
48
  } from "../chunk-VUINJM5M.js";
@@ -1,5 +1,5 @@
1
1
  import { AxAIService } from '@ax-llm/ax';
2
- import { T as TraceAnalysisStore } from './store-GmBE2pZZ.js';
2
+ import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
3
3
 
4
4
  interface AnalyzeTracesInput {
5
5
  /** The user-facing question. Domain framing belongs here, not in the