@tangle-network/agent-eval 0.79.0 → 0.81.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +101 -169
- package/dist/adapters/http.d.ts +2 -2
- package/dist/adapters/langchain.d.ts +2 -2
- package/dist/adapters/otel.d.ts +4 -4
- package/dist/{agent-profile-aSEaJ9Pl.d.ts → agent-profile-D0PBIWlV.d.ts} +14 -2
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +3 -3
- package/dist/{analyst-t7zZS3TV.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
- package/dist/belief-state/index.d.ts +524 -0
- package/dist/belief-state/index.js +1862 -0
- package/dist/belief-state/index.js.map +1 -0
- package/dist/benchmarks/index.d.ts +2 -2
- package/dist/calibration-Cpr3WaX3.d.ts +101 -0
- package/dist/campaign/index.d.ts +40 -120
- package/dist/campaign/index.js +129 -238
- package/dist/campaign/index.js.map +1 -1
- package/dist/chunk-4DIJWVUT.js +131 -0
- package/dist/chunk-4DIJWVUT.js.map +1 -0
- package/dist/{chunk-RPLZ4OIB.js → chunk-BABOZOSN.js} +7 -4
- package/dist/{chunk-RPLZ4OIB.js.map → chunk-BABOZOSN.js.map} +1 -1
- package/dist/{chunk-IHDHUN2X.js → chunk-CVVHBFGN.js} +3 -3
- package/dist/chunk-CVVHBFGN.js.map +1 -0
- package/dist/{chunk-B26KI423.js → chunk-FZWAFVAA.js} +2 -2
- package/dist/{chunk-ITBRCT73.js → chunk-IDVBLYCY.js} +2 -10
- package/dist/chunk-IDVBLYCY.js.map +1 -0
- package/dist/{chunk-5LVWPNS5.js → chunk-L5G7OUKD.js} +4 -4
- package/dist/{chunk-5LVWPNS5.js.map → chunk-L5G7OUKD.js.map} +1 -1
- package/dist/chunk-NPCTHQIO.js +91 -0
- package/dist/chunk-NPCTHQIO.js.map +1 -0
- package/dist/{chunk-GXHLRXDI.js → chunk-OTYQPHPL.js} +4 -4
- package/dist/{chunk-6REHLN5J.js → chunk-QS3RBQPI.js} +2 -2
- package/dist/{chunk-GWGO2K6Y.js → chunk-RBNA5AZT.js} +2 -2
- package/dist/chunk-S42AWHMP.js +697 -0
- package/dist/chunk-S42AWHMP.js.map +1 -0
- package/dist/chunk-VI2UW6B6.js +162 -0
- package/dist/chunk-VI2UW6B6.js.map +1 -0
- package/dist/{chunk-CF67I6QY.js → chunk-VIDQF3F5.js} +2 -2
- package/dist/{chunk-XXNIODOM.js → chunk-WJL2NJXN.js} +3 -3
- package/dist/{chunk-LB2UOI5F.js → chunk-YGYXHNAQ.js} +47 -160
- package/dist/chunk-YGYXHNAQ.js.map +1 -0
- package/dist/{chunk-KX6F6NCG.js → chunk-Z7VFTS2J.js} +2 -2
- package/dist/{chunk-ZPSKPT3V.js → chunk-ZZ2HOPME.js} +5 -2
- package/dist/chunk-ZZ2HOPME.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/code-agent-session-BRXmavYv.d.ts +80 -0
- package/dist/contract/index.d.ts +132 -18
- package/dist/contract/index.js +139 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-CehLtoET.d.ts → control-GeE8OhpN.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/governance/index.d.ts +1 -1
- package/dist/hosted/index.d.ts +4 -4
- package/dist/{index-B1RKber3.d.ts → index-DE3RXAXD.d.ts} +1 -1
- package/dist/index.d.ts +79 -288
- package/dist/index.js +87 -410
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-dlpEzQDi.d.ts → insight-report-3ADTfClO.d.ts} +1 -1
- package/dist/{kind-factory-DqV2t1Xk.d.ts → kind-factory-CVecZZG_.d.ts} +2 -2
- package/dist/{llm-client-DbjLfz-K.d.ts → llm-client-CuUg2Mn3.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +6 -99
- package/dist/meta-eval/index.js +7 -76
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/off-policy-DiwuKKg7.d.ts +132 -0
- package/dist/openapi.json +1 -1
- package/dist/{outcome-store-D6KWmYvj.d.ts → outcome-store-rnXLEqSn.d.ts} +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/{provenance-CEAJI9rm.d.ts → provenance-B9Q4886D.d.ts} +4 -4
- package/dist/{registry-BmEuU94S.d.ts → registry-DrEQ3Luj.d.ts} +2 -2
- package/dist/{release-report-CXXZlR8g.d.ts → release-report-hlNtD12q.d.ts} +2 -2
- package/dist/reporting.d.ts +6 -6
- package/dist/reporting.js +3 -3
- package/dist/{researcher-rInLj9De.d.ts → researcher-BLPHBbNV.d.ts} +3 -3
- package/dist/rl.d.ts +11 -141
- package/dist/rl.js +10 -124
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-CWyWWLBg.d.ts → rubric-predictive-validity-CnEl9Jc8.d.ts} +2 -2
- package/dist/{run-campaign-OVEZF24D.js → run-campaign-4Y5V5CN3.js} +3 -3
- package/dist/{run-improvement-loop-Bgu4C59E.d.ts → run-improvement-loop-D6PZOoQL.d.ts} +2 -2
- package/dist/{run-record-sItO5ftF.d.ts → run-record-De9VarXR.d.ts} +1 -1
- package/dist/{semantic-concept-judge-Du4ZVyef.d.ts → semantic-concept-judge-DIEgr_6v.d.ts} +6 -6
- package/dist/{statistics-B7yCbi9i.d.ts → statistics-CnC1FMbx.d.ts} +3 -7
- package/dist/{store-GmBE2pZZ.d.ts → store-C1YxJDEK.d.ts} +1 -1
- package/dist/{summary-report-BTaXq1TS.d.ts → summary-report-Db0dDSWP.d.ts} +1 -1
- package/dist/traces.d.ts +5 -5
- package/dist/{types-DRvV0zRo.d.ts → types-Cu3u_x59.d.ts} +3 -3
- package/dist/{types-QHG0KnkF.d.ts → types-D7lLRYe9.d.ts} +2 -2
- package/dist/wire/index.js +2 -2
- package/dist/workflow/index.d.ts +7 -7
- package/dist/workflow/index.js +1 -1
- package/docs/concepts.md +1 -0
- package/docs/research/belief-state-agent-eval-roadmap.md +590 -0
- package/docs/research/research-roadmap.md +1 -0
- package/docs/self-improvement-map.md +111 -0
- package/package.json +7 -2
- package/dist/chunk-IHDHUN2X.js.map +0 -1
- package/dist/chunk-ITBRCT73.js.map +0 -1
- package/dist/chunk-LB2UOI5F.js.map +0 -1
- package/dist/chunk-ZPSKPT3V.js.map +0 -1
- /package/dist/{chunk-B26KI423.js.map → chunk-FZWAFVAA.js.map} +0 -0
- /package/dist/{chunk-GXHLRXDI.js.map → chunk-OTYQPHPL.js.map} +0 -0
- /package/dist/{chunk-6REHLN5J.js.map → chunk-QS3RBQPI.js.map} +0 -0
- /package/dist/{chunk-GWGO2K6Y.js.map → chunk-RBNA5AZT.js.map} +0 -0
- /package/dist/{chunk-CF67I6QY.js.map → chunk-VIDQF3F5.js.map} +0 -0
- /package/dist/{chunk-XXNIODOM.js.map → chunk-WJL2NJXN.js.map} +0 -0
- /package/dist/{chunk-KX6F6NCG.js.map → chunk-Z7VFTS2J.js.map} +0 -0
- /package/dist/{run-campaign-OVEZF24D.js.map → run-campaign-4Y5V5CN3.js.map} +0 -0
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,20 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.81.0] — 2026-06-05 — eval-campaign scaffold prep primitives
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- **`aggregateJudgeVerdicts<D>` (root).** Generic judge-ensemble reducer: fan out N uncorrelated judges, mean each rubric dimension over the SURVIVORS, report the inter-rater disagreement spread, sum cost. Replaces the same reduction hand-rolled in legal (`aggregateEnsemble`), creative (`production-loop/judges.ts`), and tax (`judge-ensemble.ts`). Fail-loud: a failed judge (`perDimension: null`) is recorded in `failedJudges`, never folded into a zero; all-failed throws; a failed judge's cost is still summed. Composite reuses `weightedComposite`.
|
|
12
|
+
- **`createTokenRecallChecker` (root).** The deterministic, no-LLM `CorrectnessChecker` — sibling of `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its content is substantive and recalls ≥ `minRecall` of the requirement title's significant tokens. The default completion gate for apps/tests without an LLM judge.
|
|
13
|
+
- **`ErrorCluster` (root + `/analyst`).** The failure-cluster element type is now a named export, so consumers import it instead of deriving `DatasetOverview['error_clusters'][number]`.
|
|
14
|
+
|
|
15
|
+
### Fixed
|
|
16
|
+
|
|
17
|
+
- **Lint drift + non-executable pre-commit hook.** `.husky/pre-commit` was tracked `100644`, so the hook silently no-op'd and unformatted code reached `main`; marked executable and reformatted the drift.
|
|
18
|
+
|
|
19
|
+
---
|
|
20
|
+
|
|
7
21
|
## [0.72.3] — 2026-06-01 — workflow trace hardening and driver backtests
|
|
8
22
|
|
|
9
23
|
### Added
|
package/README.md
CHANGED
|
@@ -1,197 +1,106 @@
|
|
|
1
1
|
# `@tangle-network/agent-eval`
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Evaluate and improve AI agents from the runs they already produce.
|
|
4
|
+
|
|
5
|
+
`agent-eval` turns agent outputs, traces, judge scores, and production feedback into a decision packet: did this change help, what failed, what should ship, and what needs more data?
|
|
4
6
|
|
|
5
7
|
[](https://www.npmjs.com/package/@tangle-network/agent-eval)
|
|
6
8
|
[](https://pypi.org/project/agent-eval-rpc/)
|
|
7
9
|
[](https://github.com/tangle-network/agent-eval/actions/workflows/ci.yml)
|
|
8
10
|
[](./LICENSE)
|
|
9
11
|
|
|
10
|
-
|
|
12
|
+
Use it when you need to:
|
|
11
13
|
|
|
12
|
-
|
|
14
|
+
- compare a candidate agent/prompt/model against a baseline,
|
|
15
|
+
- turn production traces or human feedback into eval results,
|
|
16
|
+
- run a gated self-improvement loop,
|
|
17
|
+
- explain failures by cluster, cost, judge disagreement, and release risk.
|
|
13
18
|
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
- [What you get back](#what-you-get-back-the-decision-packet)
|
|
17
|
-
- [Quick start](#quick-start)
|
|
18
|
-
- [Closed loop — `selfImprove()`](#closed-loop--selfimprove)
|
|
19
|
-
- [Observed runs — `analyzeRuns()`](#observed-runs--analyzeruns)
|
|
20
|
-
- [Existing data — intake adapters](#existing-data--intake-adapters)
|
|
21
|
-
- [How it compares](#how-it-compares)
|
|
22
|
-
- [Customer journeys](#customer-journeys)
|
|
23
|
-
- [Subpath entry points](#subpath-entry-points)
|
|
24
|
-
- [Concepts + design](#concepts--design)
|
|
25
|
-
- [Hosted tier](#hosted-tier)
|
|
26
|
-
- [Install + run](#install--run)
|
|
27
|
-
- [Stability + versioning](#stability--versioning)
|
|
28
|
-
- [License](#license)
|
|
19
|
+
It is a library, not a SaaS requirement. TypeScript is first-class; Python can call the same wire protocol through `agent-eval-rpc`.
|
|
29
20
|
|
|
30
21
|
---
|
|
31
22
|
|
|
32
|
-
##
|
|
23
|
+
## Install
|
|
33
24
|
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
```jsonc
|
|
37
|
-
{
|
|
38
|
-
"n": 80, // runs analyzed
|
|
39
|
-
"composite": { // distributional summary
|
|
40
|
-
"mean": 0.62, "p50": 0.65, "p95": 0.88, "stddev": 0.17,
|
|
41
|
-
"histogram": [/* 12 bins */]
|
|
42
|
-
},
|
|
43
|
-
"lift": { // paired bootstrap
|
|
44
|
-
"baselineMean": 0.58, "candidateMean": 0.65,
|
|
45
|
-
"delta": 0.07,
|
|
46
|
-
"ci95": [0.04, 0.10], // 95% CI on the delta
|
|
47
|
-
"pValue": 0.0008, // paired-t
|
|
48
|
-
"cohensD": 0.41,
|
|
49
|
-
"n": 40,
|
|
50
|
-
"mde": 0.06, // min detectable effect at 80% power
|
|
51
|
-
"requiredN": 38 // n needed to detect observed delta
|
|
52
|
-
},
|
|
53
|
-
"judges": { // per-judge calibration
|
|
54
|
-
"domain-expert": { "n": 80, "meanScore": 0.64 },
|
|
55
|
-
"helpfulness-llm": { "n": 80, "meanScore": 0.61 }
|
|
56
|
-
},
|
|
57
|
-
"interRater": { // multi-rater agreement
|
|
58
|
-
"raters": 3, "jointlyRated": 80, "kappa": 0.71,
|
|
59
|
-
"disagreementCases": [/* top 20 ranked by spread */]
|
|
60
|
-
},
|
|
61
|
-
"costQuality": { // cost-vs-quality
|
|
62
|
-
"cost": { "mean": 0.024, "p95": 0.041, /* ... */ },
|
|
63
|
-
"pareto": { /* ParetoFigureSpec the dashboard renders */ }
|
|
64
|
-
},
|
|
65
|
-
"failureClusters": { // when an AnalystRegistry is wired
|
|
66
|
-
"totalFailures": 11,
|
|
67
|
-
"clusters": [
|
|
68
|
-
{ "name": "off-topic-drift", "share": 0.45, "exemplars": ["run-12", "run-19"] },
|
|
69
|
-
{ "name": "over-confidence", "share": 0.27, "exemplars": ["run-3"] },
|
|
70
|
-
{ "name": "format-mismatch", "share": 0.18, "exemplars": ["run-41"] }
|
|
71
|
-
]
|
|
72
|
-
},
|
|
73
|
-
"contamination": { "leaks": 0, "holdoutAuditPassed": true },
|
|
74
|
-
"outcomeCorrelation": { // when downstream metric supplied
|
|
75
|
-
"metric": "engagement_rate", "n": 80,
|
|
76
|
-
"pearson": 0.72, "spearman": 0.69,
|
|
77
|
-
"rewardModel": { "intercept": 0.04, "slope": 1.93, "r2": 0.52 }
|
|
78
|
-
},
|
|
79
|
-
"release": {
|
|
80
|
-
"status": "pass",
|
|
81
|
-
"axes": [
|
|
82
|
-
{ "name": "quality-lift", "status": "pass" },
|
|
83
|
-
{ "name": "contamination", "status": "pass" },
|
|
84
|
-
{ "name": "composite-distribution","status": "pass" }
|
|
85
|
-
]
|
|
86
|
-
},
|
|
87
|
-
"recommendations": [
|
|
88
|
-
{ "priority": "critical", "kind": "ship",
|
|
89
|
-
"title": "Ship — lift 0.070 (95% CI 0.040..0.100)",
|
|
90
|
-
"detail": "Holdout lift exceeds threshold 0.02 with 95% bootstrap confidence (n=40, p=0.0008, d=0.41)." },
|
|
91
|
-
{ "priority": "high", "kind": "investigate",
|
|
92
|
-
"title": "Top failure cluster: off-topic-drift (45% of failures)",
|
|
93
|
-
"detail": "11 runs failed. Drill into exemplars run-12 / run-19 to identify the pattern." }
|
|
94
|
-
]
|
|
95
|
-
}
|
|
25
|
+
```sh
|
|
26
|
+
pnpm add @tangle-network/agent-eval
|
|
96
27
|
```
|
|
97
28
|
|
|
98
|
-
|
|
29
|
+
Python clients can use the RPC package:
|
|
30
|
+
|
|
31
|
+
```sh
|
|
32
|
+
pip install agent-eval-rpc
|
|
33
|
+
```
|
|
99
34
|
|
|
100
35
|
---
|
|
101
36
|
|
|
102
37
|
## Quick start
|
|
103
38
|
|
|
104
|
-
###
|
|
39
|
+
### 1. Analyze runs you already have
|
|
105
40
|
|
|
106
|
-
|
|
41
|
+
Start here if you already have production logs, benchmark rows, human ratings, or agent run records.
|
|
107
42
|
|
|
108
43
|
```ts
|
|
109
|
-
import {
|
|
44
|
+
import { analyzeRuns } from '@tangle-network/agent-eval/contract'
|
|
110
45
|
|
|
111
|
-
const
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
await myAgent.run(scenario),
|
|
115
|
-
judges: [myJudge], // any JudgeConfig — LLM, rule, ensemble
|
|
116
|
-
baselineSurface: { systemPrompt: currentPrompt },
|
|
46
|
+
const report = await analyzeRuns({
|
|
47
|
+
runs, // RunRecord[]
|
|
48
|
+
baselineRuns,
|
|
117
49
|
})
|
|
118
50
|
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
51
|
+
console.log(report.recommendations)
|
|
52
|
+
console.log(report.lift)
|
|
53
|
+
console.log(report.failureClusters)
|
|
122
54
|
```
|
|
123
55
|
|
|
124
|
-
|
|
56
|
+
The output includes score distributions, lift confidence intervals, failure modes, cost-quality tradeoffs, judge agreement, contamination checks, and release recommendations when the input supports them.
|
|
57
|
+
|
|
58
|
+
### 2. Run a gated improvement loop
|
|
125
59
|
|
|
126
|
-
|
|
60
|
+
Use this when you have scenarios, a runnable agent, and judges.
|
|
127
61
|
|
|
128
62
|
```ts
|
|
129
|
-
import {
|
|
63
|
+
import { selfImprove } from '@tangle-network/agent-eval/contract'
|
|
130
64
|
|
|
131
|
-
const
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
},
|
|
137
|
-
canaryScenarios, // optional — contamination probe
|
|
138
|
-
analyst: myAnalystRegistry, // optional — AI-powered failure clustering
|
|
65
|
+
const result = await selfImprove({
|
|
66
|
+
scenarios,
|
|
67
|
+
dispatch: async ({ scenario }) => myAgent.run(scenario),
|
|
68
|
+
judges: [myJudge],
|
|
69
|
+
baselineSurface: { systemPrompt: currentPrompt },
|
|
139
70
|
})
|
|
140
71
|
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
72
|
+
console.log(result.gateDecision)
|
|
73
|
+
console.log(result.winnerSurface)
|
|
74
|
+
console.log(result.insight.recommendations)
|
|
144
75
|
```
|
|
145
76
|
|
|
146
|
-
|
|
77
|
+
`selfImprove()` evaluates candidates on held-out scenarios before recommending a winner.
|
|
147
78
|
|
|
148
|
-
|
|
79
|
+
### 3. Adapt existing data
|
|
149
80
|
|
|
150
81
|
```ts
|
|
151
|
-
import {
|
|
152
|
-
fromFeedbackTable,
|
|
153
|
-
fromOtelSpans,
|
|
154
|
-
analyzeRuns,
|
|
155
|
-
} from '@tangle-network/agent-eval/contract'
|
|
82
|
+
import { analyzeRuns, fromFeedbackTable, fromOtelSpans } from '@tangle-network/agent-eval/contract'
|
|
156
83
|
|
|
157
|
-
// Multi-rater approve/reject (Obsidian tags, Sheets, CSV, Postgres).
|
|
158
84
|
const { runs, raterScores } = fromFeedbackTable({
|
|
159
|
-
ratings: parseYourFeedbackTable(),
|
|
85
|
+
ratings: parseYourFeedbackTable(),
|
|
160
86
|
})
|
|
161
|
-
await analyzeRuns({ runs, raterScores })
|
|
162
87
|
|
|
163
|
-
|
|
164
|
-
const runs2 = fromOtelSpans({ spans: yourOtelStream })
|
|
165
|
-
await analyzeRuns({ runs: runs2 })
|
|
166
|
-
```
|
|
88
|
+
const traceRuns = fromOtelSpans({ spans: yourOtelSpans })
|
|
167
89
|
|
|
168
|
-
|
|
90
|
+
await analyzeRuns({ runs: [...runs, ...traceRuns], raterScores })
|
|
91
|
+
```
|
|
169
92
|
|
|
170
93
|
---
|
|
171
94
|
|
|
172
|
-
##
|
|
173
|
-
|
|
174
|
-
| | LangSmith | Braintrust | Phoenix | **agent-eval** |
|
|
175
|
-
|---|:---:|:---:|:---:|:---:|
|
|
176
|
-
| Closed-loop self-improvement | ✱ human-in-loop | ✱ experiment-driven | — | ✓ autonomous + gated |
|
|
177
|
-
| Statistical lift CI (paired bootstrap) | — | partial | — | ✓ |
|
|
178
|
-
| Judge calibration + bias detection | — | — | — | ✓ |
|
|
179
|
-
| Inter-rater agreement + disagreement triage | — | — | — | ✓ |
|
|
180
|
-
| Contamination / canary check | — | — | — | ✓ |
|
|
181
|
-
| AI-driven failure clustering | partial | — | partial | ✓ |
|
|
182
|
-
| Cost-quality Pareto | — | — | — | ✓ |
|
|
183
|
-
| Multi-language clients (TS + Python) | TS only | TS only | TS + Py | ✓ TS + Py |
|
|
184
|
-
| Self-hostable / no-SaaS option | — | — | OSS | ✓ MIT, OSS |
|
|
185
|
-
| Substrate vs SaaS shape | SaaS | SaaS | OSS server | **library** |
|
|
186
|
-
| Hosted tier (optional) | required | required | optional | optional |
|
|
95
|
+
## Core concepts
|
|
187
96
|
|
|
188
|
-
|
|
97
|
+
- **RunRecord**: the durable row for one agent run: model, prompt/config hashes, split, cost, tokens, outcome.
|
|
98
|
+
- **Scenario**: one task or case the agent attempts.
|
|
99
|
+
- **Judge**: a scoring function, rule-based or model-based.
|
|
100
|
+
- **InsightReport**: the decision packet returned by `analyzeRuns()` and embedded in `selfImprove()`.
|
|
101
|
+
- **Gate**: the policy that decides `ship`, `hold`, or `need_more_data`.
|
|
189
102
|
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
## Customer journeys
|
|
193
|
-
|
|
194
|
-
Three runnable examples — each is self-contained, each shows the actual output.
|
|
103
|
+
## Examples
|
|
195
104
|
|
|
196
105
|
| Journey | Example | Who it's for |
|
|
197
106
|
|---|---|---|
|
|
@@ -207,28 +116,55 @@ Each example: `README.md` + a single `index.ts` runnable via `pnpm tsx`. Prints
|
|
|
207
116
|
|
|
208
117
|
| Subpath | What it gives you |
|
|
209
118
|
|---|---|
|
|
210
|
-
|
|
|
211
|
-
|
|
|
212
|
-
|
|
|
213
|
-
|
|
|
214
|
-
|
|
|
215
|
-
|
|
|
216
|
-
|
|
|
217
|
-
|
|
|
218
|
-
|
|
|
219
|
-
|
|
|
220
|
-
|
|
|
221
|
-
|
|
|
222
|
-
|
|
|
223
|
-
|
|
|
224
|
-
|
|
225
|
-
|
|
119
|
+
| `…/contract` | **The headline, frozen surface — new code starts here.** `selfImprove`, `analyzeRuns`, `runEval`, `runCampaign`, `runImprovementLoop`, `diffRuns`; intake adapters (`fromFeedbackTable`, `fromOtelSpans`); drivers (`gepaDriver`, `evolutionaryDriver`); gates (`defaultProductionGate`, `heldOutGate`, `paretoSignificanceGate`, `composeGate`); the deployment-outcome store; storage; and the five core types `Scenario` / `Dispatch` / `JudgeConfig` / `Mutator` / `Gate`. |
|
|
120
|
+
| `…/hosted` | `createHostedClient` / `hostedClientFromEnv` + the wire types to ship eval-run events + trace spans to a hosted orchestrator (ours or your own implementation of the spec) |
|
|
121
|
+
| `…/adapters/otel` | `createOtelBridge` — forwards OpenTelemetry-shape spans into the hosted-tier ingest, no `@opentelemetry/*` dependency |
|
|
122
|
+
| `…/adapters/langchain` | Wrap any LangChain `Runnable` as a `Dispatch` (or `JudgeConfig`), no `@langchain/core` peer dep |
|
|
123
|
+
| `…/adapters/http` | `httpDispatch` + `runDispatchServer` — run a campaign's worker on another machine (multi-region, driver-as-a-service) |
|
|
124
|
+
| `…/campaign` | **The measurement + improvement engine** (`@experimental`): `runProfileMatrix`, `compareDrivers`, every driver (`gepaDriver`, `haloDriver`, `skillOptDriver`, `aceDriver`, `memoryCurationDriver`, …), the gates, storage backends, and loop provenance. `/contract` re-exports the stable subset. |
|
|
125
|
+
| `…/rl` | RL bridge from eval artifacts to training signal: verifiable rewards, preferences, OPE, PRM, tournaments, contamination, compute curves, plus the durable corpus + `buildRlDataset` / datasheet bundle |
|
|
126
|
+
| `…/reporting` | Release-decision statistics: `pairedBootstrap`, `benjaminiHochberg`, anytime-valid sequential e-values, `evaluateReleaseConfidence`, and the report renderers |
|
|
127
|
+
| `…/analyst` | The trace-analyst surface: `AnalystRegistry` + `buildDefaultAnalystRegistry` (run the failure-clustering panel), `FindingsStore`, and the LLM chat transports |
|
|
128
|
+
| `…/traces` | Trace stores + emitters, OTLP-JSONL deterministic replay, `analyzeTraces`, and the `traceAnalystOnRunComplete` hook |
|
|
129
|
+
| `…/control` | Agent control loop: `runAgentControlLoop` (observe → validate → decide → act), action policy, propose/review |
|
|
130
|
+
| `…/matrix` | `runAgentMatrix` — an N-axis cartesian over caller-supplied substrate values, per-axis pass/score/cost/duration |
|
|
131
|
+
| `…/multishot` | N-shot persona × shot matrix runner (`runMultishot` / `runMultishotMatrix`) |
|
|
132
|
+
| `…/wire` | The cross-language HTTP/RPC server + Zod schemas (the source-of-truth protocol the Python client speaks) + the built-in rubric registry |
|
|
133
|
+
| `…/benchmarks` | `BenchmarkAdapter` contract + `deterministicSplit` + the bundled `routing` reference benchmark |
|
|
134
|
+
|
|
135
|
+
**Specialized surfaces** (subpath-only): `…/prm` (process-reward grading + best-of-N), `…/meta-eval` (judge calibration + the deployment-outcome store), `…/pipelines` (trace-diagnostic views: budget breach, failure cluster, stuck loop, …), `…/governance` (EU AI Act / NIST AI RMF / SOC2 reports), `…/knowledge` (knowledge-readiness gating before a run), `…/builder-eval` (code-generator three-layer eval), `…/storyboard` (trace → watchable replay), `…/authenticity` (anti-Goodhart "real or convincing BS" scorer over produced files), `…/workflow` (workflow-trace eval + partner export), `…/telemetry` (Workers-safe telemetry client).
|
|
136
|
+
|
|
137
|
+
The root export remains available for backward compatibility; new code should prefer the focused subpaths above — `/contract` first.
|
|
138
|
+
|
|
139
|
+
---
|
|
140
|
+
|
|
141
|
+
## Composition with the stack
|
|
142
|
+
|
|
143
|
+
agent-eval is the bottom of the layering: consumers depend on it, it depends on none of them.
|
|
144
|
+
|
|
145
|
+
```
|
|
146
|
+
agent-runtime Runs agents (chat turns, one-shot tasks, multi-attempt loops), captures every
|
|
147
|
+
run as a trace, and calls optimizePrompt / runImprovementLoop. Produces the
|
|
148
|
+
RunRecords + traces agent-eval scores. Depends on agent-eval.
|
|
149
|
+
|
|
150
|
+
agent-eval selfImprove, analyzeRuns, runCampaign + drivers (gepaDriver, …), the gates
|
|
151
|
+
(this repo) (heldOutGate, defaultProductionGate, paretoSignificanceGate), the InsightReport
|
|
152
|
+
decision packet, the RL bridge, the wire protocol. Depends on neither consumer.
|
|
153
|
+
|
|
154
|
+
agent-knowledge proposeKnowledgeWrites / applyKnowledgeWriteBlocks. agent-eval's analyst findings
|
|
155
|
+
feed it; the knowledge gate consumes them. Depends on agent-eval.
|
|
156
|
+
|
|
157
|
+
sandbox AgentProfile, Sandbox.create, streamPrompt. The execution surface the runtime's
|
|
158
|
+
loops run on; agent-eval scores what comes back.
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
The rule: **agent-eval has zero upward dependencies on a consumer.** A concept that makes sense *without* a running agent loop — a verdict, a run record, a scenario, a judge score — is substrate and lives here; a runtime-shaped one (a sandbox profile, a validation context with an abort signal) lives in agent-runtime. When in doubt, lean substrate.
|
|
226
162
|
|
|
227
163
|
---
|
|
228
164
|
|
|
229
165
|
## Concepts + design
|
|
230
166
|
|
|
231
|
-
- [`docs/concepts.md`](./docs/concepts.md) —
|
|
167
|
+
- [`docs/concepts.md`](./docs/concepts.md) — the three top-level functions, the layering rule, and the wire-protocol contract (the five core contract types are documented in the `/contract` barrel itself)
|
|
232
168
|
- [`docs/insight-report.md`](./docs/insight-report.md) — annotated walkthrough of every section of the decision packet
|
|
233
169
|
- [`docs/customer-journeys.md`](./docs/customer-journeys.md) — three end-to-end journeys with code + expected output
|
|
234
170
|
- [`docs/adapters-observability.md`](./docs/adapters-observability.md) — composing agent-eval with LangSmith, Langfuse, Phoenix, OpenLLMetry, TraceAI
|
|
@@ -259,13 +195,7 @@ The substrate runs the loop in your process. Only the eval-run events + (optiona
|
|
|
259
195
|
|
|
260
196
|
---
|
|
261
197
|
|
|
262
|
-
##
|
|
263
|
-
|
|
264
|
-
```sh
|
|
265
|
-
pnpm add @tangle-network/agent-eval
|
|
266
|
-
# or, from Python:
|
|
267
|
-
pip install agent-eval-rpc
|
|
268
|
-
```
|
|
198
|
+
## Development
|
|
269
199
|
|
|
270
200
|
Run an example:
|
|
271
201
|
|
|
@@ -287,7 +217,9 @@ pnpm test
|
|
|
287
217
|
|
|
288
218
|
## Stability + versioning
|
|
289
219
|
|
|
290
|
-
|
|
220
|
+
The `/contract` surface is the **stability contract**: its barrel freezes the API — a `0.x` minor only *adds*; nothing there changes shape or disappears. Depend on `/contract` (and the documented subpaths) rather than the root barrel.
|
|
221
|
+
|
|
222
|
+
In the deeper subpaths, `@stable` / `@experimental` JSDoc markers (visible in IDE hover + `.d.ts`) call out what may still move — most granularly in `/rl` (tagged per export) and `/campaign` (whole barrel `@experimental`, since `/contract` re-exports only its settled subset).
|
|
291
223
|
|
|
292
224
|
| Tag | Meaning |
|
|
293
225
|
|---|---|
|
package/dist/adapters/http.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-
|
|
2
|
-
import '../run-record-
|
|
1
|
+
import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-D7lLRYe9.js';
|
|
2
|
+
import '../run-record-De9VarXR.js';
|
|
3
3
|
import '../errors-Dwqw-T_m.js';
|
|
4
4
|
import '../schema-m0gsnbt3.js';
|
|
5
5
|
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-
|
|
2
|
-
import '../run-record-
|
|
1
|
+
import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-D7lLRYe9.js';
|
|
2
|
+
import '../run-record-De9VarXR.js';
|
|
3
3
|
import '../errors-Dwqw-T_m.js';
|
|
4
4
|
import '../schema-m0gsnbt3.js';
|
|
5
5
|
|
package/dist/adapters/otel.d.ts
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
import { TraceSpanEvent, HostedClient } from '../hosted/index.js';
|
|
2
|
-
import '../types-
|
|
3
|
-
import '../run-record-
|
|
2
|
+
import '../types-D7lLRYe9.js';
|
|
3
|
+
import '../run-record-De9VarXR.js';
|
|
4
4
|
import '../errors-Dwqw-T_m.js';
|
|
5
5
|
import '../schema-m0gsnbt3.js';
|
|
6
|
-
import '../insight-report-
|
|
7
|
-
import '../summary-report-
|
|
6
|
+
import '../insight-report-3ADTfClO.js';
|
|
7
|
+
import '../summary-report-Db0dDSWP.js';
|
|
8
8
|
import '../failure-cluster-CL7IVgkJ.js';
|
|
9
9
|
import '../store-CKUAgsJz.js';
|
|
10
10
|
import '../judge-calibration-DilmB3Ml.js';
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { A as AgentEvalError } from './errors-Dwqw-T_m.js';
|
|
2
|
-
import { R as RunRecord } from './run-record-
|
|
2
|
+
import { R as RunRecord } from './run-record-De9VarXR.js';
|
|
3
3
|
import { TCloud } from '@tangle-network/tcloud';
|
|
4
4
|
|
|
5
5
|
/**
|
|
@@ -258,6 +258,18 @@ declare function parseCorrectnessResponse(raw: string): {
|
|
|
258
258
|
* fulfil a requirement — the artifact must BE the deliverable.
|
|
259
259
|
*/
|
|
260
260
|
declare function createLlmCorrectnessChecker(tc: TCloud, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
|
|
261
|
+
/**
|
|
262
|
+
* Deterministic `CorrectnessChecker` — the no-LLM counterpart to
|
|
263
|
+
* `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its
|
|
264
|
+
* content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`
|
|
265
|
+
* of the requirement title's significant tokens. No network — the default gate
|
|
266
|
+
* for apps and tests without an LLM judge. Pass to `verifyCompletion` as the
|
|
267
|
+
* checker.
|
|
268
|
+
*/
|
|
269
|
+
declare function createTokenRecallChecker(opts?: {
|
|
270
|
+
minRecall?: number;
|
|
271
|
+
minContentLength?: number;
|
|
272
|
+
}): CorrectnessChecker;
|
|
261
273
|
|
|
262
274
|
/**
|
|
263
275
|
* Produced-state extraction — normalize a run's runtime event stream into the
|
|
@@ -361,4 +373,4 @@ interface AgentProfile {
|
|
|
361
373
|
*/
|
|
362
374
|
declare function agentProfileHash(profile: AgentProfile): string;
|
|
363
375
|
|
|
364
|
-
export { type AgentProfile as A, type BackendIntegrityReport as B, type CompletionRequirement as C, type LlmCorrectnessCheckerOpts as L, type ProducedState as P, type RuntimeEventLike as R, type SatisfiedBy as S, type TaskGold as T, type ValidationContext as V, type CompletionVerdict as a, type CorrectnessChecker as b, type Artifact as c, type ArtifactEventLike as d, type ArtifactValidator as e, BackendIntegrityError as f, type ProducedProposal as g, type ProposalEventLike as h, type RequirementCheck as i, type ToolCallEventLike as j, type ValidationIssue as k, type ValidationResult as l, agentProfileHash as m, assertRealBackend as n, byteLengthRange as o, composeValidators as p, containsAll as q, createLlmCorrectnessChecker as r,
|
|
376
|
+
export { type AgentProfile as A, type BackendIntegrityReport as B, type CompletionRequirement as C, type LlmCorrectnessCheckerOpts as L, type ProducedState as P, type RuntimeEventLike as R, type SatisfiedBy as S, type TaskGold as T, type ValidationContext as V, type CompletionVerdict as a, type CorrectnessChecker as b, type Artifact as c, type ArtifactEventLike as d, type ArtifactValidator as e, BackendIntegrityError as f, type ProducedProposal as g, type ProposalEventLike as h, type RequirementCheck as i, type ToolCallEventLike as j, type ValidationIssue as k, type ValidationResult as l, agentProfileHash as m, assertRealBackend as n, byteLengthRange as o, composeValidators as p, containsAll as q, createLlmCorrectnessChecker as r, createTokenRecallChecker as s, extractProducedState as t, jsonHasKeys as u, parseCorrectnessResponse as v, regexMatch as w, summarizeBackendIntegrity as x, verifyCompletion as y };
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,21 +1,21 @@
|
|
|
1
1
|
import { AxAIService, AxFunction } from '@ax-llm/ax';
|
|
2
2
|
import { M as MultiLayerVerifier, V as VerifyOptions, S as Severity } from '../multi-layer-verifier-DlWCXuxL.js';
|
|
3
3
|
import { c as RunCritic, a as RunTrace } from '../run-critic-BAIjX99r.js';
|
|
4
|
-
import { S as SemanticConceptJudgeOptions, a as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-
|
|
5
|
-
export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, b as DefaultAnalystRegistryOptions, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FindingSubject, g as FindingSubjectKind, h as FindingSubjectStringSchema, i as FindingsDiff, j as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, k as KNOWLEDGE_GAP_KIND_SPEC, l as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, m as SKILL_USAGE_ANALYST, n as SkillUsageAnalyst, o as SkillUsageRecord, p as SkillUsageReport, q as SkillUsageScanConfig, r as buildDefaultAnalystRegistry, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as parseFindingSubject, y as renderFindingSubject } from '../semantic-concept-judge-
|
|
6
|
-
import { A as AnalyzeTracesOptions } from '../analyst-
|
|
7
|
-
import { T as TraceAnalysisStore } from '../store-
|
|
4
|
+
import { S as SemanticConceptJudgeOptions, a as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-DIEgr_6v.js';
|
|
5
|
+
export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, b as DefaultAnalystRegistryOptions, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FindingSubject, g as FindingSubjectKind, h as FindingSubjectStringSchema, i as FindingsDiff, j as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, k as KNOWLEDGE_GAP_KIND_SPEC, l as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, m as SKILL_USAGE_ANALYST, n as SkillUsageAnalyst, o as SkillUsageRecord, p as SkillUsageReport, q as SkillUsageScanConfig, r as buildDefaultAnalystRegistry, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as parseFindingSubject, y as renderFindingSubject } from '../semantic-concept-judge-DIEgr_6v.js';
|
|
6
|
+
import { A as AnalyzeTracesOptions } from '../analyst-C8HHvfJp.js';
|
|
7
|
+
import { T as TraceAnalysisStore } from '../store-C1YxJDEK.js';
|
|
8
8
|
import { b as JudgeFn, a as JudgeInput } from '../types-Croy5h7V.js';
|
|
9
|
-
import { A as Analyst, h as AnalystSeverity, c as AnalystFinding } from '../types-
|
|
10
|
-
export { a as AnalystContext, g as AnalystCost, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, k as ChatCallOpts, C as ChatClient, l as ChatRequest, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, p as CreateChatClientOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from '../types-
|
|
9
|
+
import { A as Analyst, h as AnalystSeverity, c as AnalystFinding } from '../types-Cu3u_x59.js';
|
|
10
|
+
export { a as AnalystContext, g as AnalystCost, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, k as ChatCallOpts, C as ChatClient, l as ChatRequest, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, p as CreateChatClientOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from '../types-Cu3u_x59.js';
|
|
11
11
|
import { TCloud } from '@tangle-network/tcloud';
|
|
12
|
-
export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, T as TraceAnalystGolden, c as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-
|
|
13
|
-
export {
|
|
14
|
-
import { L as LlmClientOptions } from '../llm-client-
|
|
12
|
+
export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, T as TraceAnalystGolden, c as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-CVecZZG_.js';
|
|
13
|
+
export { A as AnalystHooks, a as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, R as RegistryRunOpts } from '../registry-DrEQ3Luj.js';
|
|
14
|
+
import { L as LlmClientOptions } from '../llm-client-CuUg2Mn3.js';
|
|
15
15
|
import '../schema-m0gsnbt3.js';
|
|
16
16
|
import '../store-CKUAgsJz.js';
|
|
17
17
|
import 'zod';
|
|
18
|
-
import '../run-record-
|
|
18
|
+
import '../run-record-De9VarXR.js';
|
|
19
19
|
import '../errors-Dwqw-T_m.js';
|
|
20
20
|
import '../raw-provider-sink-C46HDghv.js';
|
|
21
21
|
|
package/dist/analyst/index.js
CHANGED
|
@@ -14,7 +14,7 @@ import {
|
|
|
14
14
|
diffFindings,
|
|
15
15
|
emitSkillUsageFindings,
|
|
16
16
|
runSemanticConceptJudge
|
|
17
|
-
} from "../chunk-
|
|
17
|
+
} from "../chunk-L5G7OUKD.js";
|
|
18
18
|
import {
|
|
19
19
|
ANALYST_SEVERITIES,
|
|
20
20
|
AnalystRegistry,
|
|
@@ -41,8 +41,8 @@ import {
|
|
|
41
41
|
renderPriorFindings,
|
|
42
42
|
stripCodeFences,
|
|
43
43
|
structureFindings
|
|
44
|
-
} from "../chunk-
|
|
45
|
-
import "../chunk-
|
|
44
|
+
} from "../chunk-VIDQF3F5.js";
|
|
45
|
+
import "../chunk-CVVHBFGN.js";
|
|
46
46
|
import {
|
|
47
47
|
analyzeTraces
|
|
48
48
|
} from "../chunk-VUINJM5M.js";
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { AxAIService } from '@ax-llm/ax';
|
|
2
|
-
import { T as TraceAnalysisStore } from './store-
|
|
2
|
+
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
3
|
|
|
4
4
|
interface AnalyzeTracesInput {
|
|
5
5
|
/** The user-facing question. Domain framing belongs here, not in the
|