@tangle-network/agent-eval 0.80.0 → 0.81.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +53 -152
- package/dist/adapters/http.d.ts +2 -2
- package/dist/adapters/langchain.d.ts +2 -2
- package/dist/adapters/otel.d.ts +4 -4
- package/dist/{agent-profile-aSEaJ9Pl.d.ts → agent-profile-D0PBIWlV.d.ts} +14 -2
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +3 -3
- package/dist/{analyst-t7zZS3TV.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
- package/dist/belief-state/index.d.ts +344 -8
- package/dist/belief-state/index.js +1518 -142
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -2
- package/dist/campaign/index.d.ts +40 -120
- package/dist/campaign/index.js +129 -238
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-RPLZ4OIB.js → chunk-BABOZOSN.js} +7 -4
- package/dist/{chunk-RPLZ4OIB.js.map → chunk-BABOZOSN.js.map} +1 -1
- package/dist/{chunk-IHDHUN2X.js → chunk-CVVHBFGN.js} +3 -3
- package/dist/chunk-CVVHBFGN.js.map +1 -0
- package/dist/{chunk-B26KI423.js → chunk-FZWAFVAA.js} +2 -2
- package/dist/{chunk-ITBRCT73.js → chunk-IDVBLYCY.js} +2 -10
- package/dist/chunk-IDVBLYCY.js.map +1 -0
- package/dist/{chunk-5LVWPNS5.js → chunk-L5G7OUKD.js} +4 -4
- package/dist/{chunk-5LVWPNS5.js.map → chunk-L5G7OUKD.js.map} +1 -1
- package/dist/{chunk-GXHLRXDI.js → chunk-OTYQPHPL.js} +4 -4
- package/dist/{chunk-6REHLN5J.js → chunk-QS3RBQPI.js} +2 -2
- package/dist/{chunk-GWGO2K6Y.js → chunk-RBNA5AZT.js} +2 -2
- package/dist/chunk-S42AWHMP.js +697 -0
- package/dist/chunk-S42AWHMP.js.map +1 -0
- package/dist/chunk-VI2UW6B6.js +162 -0
- package/dist/chunk-VI2UW6B6.js.map +1 -0
- package/dist/{chunk-CF67I6QY.js → chunk-VIDQF3F5.js} +2 -2
- package/dist/{chunk-XXNIODOM.js → chunk-WJL2NJXN.js} +3 -3
- package/dist/{chunk-LB2UOI5F.js → chunk-YGYXHNAQ.js} +47 -160
- package/dist/chunk-YGYXHNAQ.js.map +1 -0
- package/dist/{chunk-KX6F6NCG.js → chunk-Z7VFTS2J.js} +2 -2
- package/dist/{chunk-ZPSKPT3V.js → chunk-ZZ2HOPME.js} +5 -2
- package/dist/chunk-ZZ2HOPME.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/code-agent-session-BRXmavYv.d.ts +80 -0
- package/dist/contract/index.d.ts +16 -15
- package/dist/contract/index.js +23 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-CehLtoET.d.ts → control-GeE8OhpN.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/hosted/index.d.ts +4 -4
- package/dist/{index-B1RKber3.d.ts → index-DE3RXAXD.d.ts} +1 -1
- package/dist/index.d.ts +78 -287
- package/dist/index.js +87 -410
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-dlpEzQDi.d.ts → insight-report-3ADTfClO.d.ts} +1 -1
- package/dist/{kind-factory-DqV2t1Xk.d.ts → kind-factory-CVecZZG_.d.ts} +2 -2
- package/dist/{llm-client-DbjLfz-K.d.ts → llm-client-CuUg2Mn3.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/{provenance-jG-Gngg8.d.ts → provenance-B9Q4886D.d.ts} +4 -4
- package/dist/{registry-BK0Zee01.d.ts → registry-DrEQ3Luj.d.ts} +1 -1
- package/dist/{release-report-CXXZlR8g.d.ts → release-report-hlNtD12q.d.ts} +2 -2
- package/dist/reporting.d.ts +5 -5
- package/dist/reporting.js +3 -3
- package/dist/{researcher-rInLj9De.d.ts → researcher-BLPHBbNV.d.ts} +3 -3
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +4 -4
- package/dist/{rubric-predictive-validity-CLPuwiUw.d.ts → rubric-predictive-validity-CnEl9Jc8.d.ts} +1 -1
- package/dist/{run-campaign-OVEZF24D.js → run-campaign-4Y5V5CN3.js} +3 -3
- package/dist/{run-improvement-loop-BAl_aVOZ.d.ts → run-improvement-loop-D6PZOoQL.d.ts} +2 -2
- package/dist/{run-record-sItO5ftF.d.ts → run-record-De9VarXR.d.ts} +1 -1
- package/dist/{semantic-concept-judge-qXEUV2w7.d.ts → semantic-concept-judge-DIEgr_6v.d.ts} +6 -6
- package/dist/{statistics-B7yCbi9i.d.ts → statistics-CnC1FMbx.d.ts} +3 -7
- package/dist/{store-GmBE2pZZ.d.ts → store-C1YxJDEK.d.ts} +1 -1
- package/dist/{summary-report-BTaXq1TS.d.ts → summary-report-Db0dDSWP.d.ts} +1 -1
- package/dist/traces.d.ts +5 -5
- package/dist/{types-DRvV0zRo.d.ts → types-Cu3u_x59.d.ts} +3 -3
- package/dist/{types-4mm2msnR.d.ts → types-D7lLRYe9.d.ts} +1 -1
- package/dist/wire/index.js +2 -2
- package/dist/workflow/index.d.ts +7 -7
- package/dist/workflow/index.js +1 -1
- package/docs/concepts.md +1 -0
- package/docs/research/belief-state-agent-eval-roadmap.md +39 -7
- package/docs/self-improvement-map.md +111 -0
- package/package.json +2 -2
- package/dist/chunk-IHDHUN2X.js.map +0 -1
- package/dist/chunk-ITBRCT73.js.map +0 -1
- package/dist/chunk-LB2UOI5F.js.map +0 -1
- package/dist/chunk-ZPSKPT3V.js.map +0 -1
- /package/dist/{chunk-B26KI423.js.map → chunk-FZWAFVAA.js.map} +0 -0
- /package/dist/{chunk-GXHLRXDI.js.map → chunk-OTYQPHPL.js.map} +0 -0
- /package/dist/{chunk-6REHLN5J.js.map → chunk-QS3RBQPI.js.map} +0 -0
- /package/dist/{chunk-GWGO2K6Y.js.map → chunk-RBNA5AZT.js.map} +0 -0
- /package/dist/{chunk-CF67I6QY.js.map → chunk-VIDQF3F5.js.map} +0 -0
- /package/dist/{chunk-XXNIODOM.js.map → chunk-WJL2NJXN.js.map} +0 -0
- /package/dist/{chunk-KX6F6NCG.js.map → chunk-Z7VFTS2J.js.map} +0 -0
- /package/dist/{run-campaign-OVEZF24D.js.map → run-campaign-4Y5V5CN3.js.map} +0 -0
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,20 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.81.0] — 2026-06-05 — eval-campaign scaffold prep primitives
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- **`aggregateJudgeVerdicts<D>` (root).** Generic judge-ensemble reducer: fan out N uncorrelated judges, mean each rubric dimension over the SURVIVORS, report the inter-rater disagreement spread, sum cost. Replaces the same reduction hand-rolled in legal (`aggregateEnsemble`), creative (`production-loop/judges.ts`), and tax (`judge-ensemble.ts`). Fail-loud: a failed judge (`perDimension: null`) is recorded in `failedJudges`, never folded into a zero; all-failed throws; a failed judge's cost is still summed. Composite reuses `weightedComposite`.
|
|
12
|
+
- **`createTokenRecallChecker` (root).** The deterministic, no-LLM `CorrectnessChecker` — sibling of `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its content is substantive and recalls ≥ `minRecall` of the requirement title's significant tokens. The default completion gate for apps/tests without an LLM judge.
|
|
13
|
+
- **`ErrorCluster` (root + `/analyst`).** The failure-cluster element type is now a named export, so consumers import it instead of deriving `DatasetOverview['error_clusters'][number]`.
|
|
14
|
+
|
|
15
|
+
### Fixed
|
|
16
|
+
|
|
17
|
+
- **Lint drift + non-executable pre-commit hook.** `.husky/pre-commit` was tracked `100644`, so the hook silently no-op'd and unformatted code reached `main`; marked executable and reformatted the drift.
|
|
18
|
+
|
|
19
|
+
---
|
|
20
|
+
|
|
7
21
|
## [0.72.3] — 2026-06-01 — workflow trace hardening and driver backtests
|
|
8
22
|
|
|
9
23
|
### Added
|
package/README.md
CHANGED
|
@@ -1,199 +1,106 @@
|
|
|
1
1
|
# `@tangle-network/agent-eval`
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Evaluate and improve AI agents from the runs they already produce.
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
`agent-eval` turns agent outputs, traces, judge scores, and production feedback into a decision packet: did this change help, what failed, what should ship, and what needs more data?
|
|
6
6
|
|
|
7
7
|
[](https://www.npmjs.com/package/@tangle-network/agent-eval)
|
|
8
8
|
[](https://pypi.org/project/agent-eval-rpc/)
|
|
9
9
|
[](https://github.com/tangle-network/agent-eval/actions/workflows/ci.yml)
|
|
10
10
|
[](./LICENSE)
|
|
11
11
|
|
|
12
|
-
|
|
12
|
+
Use it when you need to:
|
|
13
13
|
|
|
14
|
-
|
|
14
|
+
- compare a candidate agent/prompt/model against a baseline,
|
|
15
|
+
- turn production traces or human feedback into eval results,
|
|
16
|
+
- run a gated self-improvement loop,
|
|
17
|
+
- explain failures by cluster, cost, judge disagreement, and release risk.
|
|
15
18
|
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
- [What you get back](#what-you-get-back-the-decision-packet)
|
|
19
|
-
- [Quick start](#quick-start)
|
|
20
|
-
- [Closed loop — `selfImprove()`](#closed-loop--selfimprove)
|
|
21
|
-
- [Observed runs — `analyzeRuns()`](#observed-runs--analyzeruns)
|
|
22
|
-
- [Existing data — intake adapters](#existing-data--intake-adapters)
|
|
23
|
-
- [How it compares](#how-it-compares)
|
|
24
|
-
- [Customer journeys](#customer-journeys)
|
|
25
|
-
- [Subpath entry points](#subpath-entry-points)
|
|
26
|
-
- [Concepts + design](#concepts--design)
|
|
27
|
-
- [Hosted tier](#hosted-tier)
|
|
28
|
-
- [Install + run](#install--run)
|
|
29
|
-
- [Stability + versioning](#stability--versioning)
|
|
30
|
-
- [License](#license)
|
|
19
|
+
It is a library, not a SaaS requirement. TypeScript is first-class; Python can call the same wire protocol through `agent-eval-rpc`.
|
|
31
20
|
|
|
32
21
|
---
|
|
33
22
|
|
|
34
|
-
##
|
|
35
|
-
|
|
36
|
-
Whether you call `selfImprove()` (closed loop) or `analyzeRuns()` (observed runs), the report has the same shape. Here's a real one, abridged:
|
|
23
|
+
## Install
|
|
37
24
|
|
|
38
|
-
```
|
|
39
|
-
|
|
40
|
-
"n": 80, // runs analyzed
|
|
41
|
-
"composite": { // distributional summary
|
|
42
|
-
"mean": 0.62, "p50": 0.65, "p95": 0.88, "stddev": 0.17,
|
|
43
|
-
"histogram": [/* 12 bins */]
|
|
44
|
-
},
|
|
45
|
-
"lift": { // paired bootstrap
|
|
46
|
-
"baselineMean": 0.58, "candidateMean": 0.65,
|
|
47
|
-
"delta": 0.07,
|
|
48
|
-
"ci95": [0.04, 0.10], // 95% CI on the delta
|
|
49
|
-
"pValue": 0.0008, // paired-t
|
|
50
|
-
"cohensD": 0.41,
|
|
51
|
-
"n": 40,
|
|
52
|
-
"mde": 0.06, // min detectable effect at 80% power
|
|
53
|
-
"requiredN": 38 // n needed to detect observed delta
|
|
54
|
-
},
|
|
55
|
-
"judges": { // per-judge calibration
|
|
56
|
-
"domain-expert": { "n": 80, "meanScore": 0.64 },
|
|
57
|
-
"helpfulness-llm": { "n": 80, "meanScore": 0.61 }
|
|
58
|
-
},
|
|
59
|
-
"interRater": { // multi-rater agreement
|
|
60
|
-
"raters": 3, "jointlyRated": 80, "kappa": 0.71,
|
|
61
|
-
"disagreementCases": [/* top 20 ranked by spread */]
|
|
62
|
-
},
|
|
63
|
-
"costQuality": { // cost-vs-quality
|
|
64
|
-
"cost": { "mean": 0.024, "p95": 0.041, /* ... */ },
|
|
65
|
-
"pareto": { /* ParetoFigureSpec the dashboard renders */ }
|
|
66
|
-
},
|
|
67
|
-
"failureClusters": { // when an AnalystRegistry is wired
|
|
68
|
-
"totalFailures": 11,
|
|
69
|
-
"clusters": [
|
|
70
|
-
{ "name": "off-topic-drift", "share": 0.45, "exemplars": ["run-12", "run-19"] },
|
|
71
|
-
{ "name": "over-confidence", "share": 0.27, "exemplars": ["run-3"] },
|
|
72
|
-
{ "name": "format-mismatch", "share": 0.18, "exemplars": ["run-41"] }
|
|
73
|
-
]
|
|
74
|
-
},
|
|
75
|
-
"contamination": { "leaks": 0, "holdoutAuditPassed": true },
|
|
76
|
-
"outcomeCorrelation": { // when downstream metric supplied
|
|
77
|
-
"metric": "engagement_rate", "n": 80,
|
|
78
|
-
"pearson": 0.72, "spearman": 0.69,
|
|
79
|
-
"rewardModel": { "intercept": 0.04, "slope": 1.93, "r2": 0.52 }
|
|
80
|
-
},
|
|
81
|
-
"release": {
|
|
82
|
-
"status": "pass",
|
|
83
|
-
"axes": [
|
|
84
|
-
{ "name": "quality-lift", "status": "pass" },
|
|
85
|
-
{ "name": "contamination", "status": "pass" },
|
|
86
|
-
{ "name": "composite-distribution","status": "pass" }
|
|
87
|
-
]
|
|
88
|
-
},
|
|
89
|
-
"recommendations": [
|
|
90
|
-
{ "priority": "critical", "kind": "ship",
|
|
91
|
-
"title": "Ship — lift 0.070 (95% CI 0.040..0.100)",
|
|
92
|
-
"detail": "Holdout lift exceeds threshold 0.02 with 95% bootstrap confidence (n=40, p=0.0008, d=0.41)." },
|
|
93
|
-
{ "priority": "high", "kind": "investigate",
|
|
94
|
-
"title": "Top failure cluster: off-topic-drift (45% of failures)",
|
|
95
|
-
"detail": "11 runs failed. Drill into exemplars run-12 / run-19 to identify the pattern." }
|
|
96
|
-
]
|
|
97
|
-
}
|
|
25
|
+
```sh
|
|
26
|
+
pnpm add @tangle-network/agent-eval
|
|
98
27
|
```
|
|
99
28
|
|
|
100
|
-
|
|
29
|
+
Python clients can use the RPC package:
|
|
30
|
+
|
|
31
|
+
```sh
|
|
32
|
+
pip install agent-eval-rpc
|
|
33
|
+
```
|
|
101
34
|
|
|
102
35
|
---
|
|
103
36
|
|
|
104
37
|
## Quick start
|
|
105
38
|
|
|
106
|
-
###
|
|
39
|
+
### 1. Analyze runs you already have
|
|
107
40
|
|
|
108
|
-
|
|
41
|
+
Start here if you already have production logs, benchmark rows, human ratings, or agent run records.
|
|
109
42
|
|
|
110
43
|
```ts
|
|
111
|
-
import {
|
|
44
|
+
import { analyzeRuns } from '@tangle-network/agent-eval/contract'
|
|
112
45
|
|
|
113
|
-
const
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
await myAgent.run(scenario),
|
|
117
|
-
judges: [myJudge], // any JudgeConfig — LLM, rule, ensemble
|
|
118
|
-
baselineSurface: { systemPrompt: currentPrompt },
|
|
46
|
+
const report = await analyzeRuns({
|
|
47
|
+
runs, // RunRecord[]
|
|
48
|
+
baselineRuns,
|
|
119
49
|
})
|
|
120
50
|
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
51
|
+
console.log(report.recommendations)
|
|
52
|
+
console.log(report.lift)
|
|
53
|
+
console.log(report.failureClusters)
|
|
124
54
|
```
|
|
125
55
|
|
|
126
|
-
|
|
56
|
+
The output includes score distributions, lift confidence intervals, failure modes, cost-quality tradeoffs, judge agreement, contamination checks, and release recommendations when the input supports them.
|
|
57
|
+
|
|
58
|
+
### 2. Run a gated improvement loop
|
|
127
59
|
|
|
128
|
-
|
|
60
|
+
Use this when you have scenarios, a runnable agent, and judges.
|
|
129
61
|
|
|
130
62
|
```ts
|
|
131
|
-
import {
|
|
63
|
+
import { selfImprove } from '@tangle-network/agent-eval/contract'
|
|
132
64
|
|
|
133
|
-
const
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
},
|
|
139
|
-
canaryScenarios, // optional — contamination probe
|
|
140
|
-
analyst: myAnalystRegistry, // optional — AI-powered failure clustering
|
|
65
|
+
const result = await selfImprove({
|
|
66
|
+
scenarios,
|
|
67
|
+
dispatch: async ({ scenario }) => myAgent.run(scenario),
|
|
68
|
+
judges: [myJudge],
|
|
69
|
+
baselineSurface: { systemPrompt: currentPrompt },
|
|
141
70
|
})
|
|
142
71
|
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
72
|
+
console.log(result.gateDecision)
|
|
73
|
+
console.log(result.winnerSurface)
|
|
74
|
+
console.log(result.insight.recommendations)
|
|
146
75
|
```
|
|
147
76
|
|
|
148
|
-
|
|
77
|
+
`selfImprove()` evaluates candidates on held-out scenarios before recommending a winner.
|
|
149
78
|
|
|
150
|
-
|
|
79
|
+
### 3. Adapt existing data
|
|
151
80
|
|
|
152
81
|
```ts
|
|
153
|
-
import {
|
|
154
|
-
fromFeedbackTable,
|
|
155
|
-
fromOtelSpans,
|
|
156
|
-
analyzeRuns,
|
|
157
|
-
} from '@tangle-network/agent-eval/contract'
|
|
82
|
+
import { analyzeRuns, fromFeedbackTable, fromOtelSpans } from '@tangle-network/agent-eval/contract'
|
|
158
83
|
|
|
159
|
-
// Multi-rater approve/reject (Obsidian tags, Sheets, CSV, Postgres).
|
|
160
84
|
const { runs, raterScores } = fromFeedbackTable({
|
|
161
|
-
ratings: parseYourFeedbackTable(),
|
|
85
|
+
ratings: parseYourFeedbackTable(),
|
|
162
86
|
})
|
|
163
|
-
await analyzeRuns({ runs, raterScores })
|
|
164
87
|
|
|
165
|
-
|
|
166
|
-
const runs2 = fromOtelSpans({ spans: yourOtelStream })
|
|
167
|
-
await analyzeRuns({ runs: runs2 })
|
|
168
|
-
```
|
|
88
|
+
const traceRuns = fromOtelSpans({ spans: yourOtelSpans })
|
|
169
89
|
|
|
170
|
-
|
|
90
|
+
await analyzeRuns({ runs: [...runs, ...traceRuns], raterScores })
|
|
91
|
+
```
|
|
171
92
|
|
|
172
93
|
---
|
|
173
94
|
|
|
174
|
-
##
|
|
95
|
+
## Core concepts
|
|
175
96
|
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
| Inter-rater agreement + disagreement triage | — | — | — | ✓ |
|
|
182
|
-
| Contamination / canary check | — | — | — | ✓ |
|
|
183
|
-
| AI-driven failure clustering | partial | — | partial | ✓ |
|
|
184
|
-
| Cost-quality Pareto | — | — | — | ✓ |
|
|
185
|
-
| Multi-language clients (TS + Python) | TS only | TS only | TS + Py | ✓ TS + Py |
|
|
186
|
-
| Self-hostable / no-SaaS option | — | — | OSS | ✓ MIT, OSS |
|
|
187
|
-
| Substrate vs SaaS shape | SaaS | SaaS | OSS server | **library** |
|
|
188
|
-
| Hosted tier (optional) | required | required | optional | optional |
|
|
97
|
+
- **RunRecord**: the durable row for one agent run: model, prompt/config hashes, split, cost, tokens, outcome.
|
|
98
|
+
- **Scenario**: one task or case the agent attempts.
|
|
99
|
+
- **Judge**: a scoring function, rule-based or model-based.
|
|
100
|
+
- **InsightReport**: the decision packet returned by `analyzeRuns()` and embedded in `selfImprove()`.
|
|
101
|
+
- **Gate**: the policy that decides `ship`, `hold`, or `need_more_data`.
|
|
189
102
|
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
---
|
|
193
|
-
|
|
194
|
-
## Customer journeys
|
|
195
|
-
|
|
196
|
-
Three runnable examples — each is self-contained, each shows the actual output.
|
|
103
|
+
## Examples
|
|
197
104
|
|
|
198
105
|
| Journey | Example | Who it's for |
|
|
199
106
|
|---|---|---|
|
|
@@ -288,13 +195,7 @@ The substrate runs the loop in your process. Only the eval-run events + (optiona
|
|
|
288
195
|
|
|
289
196
|
---
|
|
290
197
|
|
|
291
|
-
##
|
|
292
|
-
|
|
293
|
-
```sh
|
|
294
|
-
pnpm add @tangle-network/agent-eval
|
|
295
|
-
# or, from Python:
|
|
296
|
-
pip install agent-eval-rpc
|
|
297
|
-
```
|
|
198
|
+
## Development
|
|
298
199
|
|
|
299
200
|
Run an example:
|
|
300
201
|
|
package/dist/adapters/http.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-
|
|
2
|
-
import '../run-record-
|
|
1
|
+
import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-D7lLRYe9.js';
|
|
2
|
+
import '../run-record-De9VarXR.js';
|
|
3
3
|
import '../errors-Dwqw-T_m.js';
|
|
4
4
|
import '../schema-m0gsnbt3.js';
|
|
5
5
|
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-
|
|
2
|
-
import '../run-record-
|
|
1
|
+
import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-D7lLRYe9.js';
|
|
2
|
+
import '../run-record-De9VarXR.js';
|
|
3
3
|
import '../errors-Dwqw-T_m.js';
|
|
4
4
|
import '../schema-m0gsnbt3.js';
|
|
5
5
|
|
package/dist/adapters/otel.d.ts
CHANGED
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
import { TraceSpanEvent, HostedClient } from '../hosted/index.js';
|
|
2
|
-
import '../types-
|
|
3
|
-
import '../run-record-
|
|
2
|
+
import '../types-D7lLRYe9.js';
|
|
3
|
+
import '../run-record-De9VarXR.js';
|
|
4
4
|
import '../errors-Dwqw-T_m.js';
|
|
5
5
|
import '../schema-m0gsnbt3.js';
|
|
6
|
-
import '../insight-report-
|
|
7
|
-
import '../summary-report-
|
|
6
|
+
import '../insight-report-3ADTfClO.js';
|
|
7
|
+
import '../summary-report-Db0dDSWP.js';
|
|
8
8
|
import '../failure-cluster-CL7IVgkJ.js';
|
|
9
9
|
import '../store-CKUAgsJz.js';
|
|
10
10
|
import '../judge-calibration-DilmB3Ml.js';
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { A as AgentEvalError } from './errors-Dwqw-T_m.js';
|
|
2
|
-
import { R as RunRecord } from './run-record-
|
|
2
|
+
import { R as RunRecord } from './run-record-De9VarXR.js';
|
|
3
3
|
import { TCloud } from '@tangle-network/tcloud';
|
|
4
4
|
|
|
5
5
|
/**
|
|
@@ -258,6 +258,18 @@ declare function parseCorrectnessResponse(raw: string): {
|
|
|
258
258
|
* fulfil a requirement — the artifact must BE the deliverable.
|
|
259
259
|
*/
|
|
260
260
|
declare function createLlmCorrectnessChecker(tc: TCloud, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
|
|
261
|
+
/**
|
|
262
|
+
* Deterministic `CorrectnessChecker` — the no-LLM counterpart to
|
|
263
|
+
* `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its
|
|
264
|
+
* content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`
|
|
265
|
+
* of the requirement title's significant tokens. No network — the default gate
|
|
266
|
+
* for apps and tests without an LLM judge. Pass to `verifyCompletion` as the
|
|
267
|
+
* checker.
|
|
268
|
+
*/
|
|
269
|
+
declare function createTokenRecallChecker(opts?: {
|
|
270
|
+
minRecall?: number;
|
|
271
|
+
minContentLength?: number;
|
|
272
|
+
}): CorrectnessChecker;
|
|
261
273
|
|
|
262
274
|
/**
|
|
263
275
|
* Produced-state extraction — normalize a run's runtime event stream into the
|
|
@@ -361,4 +373,4 @@ interface AgentProfile {
|
|
|
361
373
|
*/
|
|
362
374
|
declare function agentProfileHash(profile: AgentProfile): string;
|
|
363
375
|
|
|
364
|
-
export { type AgentProfile as A, type BackendIntegrityReport as B, type CompletionRequirement as C, type LlmCorrectnessCheckerOpts as L, type ProducedState as P, type RuntimeEventLike as R, type SatisfiedBy as S, type TaskGold as T, type ValidationContext as V, type CompletionVerdict as a, type CorrectnessChecker as b, type Artifact as c, type ArtifactEventLike as d, type ArtifactValidator as e, BackendIntegrityError as f, type ProducedProposal as g, type ProposalEventLike as h, type RequirementCheck as i, type ToolCallEventLike as j, type ValidationIssue as k, type ValidationResult as l, agentProfileHash as m, assertRealBackend as n, byteLengthRange as o, composeValidators as p, containsAll as q, createLlmCorrectnessChecker as r,
|
|
376
|
+
export { type AgentProfile as A, type BackendIntegrityReport as B, type CompletionRequirement as C, type LlmCorrectnessCheckerOpts as L, type ProducedState as P, type RuntimeEventLike as R, type SatisfiedBy as S, type TaskGold as T, type ValidationContext as V, type CompletionVerdict as a, type CorrectnessChecker as b, type Artifact as c, type ArtifactEventLike as d, type ArtifactValidator as e, BackendIntegrityError as f, type ProducedProposal as g, type ProposalEventLike as h, type RequirementCheck as i, type ToolCallEventLike as j, type ValidationIssue as k, type ValidationResult as l, agentProfileHash as m, assertRealBackend as n, byteLengthRange as o, composeValidators as p, containsAll as q, createLlmCorrectnessChecker as r, createTokenRecallChecker as s, extractProducedState as t, jsonHasKeys as u, parseCorrectnessResponse as v, regexMatch as w, summarizeBackendIntegrity as x, verifyCompletion as y };
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,21 +1,21 @@
|
|
|
1
1
|
import { AxAIService, AxFunction } from '@ax-llm/ax';
|
|
2
2
|
import { M as MultiLayerVerifier, V as VerifyOptions, S as Severity } from '../multi-layer-verifier-DlWCXuxL.js';
|
|
3
3
|
import { c as RunCritic, a as RunTrace } from '../run-critic-BAIjX99r.js';
|
|
4
|
-
import { S as SemanticConceptJudgeOptions, a as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-
|
|
5
|
-
export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, b as DefaultAnalystRegistryOptions, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FindingSubject, g as FindingSubjectKind, h as FindingSubjectStringSchema, i as FindingsDiff, j as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, k as KNOWLEDGE_GAP_KIND_SPEC, l as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, m as SKILL_USAGE_ANALYST, n as SkillUsageAnalyst, o as SkillUsageRecord, p as SkillUsageReport, q as SkillUsageScanConfig, r as buildDefaultAnalystRegistry, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as parseFindingSubject, y as renderFindingSubject } from '../semantic-concept-judge-
|
|
6
|
-
import { A as AnalyzeTracesOptions } from '../analyst-
|
|
7
|
-
import { T as TraceAnalysisStore } from '../store-
|
|
4
|
+
import { S as SemanticConceptJudgeOptions, a as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-DIEgr_6v.js';
|
|
5
|
+
export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, b as DefaultAnalystRegistryOptions, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FindingSubject, g as FindingSubjectKind, h as FindingSubjectStringSchema, i as FindingsDiff, j as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, k as KNOWLEDGE_GAP_KIND_SPEC, l as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, m as SKILL_USAGE_ANALYST, n as SkillUsageAnalyst, o as SkillUsageRecord, p as SkillUsageReport, q as SkillUsageScanConfig, r as buildDefaultAnalystRegistry, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as parseFindingSubject, y as renderFindingSubject } from '../semantic-concept-judge-DIEgr_6v.js';
|
|
6
|
+
import { A as AnalyzeTracesOptions } from '../analyst-C8HHvfJp.js';
|
|
7
|
+
import { T as TraceAnalysisStore } from '../store-C1YxJDEK.js';
|
|
8
8
|
import { b as JudgeFn, a as JudgeInput } from '../types-Croy5h7V.js';
|
|
9
|
-
import { A as Analyst, h as AnalystSeverity, c as AnalystFinding } from '../types-
|
|
10
|
-
export { a as AnalystContext, g as AnalystCost, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, k as ChatCallOpts, C as ChatClient, l as ChatRequest, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, p as CreateChatClientOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from '../types-
|
|
9
|
+
import { A as Analyst, h as AnalystSeverity, c as AnalystFinding } from '../types-Cu3u_x59.js';
|
|
10
|
+
export { a as AnalystContext, g as AnalystCost, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, k as ChatCallOpts, C as ChatClient, l as ChatRequest, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, p as CreateChatClientOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from '../types-Cu3u_x59.js';
|
|
11
11
|
import { TCloud } from '@tangle-network/tcloud';
|
|
12
|
-
export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, T as TraceAnalystGolden, c as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-
|
|
13
|
-
export { A as AnalystHooks, a as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, R as RegistryRunOpts } from '../registry-
|
|
14
|
-
import { L as LlmClientOptions } from '../llm-client-
|
|
12
|
+
export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, T as TraceAnalystGolden, c as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-CVecZZG_.js';
|
|
13
|
+
export { A as AnalystHooks, a as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, R as RegistryRunOpts } from '../registry-DrEQ3Luj.js';
|
|
14
|
+
import { L as LlmClientOptions } from '../llm-client-CuUg2Mn3.js';
|
|
15
15
|
import '../schema-m0gsnbt3.js';
|
|
16
16
|
import '../store-CKUAgsJz.js';
|
|
17
17
|
import 'zod';
|
|
18
|
-
import '../run-record-
|
|
18
|
+
import '../run-record-De9VarXR.js';
|
|
19
19
|
import '../errors-Dwqw-T_m.js';
|
|
20
20
|
import '../raw-provider-sink-C46HDghv.js';
|
|
21
21
|
|
package/dist/analyst/index.js
CHANGED
|
@@ -14,7 +14,7 @@ import {
|
|
|
14
14
|
diffFindings,
|
|
15
15
|
emitSkillUsageFindings,
|
|
16
16
|
runSemanticConceptJudge
|
|
17
|
-
} from "../chunk-
|
|
17
|
+
} from "../chunk-L5G7OUKD.js";
|
|
18
18
|
import {
|
|
19
19
|
ANALYST_SEVERITIES,
|
|
20
20
|
AnalystRegistry,
|
|
@@ -41,8 +41,8 @@ import {
|
|
|
41
41
|
renderPriorFindings,
|
|
42
42
|
stripCodeFences,
|
|
43
43
|
structureFindings
|
|
44
|
-
} from "../chunk-
|
|
45
|
-
import "../chunk-
|
|
44
|
+
} from "../chunk-VIDQF3F5.js";
|
|
45
|
+
import "../chunk-CVVHBFGN.js";
|
|
46
46
|
import {
|
|
47
47
|
analyzeTraces
|
|
48
48
|
} from "../chunk-VUINJM5M.js";
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { AxAIService } from '@ax-llm/ax';
|
|
2
|
-
import { T as TraceAnalysisStore } from './store-
|
|
2
|
+
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
3
|
|
|
4
4
|
interface AnalyzeTracesInput {
|
|
5
5
|
/** The user-facing question. Domain framing belongs here, not in the
|