@tangle-network/agent-eval 0.80.0 → 0.81.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/README.md +53 -152
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/adapters/langchain.d.ts +2 -2
  5. package/dist/adapters/otel.d.ts +4 -4
  6. package/dist/{agent-profile-aSEaJ9Pl.d.ts → agent-profile-D0PBIWlV.d.ts} +14 -2
  7. package/dist/analyst/index.d.ts +10 -10
  8. package/dist/analyst/index.js +3 -3
  9. package/dist/{analyst-t7zZS3TV.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
  10. package/dist/belief-state/index.d.ts +344 -8
  11. package/dist/belief-state/index.js +1518 -142
  12. package/dist/belief-state/index.js.map +1 -1
  13. package/dist/benchmarks/index.d.ts +2 -2
  14. package/dist/campaign/index.d.ts +40 -120
  15. package/dist/campaign/index.js +129 -238
  16. package/dist/campaign/index.js.map +1 -1
  17. package/dist/{chunk-RPLZ4OIB.js → chunk-BABOZOSN.js} +7 -4
  18. package/dist/{chunk-RPLZ4OIB.js.map → chunk-BABOZOSN.js.map} +1 -1
  19. package/dist/{chunk-IHDHUN2X.js → chunk-CVVHBFGN.js} +3 -3
  20. package/dist/chunk-CVVHBFGN.js.map +1 -0
  21. package/dist/{chunk-B26KI423.js → chunk-FZWAFVAA.js} +2 -2
  22. package/dist/{chunk-ITBRCT73.js → chunk-IDVBLYCY.js} +2 -10
  23. package/dist/chunk-IDVBLYCY.js.map +1 -0
  24. package/dist/{chunk-5LVWPNS5.js → chunk-L5G7OUKD.js} +4 -4
  25. package/dist/{chunk-5LVWPNS5.js.map → chunk-L5G7OUKD.js.map} +1 -1
  26. package/dist/{chunk-GXHLRXDI.js → chunk-OTYQPHPL.js} +4 -4
  27. package/dist/{chunk-6REHLN5J.js → chunk-QS3RBQPI.js} +2 -2
  28. package/dist/{chunk-GWGO2K6Y.js → chunk-RBNA5AZT.js} +2 -2
  29. package/dist/chunk-S42AWHMP.js +697 -0
  30. package/dist/chunk-S42AWHMP.js.map +1 -0
  31. package/dist/chunk-VI2UW6B6.js +162 -0
  32. package/dist/chunk-VI2UW6B6.js.map +1 -0
  33. package/dist/{chunk-CF67I6QY.js → chunk-VIDQF3F5.js} +2 -2
  34. package/dist/{chunk-XXNIODOM.js → chunk-WJL2NJXN.js} +3 -3
  35. package/dist/{chunk-LB2UOI5F.js → chunk-YGYXHNAQ.js} +47 -160
  36. package/dist/chunk-YGYXHNAQ.js.map +1 -0
  37. package/dist/{chunk-KX6F6NCG.js → chunk-Z7VFTS2J.js} +2 -2
  38. package/dist/{chunk-ZPSKPT3V.js → chunk-ZZ2HOPME.js} +5 -2
  39. package/dist/chunk-ZZ2HOPME.js.map +1 -0
  40. package/dist/cli.js +2 -2
  41. package/dist/code-agent-session-BRXmavYv.d.ts +80 -0
  42. package/dist/contract/index.d.ts +16 -15
  43. package/dist/contract/index.js +23 -6
  44. package/dist/contract/index.js.map +1 -1
  45. package/dist/{control-CehLtoET.d.ts → control-GeE8OhpN.d.ts} +1 -1
  46. package/dist/control.d.ts +2 -2
  47. package/dist/hosted/index.d.ts +4 -4
  48. package/dist/{index-B1RKber3.d.ts → index-DE3RXAXD.d.ts} +1 -1
  49. package/dist/index.d.ts +78 -287
  50. package/dist/index.js +87 -410
  51. package/dist/index.js.map +1 -1
  52. package/dist/{insight-report-dlpEzQDi.d.ts → insight-report-3ADTfClO.d.ts} +1 -1
  53. package/dist/{kind-factory-DqV2t1Xk.d.ts → kind-factory-CVecZZG_.d.ts} +2 -2
  54. package/dist/{llm-client-DbjLfz-K.d.ts → llm-client-CuUg2Mn3.d.ts} +1 -1
  55. package/dist/meta-eval/index.d.ts +2 -2
  56. package/dist/openapi.json +1 -1
  57. package/dist/pipelines/index.js +2 -2
  58. package/dist/{provenance-jG-Gngg8.d.ts → provenance-B9Q4886D.d.ts} +4 -4
  59. package/dist/{registry-BK0Zee01.d.ts → registry-DrEQ3Luj.d.ts} +1 -1
  60. package/dist/{release-report-CXXZlR8g.d.ts → release-report-hlNtD12q.d.ts} +2 -2
  61. package/dist/reporting.d.ts +5 -5
  62. package/dist/reporting.js +3 -3
  63. package/dist/{researcher-rInLj9De.d.ts → researcher-BLPHBbNV.d.ts} +3 -3
  64. package/dist/rl.d.ts +7 -7
  65. package/dist/rl.js +4 -4
  66. package/dist/{rubric-predictive-validity-CLPuwiUw.d.ts → rubric-predictive-validity-CnEl9Jc8.d.ts} +1 -1
  67. package/dist/{run-campaign-OVEZF24D.js → run-campaign-4Y5V5CN3.js} +3 -3
  68. package/dist/{run-improvement-loop-BAl_aVOZ.d.ts → run-improvement-loop-D6PZOoQL.d.ts} +2 -2
  69. package/dist/{run-record-sItO5ftF.d.ts → run-record-De9VarXR.d.ts} +1 -1
  70. package/dist/{semantic-concept-judge-qXEUV2w7.d.ts → semantic-concept-judge-DIEgr_6v.d.ts} +6 -6
  71. package/dist/{statistics-B7yCbi9i.d.ts → statistics-CnC1FMbx.d.ts} +3 -7
  72. package/dist/{store-GmBE2pZZ.d.ts → store-C1YxJDEK.d.ts} +1 -1
  73. package/dist/{summary-report-BTaXq1TS.d.ts → summary-report-Db0dDSWP.d.ts} +1 -1
  74. package/dist/traces.d.ts +5 -5
  75. package/dist/{types-DRvV0zRo.d.ts → types-Cu3u_x59.d.ts} +3 -3
  76. package/dist/{types-4mm2msnR.d.ts → types-D7lLRYe9.d.ts} +1 -1
  77. package/dist/wire/index.js +2 -2
  78. package/dist/workflow/index.d.ts +7 -7
  79. package/dist/workflow/index.js +1 -1
  80. package/docs/concepts.md +1 -0
  81. package/docs/research/belief-state-agent-eval-roadmap.md +39 -7
  82. package/docs/self-improvement-map.md +111 -0
  83. package/package.json +2 -2
  84. package/dist/chunk-IHDHUN2X.js.map +0 -1
  85. package/dist/chunk-ITBRCT73.js.map +0 -1
  86. package/dist/chunk-LB2UOI5F.js.map +0 -1
  87. package/dist/chunk-ZPSKPT3V.js.map +0 -1
  88. /package/dist/{chunk-B26KI423.js.map → chunk-FZWAFVAA.js.map} +0 -0
  89. /package/dist/{chunk-GXHLRXDI.js.map → chunk-OTYQPHPL.js.map} +0 -0
  90. /package/dist/{chunk-6REHLN5J.js.map → chunk-QS3RBQPI.js.map} +0 -0
  91. /package/dist/{chunk-GWGO2K6Y.js.map → chunk-RBNA5AZT.js.map} +0 -0
  92. /package/dist/{chunk-CF67I6QY.js.map → chunk-VIDQF3F5.js.map} +0 -0
  93. /package/dist/{chunk-XXNIODOM.js.map → chunk-WJL2NJXN.js.map} +0 -0
  94. /package/dist/{chunk-KX6F6NCG.js.map → chunk-Z7VFTS2J.js.map} +0 -0
  95. /package/dist/{run-campaign-OVEZF24D.js.map → run-campaign-4Y5V5CN3.js.map} +0 -0
package/CHANGELOG.md CHANGED
@@ -4,6 +4,20 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
4
4
 
5
5
  ---
6
6
 
7
+ ## [0.81.0] — 2026-06-05 — eval-campaign scaffold prep primitives
8
+
9
+ ### Added
10
+
11
+ - **`aggregateJudgeVerdicts<D>` (root).** Generic judge-ensemble reducer: fan out N uncorrelated judges, mean each rubric dimension over the SURVIVORS, report the inter-rater disagreement spread, sum cost. Replaces the same reduction hand-rolled in legal (`aggregateEnsemble`), creative (`production-loop/judges.ts`), and tax (`judge-ensemble.ts`). Fail-loud: a failed judge (`perDimension: null`) is recorded in `failedJudges`, never folded into a zero; all-failed throws; a failed judge's cost is still summed. Composite reuses `weightedComposite`.
12
+ - **`createTokenRecallChecker` (root).** The deterministic, no-LLM `CorrectnessChecker` — sibling of `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its content is substantive and recalls ≥ `minRecall` of the requirement title's significant tokens. The default completion gate for apps/tests without an LLM judge.
13
+ - **`ErrorCluster` (root + `/analyst`).** The failure-cluster element type is now a named export, so consumers import it instead of deriving `DatasetOverview['error_clusters'][number]`.
14
+
15
+ ### Fixed
16
+
17
+ - **Lint drift + non-executable pre-commit hook.** `.husky/pre-commit` was tracked `100644`, so the hook silently no-op'd and unformatted code reached `main`; marked executable and reformatted the drift.
18
+
19
+ ---
20
+
7
21
  ## [0.72.3] — 2026-06-01 — workflow trace hardening and driver backtests
8
22
 
9
23
  ### Added
package/README.md CHANGED
@@ -1,199 +1,106 @@
1
1
  # `@tangle-network/agent-eval`
2
2
 
3
- **Decision-grade evals for agents.** One function call returns a decision packet — lift CI, judge calibration, contamination check, failure clusters, cost-quality Pareto, and a ranked action list — with the same shape whether you have a closed improvement loop or just production logs.
3
+ Evaluate and improve AI agents from the runs they already produce.
4
4
 
5
- It is the **substrate at the bottom of the stack**: [`@tangle-network/agent-runtime`](https://www.npmjs.com/package/@tangle-network/agent-runtime) runs agents and captures every run as a trace, then delegates scoring and the ship gate here. The dependency arrow only points up agent-eval never imports the runtime.
5
+ `agent-eval` turns agent outputs, traces, judge scores, and production feedback into a decision packet: did this change help, what failed, what should ship, and what needs more data?
6
6
 
7
7
  [![npm](https://img.shields.io/npm/v/@tangle-network/agent-eval.svg)](https://www.npmjs.com/package/@tangle-network/agent-eval)
8
8
  [![pypi](https://img.shields.io/pypi/v/agent-eval-rpc.svg)](https://pypi.org/project/agent-eval-rpc/)
9
9
  [![tests](https://github.com/tangle-network/agent-eval/actions/workflows/ci.yml/badge.svg)](https://github.com/tangle-network/agent-eval/actions/workflows/ci.yml)
10
10
  [![license: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](./LICENSE)
11
11
 
12
- > TypeScript first-class, Python (`agent-eval-rpc`) speaks the same wire protocol, hosted-tier-friendly, MIT, self-hostable, no SaaS dependency.
12
+ Use it when you need to:
13
13
 
14
- ---
14
+ - compare a candidate agent/prompt/model against a baseline,
15
+ - turn production traces or human feedback into eval results,
16
+ - run a gated self-improvement loop,
17
+ - explain failures by cluster, cost, judge disagreement, and release risk.
15
18
 
16
- ## Table of contents
17
-
18
- - [What you get back](#what-you-get-back-the-decision-packet)
19
- - [Quick start](#quick-start)
20
- - [Closed loop — `selfImprove()`](#closed-loop--selfimprove)
21
- - [Observed runs — `analyzeRuns()`](#observed-runs--analyzeruns)
22
- - [Existing data — intake adapters](#existing-data--intake-adapters)
23
- - [How it compares](#how-it-compares)
24
- - [Customer journeys](#customer-journeys)
25
- - [Subpath entry points](#subpath-entry-points)
26
- - [Concepts + design](#concepts--design)
27
- - [Hosted tier](#hosted-tier)
28
- - [Install + run](#install--run)
29
- - [Stability + versioning](#stability--versioning)
30
- - [License](#license)
19
+ It is a library, not a SaaS requirement. TypeScript is first-class; Python can call the same wire protocol through `agent-eval-rpc`.
31
20
 
32
21
  ---
33
22
 
34
- ## What you get back: the decision packet
35
-
36
- Whether you call `selfImprove()` (closed loop) or `analyzeRuns()` (observed runs), the report has the same shape. Here's a real one, abridged:
23
+ ## Install
37
24
 
38
- ```jsonc
39
- {
40
- "n": 80, // runs analyzed
41
- "composite": { // distributional summary
42
- "mean": 0.62, "p50": 0.65, "p95": 0.88, "stddev": 0.17,
43
- "histogram": [/* 12 bins */]
44
- },
45
- "lift": { // paired bootstrap
46
- "baselineMean": 0.58, "candidateMean": 0.65,
47
- "delta": 0.07,
48
- "ci95": [0.04, 0.10], // 95% CI on the delta
49
- "pValue": 0.0008, // paired-t
50
- "cohensD": 0.41,
51
- "n": 40,
52
- "mde": 0.06, // min detectable effect at 80% power
53
- "requiredN": 38 // n needed to detect observed delta
54
- },
55
- "judges": { // per-judge calibration
56
- "domain-expert": { "n": 80, "meanScore": 0.64 },
57
- "helpfulness-llm": { "n": 80, "meanScore": 0.61 }
58
- },
59
- "interRater": { // multi-rater agreement
60
- "raters": 3, "jointlyRated": 80, "kappa": 0.71,
61
- "disagreementCases": [/* top 20 ranked by spread */]
62
- },
63
- "costQuality": { // cost-vs-quality
64
- "cost": { "mean": 0.024, "p95": 0.041, /* ... */ },
65
- "pareto": { /* ParetoFigureSpec the dashboard renders */ }
66
- },
67
- "failureClusters": { // when an AnalystRegistry is wired
68
- "totalFailures": 11,
69
- "clusters": [
70
- { "name": "off-topic-drift", "share": 0.45, "exemplars": ["run-12", "run-19"] },
71
- { "name": "over-confidence", "share": 0.27, "exemplars": ["run-3"] },
72
- { "name": "format-mismatch", "share": 0.18, "exemplars": ["run-41"] }
73
- ]
74
- },
75
- "contamination": { "leaks": 0, "holdoutAuditPassed": true },
76
- "outcomeCorrelation": { // when downstream metric supplied
77
- "metric": "engagement_rate", "n": 80,
78
- "pearson": 0.72, "spearman": 0.69,
79
- "rewardModel": { "intercept": 0.04, "slope": 1.93, "r2": 0.52 }
80
- },
81
- "release": {
82
- "status": "pass",
83
- "axes": [
84
- { "name": "quality-lift", "status": "pass" },
85
- { "name": "contamination", "status": "pass" },
86
- { "name": "composite-distribution","status": "pass" }
87
- ]
88
- },
89
- "recommendations": [
90
- { "priority": "critical", "kind": "ship",
91
- "title": "Ship — lift 0.070 (95% CI 0.040..0.100)",
92
- "detail": "Holdout lift exceeds threshold 0.02 with 95% bootstrap confidence (n=40, p=0.0008, d=0.41)." },
93
- { "priority": "high", "kind": "investigate",
94
- "title": "Top failure cluster: off-topic-drift (45% of failures)",
95
- "detail": "11 runs failed. Drill into exemplars run-12 / run-19 to identify the pattern." }
96
- ]
97
- }
25
+ ```sh
26
+ pnpm add @tangle-network/agent-eval
98
27
  ```
99
28
 
100
- The `recommendations` array is the human-readable layer; everything above it is the evidence. Read the recs, act on them, the numbers are the proof.
29
+ Python clients can use the RPC package:
30
+
31
+ ```sh
32
+ pip install agent-eval-rpc
33
+ ```
101
34
 
102
35
  ---
103
36
 
104
37
  ## Quick start
105
38
 
106
- ### Closed loop `selfImprove()`
39
+ ### 1. Analyze runs you already have
107
40
 
108
- You have scenarios, a dispatch, judges, and want the loop to propose better prompts + tell you which to ship.
41
+ Start here if you already have production logs, benchmark rows, human ratings, or agent run records.
109
42
 
110
43
  ```ts
111
- import { selfImprove } from '@tangle-network/agent-eval/contract'
44
+ import { analyzeRuns } from '@tangle-network/agent-eval/contract'
112
45
 
113
- const result = await selfImprove({
114
- scenarios, // your scenario corpus
115
- dispatch: async ({ scenario }) => // your agent — anything that returns an artifact
116
- await myAgent.run(scenario),
117
- judges: [myJudge], // any JudgeConfig — LLM, rule, ensemble
118
- baselineSurface: { systemPrompt: currentPrompt },
46
+ const report = await analyzeRuns({
47
+ runs, // RunRecord[]
48
+ baselineRuns,
119
49
  })
120
50
 
121
- result.gateDecision // 'ship' | 'hold' | 'need_more_work' | ...
122
- result.lift // raw delta on holdout
123
- result.insight // the full decision packet above
51
+ console.log(report.recommendations)
52
+ console.log(report.lift)
53
+ console.log(report.failureClusters)
124
54
  ```
125
55
 
126
- ### Observed runs `analyzeRuns()`
56
+ The output includes score distributions, lift confidence intervals, failure modes, cost-quality tradeoffs, judge agreement, contamination checks, and release recommendations when the input supports them.
57
+
58
+ ### 2. Run a gated improvement loop
127
59
 
128
- You don't have a closed loop yet — you have observed runs (production traces, an approve/reject corpus, a CSV gold set). Same report shape, no agent invocation.
60
+ Use this when you have scenarios, a runnable agent, and judges.
129
61
 
130
62
  ```ts
131
- import { analyzeRuns } from '@tangle-network/agent-eval/contract'
63
+ import { selfImprove } from '@tangle-network/agent-eval/contract'
132
64
 
133
- const report = await analyzeRuns({
134
- runs, // RunRecord[]
135
- outcomeSignal: { // optional closes the loop on real outcomes
136
- metric: 'engagement_rate',
137
- valueByRunId: enrichedFromProd,
138
- },
139
- canaryScenarios, // optional — contamination probe
140
- analyst: myAnalystRegistry, // optional — AI-powered failure clustering
65
+ const result = await selfImprove({
66
+ scenarios,
67
+ dispatch: async ({ scenario }) => myAgent.run(scenario),
68
+ judges: [myJudge],
69
+ baselineSurface: { systemPrompt: currentPrompt },
141
70
  })
142
71
 
143
- report.recommendations // ranked actions
144
- report.failureClusters // grouped failure modes
145
- report.outcomeCorrelation // judge↔outcome correlation + linear reward model
72
+ console.log(result.gateDecision)
73
+ console.log(result.winnerSurface)
74
+ console.log(result.insight.recommendations)
146
75
  ```
147
76
 
148
- ### Existing data intake adapters
77
+ `selfImprove()` evaluates candidates on held-out scenarios before recommending a winner.
149
78
 
150
- You have data already. Don't reshape it — pipe it through an adapter.
79
+ ### 3. Adapt existing data
151
80
 
152
81
  ```ts
153
- import {
154
- fromFeedbackTable,
155
- fromOtelSpans,
156
- analyzeRuns,
157
- } from '@tangle-network/agent-eval/contract'
82
+ import { analyzeRuns, fromFeedbackTable, fromOtelSpans } from '@tangle-network/agent-eval/contract'
158
83
 
159
- // Multi-rater approve/reject (Obsidian tags, Sheets, CSV, Postgres).
160
84
  const { runs, raterScores } = fromFeedbackTable({
161
- ratings: parseYourFeedbackTable(), // Array<{ runId, rater, rating }>
85
+ ratings: parseYourFeedbackTable(),
162
86
  })
163
- await analyzeRuns({ runs, raterScores })
164
87
 
165
- // Production OTel traces group by tangle.runId or traceId.
166
- const runs2 = fromOtelSpans({ spans: yourOtelStream })
167
- await analyzeRuns({ runs: runs2 })
168
- ```
88
+ const traceRuns = fromOtelSpans({ spans: yourOtelSpans })
169
89
 
170
- Both intake adapters preserve every signal in the source — multi-rater scores stay rater-keyed so the report can compute inter-rater agreement and surface the disagreement triage list.
90
+ await analyzeRuns({ runs: [...runs, ...traceRuns], raterScores })
91
+ ```
171
92
 
172
93
  ---
173
94
 
174
- ## How it compares
95
+ ## Core concepts
175
96
 
176
- | | LangSmith | Braintrust | Phoenix | **agent-eval** |
177
- |---|:---:|:---:|:---:|:---:|
178
- | Closed-loop self-improvement | human-in-loop | ✱ experiment-driven | — | ✓ autonomous + gated |
179
- | Statistical lift CI (paired bootstrap) | | partial | — | ✓ |
180
- | Judge calibration + bias detection | | — | — | ✓ |
181
- | Inter-rater agreement + disagreement triage | — | — | — | ✓ |
182
- | Contamination / canary check | — | — | — | ✓ |
183
- | AI-driven failure clustering | partial | — | partial | ✓ |
184
- | Cost-quality Pareto | — | — | — | ✓ |
185
- | Multi-language clients (TS + Python) | TS only | TS only | TS + Py | ✓ TS + Py |
186
- | Self-hostable / no-SaaS option | — | — | OSS | ✓ MIT, OSS |
187
- | Substrate vs SaaS shape | SaaS | SaaS | OSS server | **library** |
188
- | Hosted tier (optional) | required | required | optional | optional |
97
+ - **RunRecord**: the durable row for one agent run: model, prompt/config hashes, split, cost, tokens, outcome.
98
+ - **Scenario**: one task or case the agent attempts.
99
+ - **Judge**: a scoring function, rule-based or model-based.
100
+ - **InsightReport**: the decision packet returned by `analyzeRuns()` and embedded in `selfImprove()`.
101
+ - **Gate**: the policy that decides `ship`, `hold`, or `need_more_data`.
189
102
 
190
- Position: agent-eval is the **substrate** (one library, decision-grade output) the others are SaaS *around* the substrate. If you want a closed loop that ships your prompt under statistical confidence, you call agent-eval. If you want a dashboard rendered from your data, you pipe agent-eval into the hosted tier or your own renderer.
191
-
192
- ---
193
-
194
- ## Customer journeys
195
-
196
- Three runnable examples — each is self-contained, each shows the actual output.
103
+ ## Examples
197
104
 
198
105
  | Journey | Example | Who it's for |
199
106
  |---|---|---|
@@ -288,13 +195,7 @@ The substrate runs the loop in your process. Only the eval-run events + (optiona
288
195
 
289
196
  ---
290
197
 
291
- ## Install + run
292
-
293
- ```sh
294
- pnpm add @tangle-network/agent-eval
295
- # or, from Python:
296
- pip install agent-eval-rpc
297
- ```
198
+ ## Development
298
199
 
299
200
  Run an example:
300
201
 
@@ -1,5 +1,5 @@
1
- import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-4mm2msnR.js';
2
- import '../run-record-sItO5ftF.js';
1
+ import { S as Scenario, D as DispatchFn, b as DispatchContext } from '../types-D7lLRYe9.js';
2
+ import '../run-record-De9VarXR.js';
3
3
  import '../errors-Dwqw-T_m.js';
4
4
  import '../schema-m0gsnbt3.js';
5
5
 
@@ -1,5 +1,5 @@
1
- import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-4mm2msnR.js';
2
- import '../run-record-sItO5ftF.js';
1
+ import { S as Scenario, J as JudgeScore, D as DispatchFn, a as JudgeConfig } from '../types-D7lLRYe9.js';
2
+ import '../run-record-De9VarXR.js';
3
3
  import '../errors-Dwqw-T_m.js';
4
4
  import '../schema-m0gsnbt3.js';
5
5
 
@@ -1,10 +1,10 @@
1
1
  import { TraceSpanEvent, HostedClient } from '../hosted/index.js';
2
- import '../types-4mm2msnR.js';
3
- import '../run-record-sItO5ftF.js';
2
+ import '../types-D7lLRYe9.js';
3
+ import '../run-record-De9VarXR.js';
4
4
  import '../errors-Dwqw-T_m.js';
5
5
  import '../schema-m0gsnbt3.js';
6
- import '../insight-report-dlpEzQDi.js';
7
- import '../summary-report-BTaXq1TS.js';
6
+ import '../insight-report-3ADTfClO.js';
7
+ import '../summary-report-Db0dDSWP.js';
8
8
  import '../failure-cluster-CL7IVgkJ.js';
9
9
  import '../store-CKUAgsJz.js';
10
10
  import '../judge-calibration-DilmB3Ml.js';
@@ -1,5 +1,5 @@
1
1
  import { A as AgentEvalError } from './errors-Dwqw-T_m.js';
2
- import { R as RunRecord } from './run-record-sItO5ftF.js';
2
+ import { R as RunRecord } from './run-record-De9VarXR.js';
3
3
  import { TCloud } from '@tangle-network/tcloud';
4
4
 
5
5
  /**
@@ -258,6 +258,18 @@ declare function parseCorrectnessResponse(raw: string): {
258
258
  * fulfil a requirement — the artifact must BE the deliverable.
259
259
  */
260
260
  declare function createLlmCorrectnessChecker(tc: TCloud, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
261
+ /**
262
+ * Deterministic `CorrectnessChecker` — the no-LLM counterpart to
263
+ * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its
264
+ * content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`
265
+ * of the requirement title's significant tokens. No network — the default gate
266
+ * for apps and tests without an LLM judge. Pass to `verifyCompletion` as the
267
+ * checker.
268
+ */
269
+ declare function createTokenRecallChecker(opts?: {
270
+ minRecall?: number;
271
+ minContentLength?: number;
272
+ }): CorrectnessChecker;
261
273
 
262
274
  /**
263
275
  * Produced-state extraction — normalize a run's runtime event stream into the
@@ -361,4 +373,4 @@ interface AgentProfile {
361
373
  */
362
374
  declare function agentProfileHash(profile: AgentProfile): string;
363
375
 
364
- export { type AgentProfile as A, type BackendIntegrityReport as B, type CompletionRequirement as C, type LlmCorrectnessCheckerOpts as L, type ProducedState as P, type RuntimeEventLike as R, type SatisfiedBy as S, type TaskGold as T, type ValidationContext as V, type CompletionVerdict as a, type CorrectnessChecker as b, type Artifact as c, type ArtifactEventLike as d, type ArtifactValidator as e, BackendIntegrityError as f, type ProducedProposal as g, type ProposalEventLike as h, type RequirementCheck as i, type ToolCallEventLike as j, type ValidationIssue as k, type ValidationResult as l, agentProfileHash as m, assertRealBackend as n, byteLengthRange as o, composeValidators as p, containsAll as q, createLlmCorrectnessChecker as r, extractProducedState as s, jsonHasKeys as t, parseCorrectnessResponse as u, regexMatch as v, summarizeBackendIntegrity as w, verifyCompletion as x };
376
+ export { type AgentProfile as A, type BackendIntegrityReport as B, type CompletionRequirement as C, type LlmCorrectnessCheckerOpts as L, type ProducedState as P, type RuntimeEventLike as R, type SatisfiedBy as S, type TaskGold as T, type ValidationContext as V, type CompletionVerdict as a, type CorrectnessChecker as b, type Artifact as c, type ArtifactEventLike as d, type ArtifactValidator as e, BackendIntegrityError as f, type ProducedProposal as g, type ProposalEventLike as h, type RequirementCheck as i, type ToolCallEventLike as j, type ValidationIssue as k, type ValidationResult as l, agentProfileHash as m, assertRealBackend as n, byteLengthRange as o, composeValidators as p, containsAll as q, createLlmCorrectnessChecker as r, createTokenRecallChecker as s, extractProducedState as t, jsonHasKeys as u, parseCorrectnessResponse as v, regexMatch as w, summarizeBackendIntegrity as x, verifyCompletion as y };
@@ -1,21 +1,21 @@
1
1
  import { AxAIService, AxFunction } from '@ax-llm/ax';
2
2
  import { M as MultiLayerVerifier, V as VerifyOptions, S as Severity } from '../multi-layer-verifier-DlWCXuxL.js';
3
3
  import { c as RunCritic, a as RunTrace } from '../run-critic-BAIjX99r.js';
4
- import { S as SemanticConceptJudgeOptions, a as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-qXEUV2w7.js';
5
- export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, b as DefaultAnalystRegistryOptions, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FindingSubject, g as FindingSubjectKind, h as FindingSubjectStringSchema, i as FindingsDiff, j as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, k as KNOWLEDGE_GAP_KIND_SPEC, l as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, m as SKILL_USAGE_ANALYST, n as SkillUsageAnalyst, o as SkillUsageRecord, p as SkillUsageReport, q as SkillUsageScanConfig, r as buildDefaultAnalystRegistry, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as parseFindingSubject, y as renderFindingSubject } from '../semantic-concept-judge-qXEUV2w7.js';
6
- import { A as AnalyzeTracesOptions } from '../analyst-t7zZS3TV.js';
7
- import { T as TraceAnalysisStore } from '../store-GmBE2pZZ.js';
4
+ import { S as SemanticConceptJudgeOptions, a as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-DIEgr_6v.js';
5
+ export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, b as DefaultAnalystRegistryOptions, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FindingSubject, g as FindingSubjectKind, h as FindingSubjectStringSchema, i as FindingsDiff, j as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, k as KNOWLEDGE_GAP_KIND_SPEC, l as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, m as SKILL_USAGE_ANALYST, n as SkillUsageAnalyst, o as SkillUsageRecord, p as SkillUsageReport, q as SkillUsageScanConfig, r as buildDefaultAnalystRegistry, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as parseFindingSubject, y as renderFindingSubject } from '../semantic-concept-judge-DIEgr_6v.js';
6
+ import { A as AnalyzeTracesOptions } from '../analyst-C8HHvfJp.js';
7
+ import { T as TraceAnalysisStore } from '../store-C1YxJDEK.js';
8
8
  import { b as JudgeFn, a as JudgeInput } from '../types-Croy5h7V.js';
9
- import { A as Analyst, h as AnalystSeverity, c as AnalystFinding } from '../types-DRvV0zRo.js';
10
- export { a as AnalystContext, g as AnalystCost, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, k as ChatCallOpts, C as ChatClient, l as ChatRequest, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, p as CreateChatClientOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from '../types-DRvV0zRo.js';
9
+ import { A as Analyst, h as AnalystSeverity, c as AnalystFinding } from '../types-Cu3u_x59.js';
10
+ export { a as AnalystContext, g as AnalystCost, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, k as ChatCallOpts, C as ChatClient, l as ChatRequest, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, p as CreateChatClientOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from '../types-Cu3u_x59.js';
11
11
  import { TCloud } from '@tangle-network/tcloud';
12
- export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, T as TraceAnalystGolden, c as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-DqV2t1Xk.js';
13
- export { A as AnalystHooks, a as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, R as RegistryRunOpts } from '../registry-BK0Zee01.js';
14
- import { L as LlmClientOptions } from '../llm-client-DbjLfz-K.js';
12
+ export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, T as TraceAnalystGolden, c as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-CVecZZG_.js';
13
+ export { A as AnalystHooks, a as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, R as RegistryRunOpts } from '../registry-DrEQ3Luj.js';
14
+ import { L as LlmClientOptions } from '../llm-client-CuUg2Mn3.js';
15
15
  import '../schema-m0gsnbt3.js';
16
16
  import '../store-CKUAgsJz.js';
17
17
  import 'zod';
18
- import '../run-record-sItO5ftF.js';
18
+ import '../run-record-De9VarXR.js';
19
19
  import '../errors-Dwqw-T_m.js';
20
20
  import '../raw-provider-sink-C46HDghv.js';
21
21
 
@@ -14,7 +14,7 @@ import {
14
14
  diffFindings,
15
15
  emitSkillUsageFindings,
16
16
  runSemanticConceptJudge
17
- } from "../chunk-5LVWPNS5.js";
17
+ } from "../chunk-L5G7OUKD.js";
18
18
  import {
19
19
  ANALYST_SEVERITIES,
20
20
  AnalystRegistry,
@@ -41,8 +41,8 @@ import {
41
41
  renderPriorFindings,
42
42
  stripCodeFences,
43
43
  structureFindings
44
- } from "../chunk-CF67I6QY.js";
45
- import "../chunk-IHDHUN2X.js";
44
+ } from "../chunk-VIDQF3F5.js";
45
+ import "../chunk-CVVHBFGN.js";
46
46
  import {
47
47
  analyzeTraces
48
48
  } from "../chunk-VUINJM5M.js";
@@ -1,5 +1,5 @@
1
1
  import { AxAIService } from '@ax-llm/ax';
2
- import { T as TraceAnalysisStore } from './store-GmBE2pZZ.js';
2
+ import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
3
3
 
4
4
  interface AnalyzeTracesInput {
5
5
  /** The user-facing question. Domain framing belongs here, not in the