@tangle-network/agent-eval 0.117.1 → 0.118.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (143) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/dist/analyst/index.d.ts +2772 -21
  3. package/dist/analyst/index.js +7 -6
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/belief-state/index.d.ts +706 -10
  6. package/dist/belief-state/index.js +2 -1
  7. package/dist/belief-state/index.js.map +1 -1
  8. package/dist/benchmarks/index.d.ts +958 -14
  9. package/dist/benchmarks/index.js +12 -10
  10. package/dist/builder-eval/index.d.ts +449 -4
  11. package/dist/builder-eval/index.js +4 -3
  12. package/dist/builder-eval/index.js.map +1 -1
  13. package/dist/campaign/index.d.ts +4275 -73
  14. package/dist/campaign/index.js +12 -10
  15. package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
  16. package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
  17. package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
  18. package/dist/{chunk-YZPO4UHR.js → chunk-FTUMG2U7.js} +124 -149
  19. package/dist/chunk-FTUMG2U7.js.map +1 -0
  20. package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
  21. package/dist/chunk-HKUCJ437.js.map +1 -0
  22. package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
  23. package/dist/chunk-KSDQVPLR.js +286 -0
  24. package/dist/chunk-KSDQVPLR.js.map +1 -0
  25. package/dist/chunk-MA6HLL3S.js +65 -0
  26. package/dist/chunk-MA6HLL3S.js.map +1 -0
  27. package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
  28. package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
  29. package/dist/chunk-PXE2VKMX.js +140 -0
  30. package/dist/chunk-PXE2VKMX.js.map +1 -0
  31. package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
  32. package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
  33. package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
  34. package/dist/chunk-QKEGNI5B.js.map +1 -0
  35. package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
  36. package/dist/chunk-S3UZOQ5Y.js.map +1 -0
  37. package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
  38. package/dist/chunk-SVH2ANFD.js.map +1 -0
  39. package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
  40. package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
  41. package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
  42. package/dist/chunk-WDHBCA3M.js +31 -0
  43. package/dist/chunk-WDHBCA3M.js.map +1 -0
  44. package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
  45. package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
  46. package/dist/chunk-ZET2UAYW.js.map +1 -0
  47. package/dist/contract/index.d.ts +4012 -38
  48. package/dist/contract/index.js +56 -29
  49. package/dist/contract/index.js.map +1 -1
  50. package/dist/control.d.ts +1013 -9
  51. package/dist/control.js +4 -3
  52. package/dist/fuzz.d.ts +194 -4
  53. package/dist/fuzz.js +3 -3
  54. package/dist/hosted/index.d.ts +498 -17
  55. package/dist/index.d.ts +10992 -1288
  56. package/dist/index.js +101 -80
  57. package/dist/index.js.map +1 -1
  58. package/dist/matrix/index.d.ts +139 -4
  59. package/dist/meta-eval/index.d.ts +862 -15
  60. package/dist/meta-eval/index.js +2 -1
  61. package/dist/meta-eval/index.js.map +1 -1
  62. package/dist/multishot/index.d.ts +214 -14
  63. package/dist/openapi.json +1 -1
  64. package/dist/pipelines/index.d.ts +392 -7
  65. package/dist/pipelines/index.js +5 -3
  66. package/dist/pipelines/index.js.map +1 -1
  67. package/dist/reporting.d.ts +1277 -17
  68. package/dist/rl.d.ts +2359 -28
  69. package/dist/rl.js +6 -5
  70. package/dist/rl.js.map +1 -1
  71. package/dist/storyboard/index.d.ts +86 -1
  72. package/dist/trace-attributes.d.ts +16 -0
  73. package/dist/trace-attributes.js +32 -0
  74. package/dist/trace-attributes.js.map +1 -0
  75. package/dist/traces.d.ts +1978 -697
  76. package/dist/traces.js +53 -32
  77. package/dist/wire/index.d.ts +655 -9
  78. package/docs/insight-report.md +44 -0
  79. package/package.json +9 -3
  80. package/dist/adversarial-B7loGVVX.d.ts +0 -19
  81. package/dist/analyst-C8HHvfJp.d.ts +0 -88
  82. package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
  83. package/dist/baseline-DKq3gJpP.d.ts +0 -141
  84. package/dist/calibration-C8MTS7cw.d.ts +0 -101
  85. package/dist/chunk-5UF54T55.js.map +0 -1
  86. package/dist/chunk-E4BUPP7Z.js.map +0 -1
  87. package/dist/chunk-FQNLDL4D.js.map +0 -1
  88. package/dist/chunk-LQUTGLOZ.js.map +0 -1
  89. package/dist/chunk-S2F4J57L.js.map +0 -1
  90. package/dist/chunk-YZPO4UHR.js.map +0 -1
  91. package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
  92. package/dist/control-6vuGfmDH.d.ts +0 -258
  93. package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
  94. package/dist/dataset-NENEzRgk.d.ts +0 -115
  95. package/dist/default-registry-DaK8b3fv.d.ts +0 -155
  96. package/dist/emitter-CjD7vUwv.d.ts +0 -122
  97. package/dist/errors-oeQrLqXC.d.ts +0 -74
  98. package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
  99. package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
  100. package/dist/gepa-eESocoDi.d.ts +0 -642
  101. package/dist/index-PdX4VnPA.d.ts +0 -423
  102. package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
  103. package/dist/integrity-DqlBiLyK.d.ts +0 -81
  104. package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
  105. package/dist/kind-factory-ClZmO25A.d.ts +0 -171
  106. package/dist/llm-client-qoDd18Qz.d.ts +0 -289
  107. package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
  108. package/dist/off-policy-DiwuKKg7.d.ts +0 -132
  109. package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
  110. package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
  111. package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
  112. package/dist/provenance-DpjwyseI.d.ts +0 -541
  113. package/dist/query-CF7PG61p.d.ts +0 -35
  114. package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
  115. package/dist/release-report-C8G2i5Xi.d.ts +0 -236
  116. package/dist/researcher-C8XyxQsu.d.ts +0 -387
  117. package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
  118. package/dist/run-record-BDH49H2E.d.ts +0 -360
  119. package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
  120. package/dist/schema-B3Q3l9Z_.d.ts +0 -201
  121. package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
  122. package/dist/sequential-5iSVfzl2.d.ts +0 -139
  123. package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
  124. package/dist/statistics-KUnG73jH.d.ts +0 -494
  125. package/dist/storage-DrX3v_5B.d.ts +0 -50
  126. package/dist/store-C1YxJDEK.d.ts +0 -248
  127. package/dist/store-DGqD0Pyo.d.ts +0 -116
  128. package/dist/summary-report-C5bKFfm-.d.ts +0 -445
  129. package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
  130. package/dist/types-BSw1rOUB.d.ts +0 -634
  131. package/dist/types-BUxNaJ8c.d.ts +0 -108
  132. package/dist/types-BkfcQnxV.d.ts +0 -313
  133. package/dist/verdict-C9MlYujm.d.ts +0 -35
  134. /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
  135. /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
  136. /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
  137. /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
  138. /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
  139. /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
  140. /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
  141. /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
  142. /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
  143. /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
@@ -1,761 +0,0 @@
1
- import { A as AgentEvalError } from './errors-oeQrLqXC.js';
2
- import { R as RunRecord } from './run-record-BDH49H2E.js';
3
- import { P as PairedBootstrapOptions, M as McNemarResult, R as RiskDifferenceResult, a as PairedBootstrapResult } from './statistics-KUnG73jH.js';
4
- import { z } from 'zod';
5
- import { C as ChatClient } from './policy-edit-wG9uFEFm.js';
6
- import { a as JudgeDimension, S as Scenario, b as JudgeConfig } from './types-BSw1rOUB.js';
7
- import { C as CostLedger, c as CostReceiptInput } from './cost-ledger-DWy3XdJc.js';
8
- import { TCloud } from '@tangle-network/tcloud';
9
- import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
10
- import { D as DefaultVerdict } from './verdict-C9MlYujm.js';
11
-
12
- /**
13
- * Backend-integrity guard: distinguish "agent failed" from "eval ran against
14
- * a stub / unconfigured backend." Without this guard a canonical eval can
15
- * silently report `0/N passed` and look like an agent-quality problem when
16
- * the LLM was never actually called — the failure mode we just hit running
17
- * the 4-vertical parallel eval (legal-sandbox-stub returned hard-coded 33-104
18
- * char strings; gtm/creative defaulted to a cli-bridge that wasn't running).
19
- *
20
- * The shape:
21
- *
22
- * const report = summarizeBackendIntegrity(records)
23
- * assertRealBackend(records) // throws BackendIntegrityError if 100% stub
24
- *
25
- * A record is "stub-mode" if its `tokenUsage.input === 0 && tokenUsage.output === 0`.
26
- * (`costUsd` alone is unreliable — some backends successfully call LLMs but
27
- * don't propagate pricing, producing real tokens with $0 cost.)
28
- *
29
- * Verdicts:
30
- * - `real` — at least one record has nonzero token usage
31
- * - `stub` — every record is stub-mode (eval ran blind)
32
- * - `mixed` — some records real, some stub (partial backend failure;
33
- * often the 429-cascade or auth-half-failed case)
34
- */
35
-
36
- interface BackendIntegrityReport {
37
- /** Total records inspected. */
38
- totalRecords: number;
39
- /** Records with input=0 AND output=0 (a stub fingerprint). */
40
- stubRecords: number;
41
- /** Records with nonzero token usage (real LLM activity). */
42
- realRecords: number;
43
- /** Records where output>0 but costUsd=0 (real LLM, broken cost ledger). */
44
- uncostedRecords: number;
45
- /** Sum of input tokens across all records. */
46
- totalInputTokens: number;
47
- /** Sum of output tokens across all records. */
48
- totalOutputTokens: number;
49
- /** Sum of costUsd across all records. */
50
- totalCostUsd: number;
51
- /** Worst-case integrity verdict. */
52
- verdict: 'real' | 'mixed' | 'stub';
53
- /** Human-readable diagnosis suitable for terminal output. */
54
- diagnosis: string;
55
- }
56
- /**
57
- * Error thrown when an integrity assertion fails. Caller can pattern-match
58
- * by `code === 'AGENT_EVAL_BACKEND_STUB'` to differentiate from other
59
- * errors.
60
- */
61
- declare class BackendIntegrityError extends AgentEvalError {
62
- readonly report: BackendIntegrityReport;
63
- constructor(message: string, report: BackendIntegrityReport);
64
- }
65
- /**
66
- * Inspect a batch of RunRecords and return an integrity report. Pure
67
- * function — no I/O, no logging. The caller decides what to do with the
68
- * verdict (print warning, throw, gate CI, etc.).
69
- */
70
- declare function summarizeBackendIntegrity(records: ReadonlyArray<RunRecord>): BackendIntegrityReport;
71
- /**
72
- * Throw BackendIntegrityError if the verdict is 'stub' — i.e. every record
73
- * shows zero LLM activity. Non-strict callers can pass `{ allowMixed: false }`
74
- * to also reject mixed verdicts (recommended for CI gates).
75
- *
76
- * Real backends pass through silently.
77
- */
78
- declare function assertRealBackend(records: ReadonlyArray<RunRecord>, opts?: {
79
- allowMixed?: boolean;
80
- }): BackendIntegrityReport;
81
-
82
- /**
83
- * Matched-pair arm comparison — "did the treatment arm beat the baseline arm
84
- * on the SAME work items?"
85
- *
86
- * An arm A/B over run records is only trustworthy when it is PAIRED: the same
87
- * task/scenario/seed evaluated under both arms, compared item-by-item, so
88
- * inter-item difficulty variance cancels instead of masquerading as an arm
89
- * effect. This module owns the two error-prone steps every consumer otherwise
90
- * hand-rolls:
91
- *
92
- * 1. Pairing — matching rows across arms by `pairKey` (and by `repKey`
93
- * within multi-rep items), with leftovers REPORTED rather than silently
94
- * dropped (a silently unbalanced pairing biases every paired statistic
95
- * downstream). Pairing never keys on outcome content: matching reps by
96
- * their outcomes deflates discordant-pair counts and makes McNemar
97
- * anti-conservative, so reps pair only by (`pairKey`, `repKey`) identity.
98
- * 2. Composition — feeding the matched pairs to the correct paired
99
- * estimators that already live in `statistics`: `mcnemar` +
100
- * `pairedRiskDifference` for pass/fail, `pairedBootstrap` +
101
- * `wilcoxonSignedRank` for continuous metrics. No statistic is
102
- * re-implemented here.
103
- *
104
- * The row shape is deliberately structural — callers project a `RunRecord`
105
- * (or any record) into `{ pairKey, arm, pass?, metrics? }`. Arm names are
106
- * caller-supplied parameters; the module ships no domain literal.
107
- */
108
-
109
- /** One arm observation of one work item. Structural on purpose: callers
110
- * project their own record type (e.g. a `RunRecord`) into this shape. */
111
- interface PairedArmRow {
112
- /** Matching key — rows sharing a `pairKey` across both arms form pairs
113
- * (typically the task/scenario/seed identity). */
114
- pairKey: string;
115
- /** Rep identity within a `pairKey` (e.g. a seed or rep number). Required on
116
- * every row of a `pairKey` that has more than one rep in either arm; reps
117
- * then pair only on exact (`pairKey`, `repKey`) match, never on outcome
118
- * content. Optional when each arm has at most one rep of the item. */
119
- repKey?: string;
120
- /** Arm label this row was produced under. */
121
- arm: string;
122
- /** Binary outcome; omit when the comparison has no pass/fail notion. */
123
- pass?: boolean;
124
- /** Named numeric measurements (score, cost, latency, …). */
125
- metrics?: Record<string, number>;
126
- }
127
- interface PairArmsOptions {
128
- /** Arm treated as the control side of every pair. */
129
- baselineArm: string;
130
- /** Arm treated as the treatment side of every pair. */
131
- treatmentArm: string;
132
- }
133
- /** One matched (baseline, treatment) observation of the same work item. */
134
- interface MatchedPair {
135
- pairKey: string;
136
- /** 0-based position of this pair within its `pairKey`, ordered by sorted
137
- * `repKey` (always 0 for a single-rep item). The rep identity itself is on
138
- * the rows (`baseline.repKey` / `treatment.repKey`). */
139
- repIndex: number;
140
- baseline: PairedArmRow;
141
- treatment: PairedArmRow;
142
- }
143
- interface PairArmsResult {
144
- /** Matched pairs, ordered by (`pairKey`, `repIndex`). */
145
- pairs: MatchedPair[];
146
- /** Baseline rows left without a treatment counterpart — reported, never
147
- * silently dropped. */
148
- unpairedBaseline: PairedArmRow[];
149
- /** Treatment rows left without a baseline counterpart. */
150
- unpairedTreatment: PairedArmRow[];
151
- }
152
- /**
153
- * Match rows across two arms into (baseline, treatment) pairs by `pairKey`.
154
- *
155
- * A `pairKey` with at most one row per arm pairs directly, no `repKey`
156
- * needed. A `pairKey` with multiple reps in either arm requires `repKey` on
157
- * every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)
158
- * match — pairing is keyed purely on row identity, never on outcome content
159
- * (outcome-keyed matching deflates discordant counts and biases McNemar), and
160
- * is therefore independent of input order. Reps whose `repKey` has no
161
- * counterpart in the other arm, and items present in only one arm, land in
162
- * the unpaired lists — reported, never truncated.
163
- *
164
- * Fail-loud: throws when either named arm has zero rows (an unknown arm
165
- * name would otherwise read as "everything unpaired"), when the two arm
166
- * names are equal, when a multi-rep `pairKey` has a row without `repKey`, or
167
- * when a (`pairKey`, arm) group repeats a `repKey` (the match would be
168
- * ambiguous).
169
- */
170
- declare function pairArms(rows: readonly PairedArmRow[], opts: PairArmsOptions): PairArmsResult;
171
- /** Paired pass/fail comparison over the pairs where BOTH sides carry `pass`. */
172
- interface PairedCorrectness {
173
- /** Discordant pairs where the treatment passed and the baseline failed. */
174
- b10: number;
175
- /** Discordant pairs where the baseline passed and the treatment failed. */
176
- b01: number;
177
- /** Exact McNemar significance over the paired outcomes (`b === b10`, `c === b01`). */
178
- mcnemar: McNemarResult;
179
- /** Paired effect size: p(treatment) − p(baseline) with a paired-variance CI. */
180
- riskDifference: RiskDifferenceResult;
181
- }
182
- /** Paired delta summary for one named metric (delta = treatment − baseline). */
183
- interface PairedMetricDelta {
184
- name: string;
185
- /** Pairs where BOTH sides carry a finite value for this metric. */
186
- n: number;
187
- /** Pairs where at least one side does not carry the metric. */
188
- nMissing: number;
189
- /** Median paired delta; NaN when `n === 0` (no data ≠ measured zero). */
190
- medianDelta: number;
191
- /** Mean paired delta; NaN when `n === 0`. */
192
- meanDelta: number;
193
- /** Bootstrap CI on the paired delta (`pairedBootstrap`); null when
194
- * `n === 0` — a zero-width [0, 0] interval on no data would read as a
195
- * measured tight null. */
196
- bootstrapCi: PairedBootstrapResult | null;
197
- /** Wilcoxon signed-rank test on the paired deltas; null when `n === 0`. */
198
- wilcoxon: {
199
- w: number;
200
- p: number;
201
- } | null;
202
- }
203
- interface ComparePairedArmsOptions extends PairArmsOptions {
204
- /** Metrics to compare. Default: every metric name observed on any matched
205
- * pair, sorted. A name that appears on no pair is still reported (with
206
- * `n = 0`) so a misspelled metric is visible instead of vanishing. */
207
- metricNames?: string[];
208
- /** Passed through to `pairedBootstrap` — set `seed` for reproducible CIs. */
209
- bootstrap?: PairedBootstrapOptions;
210
- }
211
- interface PairedArmsComparison {
212
- nPairs: number;
213
- nUnpairedBaseline: number;
214
- nUnpairedTreatment: number;
215
- /** null when no matched pair carries `pass` on both sides — a pass/fail
216
- * verdict over rows that never measured pass/fail would be fabricated. */
217
- correctness: PairedCorrectness | null;
218
- metricDeltas: PairedMetricDelta[];
219
- }
220
- /**
221
- * Full matched-pair arm comparison: pair via {@link pairArms}, then compose
222
- * the paired estimators from `statistics` over the matched pairs.
223
- *
224
- * Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`
225
- * is that subset's size); each metric uses only the pairs where both sides
226
- * carry a finite value for it, with the remainder counted in `nMissing`.
227
- * Deltas are treatment − baseline throughout.
228
- *
229
- * Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a
230
- * non-finite metric value — silently treating corrupt telemetry as "metric
231
- * absent" would misreport it as missing coverage.
232
- */
233
- declare function comparePairedArms(rows: readonly PairedArmRow[], opts: ComparePairedArmsOptions): PairedArmsComparison;
234
-
235
- /**
236
- * Artifact validators.
237
- *
238
- * Generic "score a produced artifact" primitive. Tax uses it for PDF form
239
- * correctness, research for sourced briefs, browser for task assertions, coding
240
- * for social posts. One interface, many validators; all plug into
241
- * `BenchmarkRunner` the same way.
242
- *
243
- * A validator receives an `Artifact` (file on disk, JSON blob, text, binary)
244
- * plus a `ValidationContext` (scenario id, the turns that produced it) and
245
- * returns a `ValidationResult` with pass/fail + 0..1 score + structured
246
- * issues.
247
- */
248
- interface Artifact {
249
- /** Logical kind — validators type-guard on this */
250
- kind: 'file' | 'json' | 'text' | 'binary' | string;
251
- /** Filesystem-style path, optional */
252
- path?: string;
253
- /** String content for text/json/file kinds */
254
- content?: string;
255
- /** Binary content (if kind === 'binary') */
256
- bytes?: Uint8Array;
257
- /** Caller-supplied metadata (mimeType, sha256, size, etc.) */
258
- metadata?: Record<string, unknown>;
259
- }
260
- interface ValidationContext {
261
- scenarioId: string;
262
- turnIndex?: number;
263
- /** Prior artifacts for multi-artifact scenarios */
264
- priorArtifacts?: Artifact[];
265
- /** Free-form hints the validator uses for domain-specific checks */
266
- hints?: Record<string, unknown>;
267
- }
268
- interface ValidationIssue {
269
- severity: 'error' | 'warning' | 'info';
270
- message: string;
271
- /** Optional path into the artifact (e.g. JSON path or byte offset) */
272
- locus?: string;
273
- }
274
- interface ValidationResult {
275
- pass: boolean;
276
- /** 0–1 normalized score. Validators should be monotonic in pass-ness. */
277
- score: number;
278
- issues: ValidationIssue[];
279
- /** Diagnostic payload for reporters */
280
- evidence?: Record<string, unknown>;
281
- }
282
- interface ArtifactValidator {
283
- /** Stable identifier for the validator; appears in reports. */
284
- name: string;
285
- /** Optional description for human-facing reports. */
286
- description?: string;
287
- /** Called once per artifact; validators are expected to be pure + idempotent. */
288
- validate(artifact: Artifact, context: ValidationContext): Promise<ValidationResult>;
289
- }
290
- /**
291
- * Run every validator on the same artifact; aggregate pass as AND, score as
292
- * (weighted) mean, issues concatenated. Weights default to 1 each.
293
- */
294
- declare function composeValidators(validators: ArtifactValidator[], options?: {
295
- name?: string;
296
- weights?: number[];
297
- }): ArtifactValidator;
298
- /** Pass if the artifact body matches a provided regex. */
299
- declare function regexMatch(name: string, pattern: RegExp): ArtifactValidator;
300
- /** Pass if JSON parses and every required key is present. */
301
- declare function jsonHasKeys(name: string, requiredPaths: string[]): ArtifactValidator;
302
- /** Pass if min ≤ byte length ≤ max. */
303
- declare function byteLengthRange(name: string, min: number, max: number): ArtifactValidator;
304
- /** Pass if the artifact contains every required substring (case-insensitive by default). */
305
- declare function containsAll(name: string, required: string[], options?: {
306
- caseSensitive?: boolean;
307
- }): ArtifactValidator;
308
-
309
- /**
310
- * Completion verifier — the task-completion oracle.
311
- *
312
- * Answers the only eval question that is not a proxy: did the agent actually
313
- * COMPLETE the task — produce every required deliverable, persisted and
314
- * correct — rather than describe what should be done. A fluent transcript
315
- * that never produces the artifact scores zero here.
316
- *
317
- * Per requirement, a two-stage check:
318
- * 1. Structural — a produced item (vault artifact / approved proposal /
319
- * tool call) of the right kind is matched against the requirement and
320
- * carries non-empty content. Deterministic; no LLM.
321
- * 2. Correctness — only if structurally present AND the matched item
322
- * carries content, one targeted check decides whether that item
323
- * actually fulfils the requirement. A hallucinated artifact fails here;
324
- * an absent one already failed stage 1.
325
- *
326
- * `completionRate` is satisfied / MEASURABLE requirements (unmeasured rows —
327
- * checker failures — are excluded from the denominator, never scored as
328
- * zeros). Quality dimensions are meaningless on an incomplete task — callers
329
- * gate on `fullyComplete` / `completionRate` before scoring quality.
330
- */
331
-
332
- /** What kind of produced state can satisfy a requirement structurally. */
333
- type SatisfiedBy = 'artifact' | 'proposal' | 'tool-call' | 'any';
334
- interface CompletionRequirement {
335
- /** Stable id from the task gold (e.g. a persona's `expected_requirements[].req_id`). */
336
- reqId: string;
337
- /** Human-readable description of the required deliverable. */
338
- title: string;
339
- /** Optional kind/category hint, matched against a produced item's kind. */
340
- category?: string;
341
- /** What produced state satisfies this requirement. Defaults to 'any'. */
342
- satisfiedBy?: SatisfiedBy;
343
- }
344
- interface TaskGold {
345
- taskId: string;
346
- requirements: CompletionRequirement[];
347
- }
348
- interface ProducedProposal {
349
- id: string;
350
- title: string;
351
- status: 'pending' | 'approved' | 'rejected';
352
- /** Optional persisted body — when present, enables a correctness check. */
353
- content?: string;
354
- }
355
- /** Everything observable about what a run actually produced. */
356
- interface ProducedState {
357
- /** Persisted vault artifacts. Reuses the shared `Artifact` shape. */
358
- artifacts: Artifact[];
359
- /** Proposals / filings the agent created. */
360
- proposals: ProducedProposal[];
361
- /** Names of tools the agent invoked. */
362
- toolCalls: string[];
363
- }
364
- interface RequirementCheck {
365
- reqId: string;
366
- title: string;
367
- /** A produced item of the right kind matched the requirement, non-empty. */
368
- structurallyPresent: boolean;
369
- /**
370
- * Whether the matched item actually fulfils the requirement. `null` when
371
- * not structurally present, when the matched item carries no content
372
- * to assess, or when the correctness check itself failed (`unmeasured`).
373
- */
374
- correct: boolean | null;
375
- /** structurallyPresent && !unmeasured && correct !== false. */
376
- satisfied: boolean;
377
- /**
378
- * Set when the correctness check itself errored (LLM call failure or an
379
- * unparseable response after retry). The requirement's fulfilment is
380
- * UNKNOWN — `correct` stays null, `satisfied` is false, and
381
- * `completionVerdict` excludes the row from `completionRate`'s
382
- * denominator. Never folded into a zero: a synthetic zero is
383
- * indistinguishable from a real failure (see `JudgeParseError`).
384
- */
385
- unmeasured?: true;
386
- /** Why the correctness check could not be measured (present iff `unmeasured`). */
387
- unmeasuredReason?: string;
388
- /** Human-readable evidence for the verdict. */
389
- evidence: string[];
390
- }
391
- /** Extends the substrate verdict spine: `valid` = `fullyComplete` and
392
- * `score` = `completionRate` — derived in `completionVerdict()`, the one
393
- * place those equalities hold by construction. */
394
- interface CompletionVerdict extends DefaultVerdict {
395
- taskId: string;
396
- requirements: RequirementCheck[];
397
- /** satisfied / MEASURABLE requirements (unmeasured rows leave the denominator). */
398
- completionRate: number;
399
- /** Every measurable requirement satisfied (false when anything is unmeasured). */
400
- fullyComplete: boolean;
401
- /** Requirements whose correctness check errored — reported, never scored as zero. */
402
- unmeasuredCount: number;
403
- }
404
- /**
405
- * Construct a `CompletionVerdict` from the per-requirement checks, deriving
406
- * `completionRate` / `fullyComplete` and the spine fields (`valid` =
407
- * `fullyComplete`, `score` = `completionRate`) in one place. Throws on zero
408
- * requirements — a verdict over nothing is a misconfiguration, mirroring
409
- * `verifyCompletion`'s gold-spec guard.
410
- */
411
- declare function completionVerdict(input: {
412
- taskId: string;
413
- requirements: RequirementCheck[];
414
- }): CompletionVerdict;
415
- /**
416
- * Decides whether a produced item's content actually fulfils a requirement.
417
- * Injected so the structural verifier stays pure and unit-testable; the
418
- * production implementation is `createLlmCorrectnessChecker`.
419
- */
420
- type CorrectnessChecker = (requirement: CompletionRequirement, content: string) => Promise<{
421
- correct: boolean;
422
- reason: string;
423
- }>;
424
- /**
425
- * Verify whether a run completed the task. `checkCorrectness` is injected —
426
- * `createLlmCorrectnessChecker` for production, a deterministic stub in tests.
427
- *
428
- * Throws on a gold spec with no requirements: an eval task that requires
429
- * nothing is a misconfiguration, not a vacuously-complete task.
430
- */
431
- declare function verifyCompletion(gold: TaskGold, state: ProducedState, checkCorrectness: CorrectnessChecker): Promise<CompletionVerdict>;
432
- interface LlmCorrectnessCheckerOpts {
433
- model?: string;
434
- /** Optional ledger for direct use. */
435
- costLedger?: CostLedger;
436
- costPhase?: string;
437
- costTags?: Record<string, string>;
438
- signal?: AbortSignal;
439
- /** Exact maximum provider attempts configured on the supplied TCloud client. */
440
- tcloudMaximumAttempts?: number;
441
- /** Usage/cost retained by a failed provider response; enables a safe retry. */
442
- receiptFromError?: (error: Error, attempt: number) => CostReceiptInput | undefined;
443
- /** Max chars of artifact content sent to the checker. */
444
- maxContentChars?: number;
445
- /**
446
- * Checker LLM calls per requirement before giving up (parse failures and
447
- * call errors both consume attempts). The failure then surfaces as an
448
- * `unmeasured` requirement, never a zero.
449
- */
450
- maxAttempts?: number;
451
- /**
452
- * Forensic capture of every checker request/response/error — without it a
453
- * checker failure is unauditable (the agent-turn raws never contain the
454
- * checker's own calls). Same sink contract as `LlmClient`.
455
- */
456
- rawSink?: RawProviderSink;
457
- }
458
- /**
459
- * Parse the correctness checker's model response. Tolerates a response
460
- * truncated mid-JSON (max_tokens cap) by auto-closing the prefix — the
461
- * verdict boolean usually lands in the first few tokens, so a recovered
462
- * prefix with a boolean `correct` is a real measurement, not a guess.
463
- * Fails loud (JudgeParseError) when no boolean verdict is recoverable.
464
- */
465
- declare function parseCorrectnessResponse(raw: string): {
466
- correct: boolean;
467
- reason: string;
468
- };
469
- /**
470
- * Production `CorrectnessChecker` — one LLM call per matched artifact,
471
- * deterministic (temperature 0), structured JSON out. Judges fulfilment
472
- * only: a plan, a gesture, or a description of what should be done does not
473
- * fulfil a requirement — the artifact must BE the deliverable.
474
- */
475
- declare function createLlmCorrectnessChecker(tc: TCloud, opts?: LlmCorrectnessCheckerOpts): CorrectnessChecker;
476
- /**
477
- * Deterministic `CorrectnessChecker` — the no-LLM counterpart to
478
- * `createLlmCorrectnessChecker`. A produced item fulfils a requirement when its
479
- * content is substantive (≥ `minContentLength` chars) AND recalls ≥ `minRecall`
480
- * of the requirement title's significant tokens. No network.
481
- *
482
- * Polarity-blind: token recall credits a negation that contains the
483
- * requirement's tokens ("I will NOT produce the comparison" recalls every token
484
- * of "produce the comparison"). The structural match stage is ALSO lexical, so
485
- * pairing the two collapses to a single gameable gate. Use this only as an
486
- * opt-in structural pre-filter or for tasks whose requirements have no polarity
487
- * to invert; for produced-state grading the correctness checker MUST be semantic
488
- * (`createLlmCorrectnessChecker`). See the anti-game fixtures in the test suite.
489
- */
490
- declare function createTokenRecallChecker(opts?: {
491
- minRecall?: number;
492
- minContentLength?: number;
493
- }): CorrectnessChecker;
494
-
495
- /**
496
- * `llmJudge` — the single-LLM-call bridge that turns a rubric prompt into a
497
- * canonical campaign `JudgeConfig`.
498
- *
499
- * The `JudgeConfig` contract (src/campaign/types.ts) is deliberately a
500
- * function, not a fixed LLM-prompt shape: real consumers judge with
501
- * ensembles, deterministic checks, or one LLM call. `ensembleJudge`
502
- * (src/judge-panel.ts) covers the multi-model case; `buildAgreementJudge`
503
- * (src/campaign/distillation) covers the pure-comparator case. `llmJudge`
504
- * covers the common single-call case the `JudgeConfig` doc-comment names:
505
- * one model call against `prompt`, parsed into the canonical `JudgeScore`
506
- * (`{ dimensions, composite, notes }`) on the campaign [0,1] scale.
507
- *
508
- * Transport is injected as a `ChatClient` (src/analyst/chat-client.ts) — the
509
- * substrate's transport-agnostic LLM seam — so the judge stays decoupled from
510
- * router-vs-sandbox-vs-cli-bridge and is unit-testable with the `mock`
511
- * transport. The composite is computed by `weightedComposite` (the same
512
- * sum-normalized weighting `ensembleJudge` uses), so a lift is attributable to
513
- * the dimension scores, not to a bespoke reducer.
514
- *
515
- * Fail-loud throughout: an unparseable model response throws `JudgeParseError`;
516
- * a response missing a declared dimension throws; an out-of-range score throws.
517
- * A thrown judge is recorded by the campaign engine as a failed cell, never
518
- * folded into a silent zero.
519
- */
520
-
521
- /** A rubric dimension as a bare key or the full `{ key, description }` shape. A
522
- * bare string uses the key as its own description. */
523
- type LlmJudgeDimension = string | JudgeDimension;
524
- interface LlmJudgeOptions<TArtifact, TScenario extends Scenario = Scenario> {
525
- /** The injected LLM transport. One `chat()` call per `score()`. Required —
526
- * there is no default route, so a misconfigured judge fails at construction,
527
- * never silently against the free-tier router. */
528
- chat: ChatClient;
529
- /** Rubric dimensions the model scores. Each becomes a `[0,1]` field of the
530
- * returned `JudgeScore.dimensions`. Defaults to a single `quality` dimension. */
531
- dimensions?: LlmJudgeDimension[];
532
- /** Model id. Falls back to `chat.defaultModel`; one of the two MUST resolve. */
533
- model?: string;
534
- /** Explicit scoring revision for opaque transport or renderer changes. */
535
- judgeVersion?: string;
536
- temperature?: number;
537
- maxTokens?: number;
538
- /** Composite weights forwarded to `weightedComposite`: a partial map selects
539
- * AND weights exactly the named dimensions. Omit for a uniform mean. */
540
- weights?: Record<string, number>;
541
- /** Scale the model is prompted to score on, normalized into `[0,1]`:
542
- * - `'unit'` (default): the model returns `[0,1]` directly.
543
- * - `'ten'`: the model returns `[0,10]`; divided by 10 here.
544
- * The prompt is annotated with the expected range either way. */
545
- scale?: 'unit' | 'ten';
546
- /** Run this judge only on matching scenarios (mirrors `JudgeConfig.appliesTo`). */
547
- appliesTo?: (scenario: TScenario) => boolean;
548
- /** Render the artifact + scenario into the user message. Default:
549
- * pretty-printed JSON of `{ scenario, artifact }`. */
550
- renderUser?: (input: {
551
- artifact: TArtifact;
552
- scenario: TScenario;
553
- }) => string;
554
- /** Strict runtime contract; its JSON Schema is sent to the provider. */
555
- costLedger?: CostLedger;
556
- responseSchema?: {
557
- name: string;
558
- schema: z.ZodObject;
559
- };
560
- }
561
- /**
562
- * Build a campaign-shaped `JudgeConfig` whose `score()` makes ONE LLM call
563
- * against `prompt` and reduces the model's per-dimension scores to a canonical
564
- * `JudgeScore` in `[0,1]`.
565
- *
566
- * The model is instructed to return JSON `{ "dimensions": { <key>: <number>, … },
567
- * "notes": "…" }`; the helper strips fenced JSON, validates every declared
568
- * dimension is present and in range, normalizes by `scale`, and composites via
569
- * `weightedComposite`.
570
- */
571
- declare function llmJudge<TArtifact = unknown, TScenario extends Scenario = Scenario>(name: string, prompt: string, opts: LlmJudgeOptions<TArtifact, TScenario>): JudgeConfig<TArtifact, TScenario>;
572
-
573
- /**
574
- * Produced-state extraction — normalize a run's runtime event stream into the
575
- * typed `ProducedState` the completion oracle consumes.
576
- *
577
- * `ProducedState` answers "what did the agent actually produce" — vault
578
- * artifacts, proposals, tool calls. The runtime emits these as a stream of
579
- * events; this module is the single normalization point from that stream to
580
- * the shape `verifyCompletion` expects.
581
- *
582
- * Input is structurally typed (`RuntimeEventLike`) so this module does not
583
- * depend on agent-runtime — agent-runtime's `RuntimeStreamEvent` satisfies it
584
- * structurally. The `content` on `ArtifactEventLike` and the whole
585
- * `proposal_created` variant are the runtime-side enrichments this contract
586
- * requires; the runtime emits them, this module consumes them.
587
- */
588
-
589
- /** A tool the agent invoked. */
590
- interface ToolCallEventLike {
591
- type: 'tool_call';
592
- toolName: string;
593
- }
594
- /**
595
- * An artifact the agent produced. `content` is the enriched field — the
596
- * runtime's base `artifact` event carries only metadata; the completion
597
- * oracle needs the body to verify the deliverable, so the runtime emits it.
598
- */
599
- interface ArtifactEventLike {
600
- type: 'artifact';
601
- artifactId: string;
602
- name?: string;
603
- mimeType?: string;
604
- uri?: string;
605
- content?: string;
606
- }
607
- /** A proposal / filing the agent created. */
608
- interface ProposalEventLike {
609
- type: 'proposal_created';
610
- proposalId: string;
611
- title: string;
612
- status?: 'pending' | 'approved' | 'rejected';
613
- content?: string;
614
- }
615
- /**
616
- * The subset of runtime stream events `extractProducedState` consumes.
617
- * agent-runtime's full `RuntimeStreamEvent` union satisfies this structurally;
618
- * the `{ type: string }` catch-all keeps the input permissive so callers can
619
- * pass the whole unfiltered telemetry stream — unrecognized events are skipped.
620
- */
621
- type RuntimeEventLike = ToolCallEventLike | ArtifactEventLike | ProposalEventLike | {
622
- type: string;
623
- };
624
- /**
625
- * Normalize a run's runtime event stream into `ProducedState`.
626
- *
627
- * Pure and total — unrecognized event types are skipped. `toolCalls` is
628
- * deduplicated by name in first-seen order (completion cares about a tool's
629
- * presence, not its call count). An artifact with neither a name nor a uri
630
- * still yields an entry keyed by its `artifactId` so it is never silently
631
- * dropped; an artifact with no `content` yields empty content, which the
632
- * completion oracle's structural check then rejects on its own.
633
- */
634
- declare function extractProducedState(events: readonly RuntimeEventLike[]): ProducedState;
635
-
636
- /**
637
- * Pre-registered hypotheses — declare what you're testing BEFORE the
638
- * run, check it AFTER. Prevents p-hacking, optional stopping, and the
639
- * "we ran until it looked good" failure mode.
640
- *
641
- * Manifest is a plain JSON-friendly object. Sign it with a content hash
642
- * + timestamp; the registered record becomes immutable. Post-run,
643
- * evaluate the manifest against observed results — the library refuses
644
- * to let you re-interpret a different metric as the declared one.
645
- */
646
- interface HypothesisManifest {
647
- id: string;
648
- /** Human prose — goes into the audit trail. */
649
- hypothesis: string;
650
- /** Metric the hypothesis claims to move. */
651
- metric: string;
652
- /** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */
653
- direction: 'increase' | 'decrease';
654
- /** Minimum effect size to count (same units as the metric). */
655
- minEffect: number;
656
- /** Alpha threshold. */
657
- alpha: number;
658
- /** Target statistical power at which sample size was pre-computed. */
659
- power: number;
660
- /** Declared N per arm before running. */
661
- preRegisteredN: number;
662
- /** ISO8601 timestamp the manifest was registered. */
663
- registeredAt: string;
664
- /** Optional identifiers to tie into the trace corpus. */
665
- baselineLabel?: string;
666
- candidateLabel?: string;
667
- }
668
- /**
669
- * Identifier for the hashing scheme used to produce `contentHash`.
670
- *
671
- * `'sha256-content'` — sha256 hex over the canonicalized manifest with
672
- * the `contentHash` and `algo` fields stripped. Held as a string union
673
- * so future schemes can be added without breaking parsers; SignedManifest
674
- * values without `algo` deserialize cleanly because the field is optional.
675
- */
676
- type SignedManifestAlgo = 'sha256-content';
677
- interface SignedManifest extends HypothesisManifest {
678
- /** sha256 hex of canonicalized manifest (everything except contentHash and algo). */
679
- contentHash: string;
680
- /**
681
- * Algorithm string describing how `contentHash` was produced.
682
- *
683
- * Optional on the type so serialized manifests without it still parse,
684
- * but ALWAYS populated by {@link signManifest}. Consumers that want to
685
- * enforce a known algorithm should reject manifests where this field
686
- * is missing or unrecognized.
687
- */
688
- algo?: SignedManifestAlgo;
689
- }
690
- interface HypothesisResult {
691
- manifest: SignedManifest;
692
- observedN: number;
693
- observedEffect: number;
694
- observedPValue: number;
695
- /** True iff the observed effect hits the pre-declared direction with
696
- * magnitude ≥ minEffect AND p < alpha. */
697
- confirmed: boolean;
698
- /** Enumerated reasons the hypothesis was rejected (each a machine-tag). */
699
- rejectionReasons: Array<'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'>;
700
- notes?: string;
701
- }
702
- /**
703
- * Deterministic JSON canonicalization — sort object keys recursively.
704
- *
705
- * Two semantically-equal objects produce byte-identical canonicalized output;
706
- * this is what makes a content-hash stable across encoders, key insertion
707
- * orders, and runtime versions. Exported for any consumer that needs the same
708
- * canonicalization guarantee outside the manifest-signing path (e.g., signing
709
- * an artifact bundle, hashing a dataset version, etc.).
710
- */
711
- declare function canonicalize(v: unknown): unknown;
712
- /**
713
- * SHA-256 hex (full 64 chars) over the canonicalized JSON encoding of `obj`.
714
- *
715
- * The same primitive `signManifest` and `verifyManifest` are built on, exposed
716
- * directly so consumers signing arbitrary structured content (artifact bundles,
717
- * production packets, dataset manifests, etc.) don't have to re-derive
718
- * canonicalize+sha256 from scratch.
719
- *
720
- * Stable across:
721
- * - object key insertion order (canonicalization sorts keys recursively)
722
- * - encoder choice (UTF-8 via TextEncoder, fixed)
723
- * - runtime (uses the Web Crypto subtle digest, present in Node ≥18 and browsers)
724
- *
725
- * Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,
726
- * which takes a string input and returns a truncated 12-char prompt id.
727
- * Use `hashJson` when you mean "canonicalize then hash."
728
- *
729
- * @example
730
- * const hash = await hashJson({ id: '1', kind: 'spec' })
731
- * // 'a3f1...' (64 hex chars)
732
- */
733
- declare function hashJson<T>(obj: T): Promise<string>;
734
- /**
735
- * Sign a manifest with a SHA-256 content hash.
736
- *
737
- * The hash covers the canonicalized manifest with the `contentHash`
738
- * and `algo` fields stripped; this lets verifiers re-sign the rest and
739
- * compare. Returned manifest always carries `algo: 'sha256-content'`
740
- * so downstream consumers can identify the scheme; manifests without
741
- * `algo` still verify because it is stripped before hashing on both sides.
742
- */
743
- declare function signManifest(m: HypothesisManifest): Promise<SignedManifest>;
744
- /**
745
- * Verify that a signed manifest has not been tampered with.
746
- *
747
- * Strips `contentHash` and `algo` before re-signing so manifests without
748
- * `algo` verify identically to ones that carry it.
749
- */
750
- declare function verifyManifest(m: SignedManifest): Promise<boolean>;
751
- /**
752
- * Evaluate a pre-registered hypothesis against observed results.
753
- * Mechanical — no re-interpretation permitted.
754
- */
755
- declare function evaluateHypothesis(manifest: SignedManifest, observed: {
756
- n: number;
757
- effect: number;
758
- pValue: number;
759
- }): Promise<HypothesisResult>;
760
-
761
- export { verifyCompletion as $, type Artifact as A, type BackendIntegrityReport as B, type CompletionRequirement as C, canonicalize as D, comparePairedArms as E, completionVerdict as F, composeValidators as G, type HypothesisManifest as H, containsAll as I, createLlmCorrectnessChecker as J, createTokenRecallChecker as K, type LlmJudgeDimension as L, type MatchedPair as M, evaluateHypothesis as N, extractProducedState as O, type PairedArmsComparison as P, hashJson as Q, type RuntimeEventLike as R, type SignedManifest as S, type TaskGold as T, jsonHasKeys as U, type ValidationContext as V, pairArms as W, parseCorrectnessResponse as X, regexMatch as Y, signManifest as Z, summarizeBackendIntegrity as _, type CompletionVerdict as a, verifyManifest as a0, type ProducedState as b, type CorrectnessChecker as c, type LlmJudgeOptions as d, type ArtifactEventLike as e, type ArtifactValidator as f, BackendIntegrityError as g, type ComparePairedArmsOptions as h, type HypothesisResult as i, type LlmCorrectnessCheckerOpts as j, type PairArmsOptions as k, llmJudge as l, type PairArmsResult as m, type PairedArmRow as n, type PairedCorrectness as o, type PairedMetricDelta as p, type ProducedProposal as q, type ProposalEventLike as r, type RequirementCheck as s, type SatisfiedBy as t, type SignedManifestAlgo as u, type ToolCallEventLike as v, type ValidationIssue as w, type ValidationResult as x, assertRealBackend as y, byteLengthRange as z };