@tangle-network/agent-eval 0.120.0 → 0.120.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/CHANGELOG.md +12 -0
  2. package/package.json +1 -1
  3. package/dist/analyst/index.d.ts +0 -3111
  4. package/dist/analyst/index.js +0 -403
  5. package/dist/analyst/index.js.map +0 -1
  6. package/dist/authenticity/index.d.ts +0 -161
  7. package/dist/authenticity/index.js +0 -215
  8. package/dist/authenticity/index.js.map +0 -1
  9. package/dist/belief-state/index.d.ts +0 -1301
  10. package/dist/belief-state/index.js +0 -2152
  11. package/dist/belief-state/index.js.map +0 -1
  12. package/dist/benchmarks/index.d.ts +0 -974
  13. package/dist/benchmarks/index.js +0 -60
  14. package/dist/benchmarks/index.js.map +0 -1
  15. package/dist/builder-eval/index.d.ts +0 -695
  16. package/dist/builder-eval/index.js +0 -366
  17. package/dist/builder-eval/index.js.map +0 -1
  18. package/dist/campaign/index.d.ts +0 -7454
  19. package/dist/campaign/index.js +0 -272
  20. package/dist/campaign/index.js.map +0 -1
  21. package/dist/chunk-32BZXMSO.js +0 -3878
  22. package/dist/chunk-32BZXMSO.js.map +0 -1
  23. package/dist/chunk-3A246TSA.js +0 -998
  24. package/dist/chunk-3A246TSA.js.map +0 -1
  25. package/dist/chunk-3RF76KTD.js +0 -84
  26. package/dist/chunk-3RF76KTD.js.map +0 -1
  27. package/dist/chunk-3YYRZDON.js +0 -45
  28. package/dist/chunk-3YYRZDON.js.map +0 -1
  29. package/dist/chunk-4I2E3LLO.js +0 -1030
  30. package/dist/chunk-4I2E3LLO.js.map +0 -1
  31. package/dist/chunk-ARU2PZFM.js +0 -312
  32. package/dist/chunk-ARU2PZFM.js.map +0 -1
  33. package/dist/chunk-BOD4O7OF.js +0 -40
  34. package/dist/chunk-BOD4O7OF.js.map +0 -1
  35. package/dist/chunk-DPZAEKA6.js +0 -880
  36. package/dist/chunk-DPZAEKA6.js.map +0 -1
  37. package/dist/chunk-DTJ6QUQB.js +0 -131
  38. package/dist/chunk-DTJ6QUQB.js.map +0 -1
  39. package/dist/chunk-GGE4NNQT.js +0 -65
  40. package/dist/chunk-GGE4NNQT.js.map +0 -1
  41. package/dist/chunk-H5UD2323.js +0 -286
  42. package/dist/chunk-H5UD2323.js.map +0 -1
  43. package/dist/chunk-HHWE3POT.js +0 -94
  44. package/dist/chunk-HHWE3POT.js.map +0 -1
  45. package/dist/chunk-HKUCJ437.js +0 -787
  46. package/dist/chunk-HKUCJ437.js.map +0 -1
  47. package/dist/chunk-JHCHEVET.js +0 -274
  48. package/dist/chunk-JHCHEVET.js.map +0 -1
  49. package/dist/chunk-JHOJHHU7.js +0 -867
  50. package/dist/chunk-JHOJHHU7.js.map +0 -1
  51. package/dist/chunk-JM2SKQMS.js +0 -750
  52. package/dist/chunk-JM2SKQMS.js.map +0 -1
  53. package/dist/chunk-JN2FCO5W.js +0 -7958
  54. package/dist/chunk-JN2FCO5W.js.map +0 -1
  55. package/dist/chunk-K4DBDHLK.js +0 -158
  56. package/dist/chunk-K4DBDHLK.js.map +0 -1
  57. package/dist/chunk-K6N6XJJX.js +0 -306
  58. package/dist/chunk-K6N6XJJX.js.map +0 -1
  59. package/dist/chunk-MA6HLL3S.js +0 -65
  60. package/dist/chunk-MA6HLL3S.js.map +0 -1
  61. package/dist/chunk-MAZ26DC7.js +0 -99
  62. package/dist/chunk-MAZ26DC7.js.map +0 -1
  63. package/dist/chunk-MOXWMGPC.js +0 -577
  64. package/dist/chunk-MOXWMGPC.js.map +0 -1
  65. package/dist/chunk-NJC7U437.js +0 -626
  66. package/dist/chunk-NJC7U437.js.map +0 -1
  67. package/dist/chunk-NPCTHQIO.js +0 -91
  68. package/dist/chunk-NPCTHQIO.js.map +0 -1
  69. package/dist/chunk-ONWEPEDO.js +0 -57
  70. package/dist/chunk-ONWEPEDO.js.map +0 -1
  71. package/dist/chunk-OYZAPX5G.js +0 -1526
  72. package/dist/chunk-OYZAPX5G.js.map +0 -1
  73. package/dist/chunk-PC4UYEBM.js +0 -166
  74. package/dist/chunk-PC4UYEBM.js.map +0 -1
  75. package/dist/chunk-PICTDURQ.js +0 -766
  76. package/dist/chunk-PICTDURQ.js.map +0 -1
  77. package/dist/chunk-PJQFMIOX.js +0 -1182
  78. package/dist/chunk-PJQFMIOX.js.map +0 -1
  79. package/dist/chunk-PXD6ZFNY.js +0 -1107
  80. package/dist/chunk-PXD6ZFNY.js.map +0 -1
  81. package/dist/chunk-PXE2VKMX.js +0 -140
  82. package/dist/chunk-PXE2VKMX.js.map +0 -1
  83. package/dist/chunk-PZ5AY32C.js +0 -10
  84. package/dist/chunk-PZ5AY32C.js.map +0 -1
  85. package/dist/chunk-QBRSJK47.js +0 -622
  86. package/dist/chunk-QBRSJK47.js.map +0 -1
  87. package/dist/chunk-QWMPPZ3X.js +0 -550
  88. package/dist/chunk-QWMPPZ3X.js.map +0 -1
  89. package/dist/chunk-S3UZOQ5Y.js +0 -328
  90. package/dist/chunk-S3UZOQ5Y.js.map +0 -1
  91. package/dist/chunk-S5TT5R3L.js +0 -2668
  92. package/dist/chunk-S5TT5R3L.js.map +0 -1
  93. package/dist/chunk-T4SQEITX.js +0 -95
  94. package/dist/chunk-T4SQEITX.js.map +0 -1
  95. package/dist/chunk-TT4KNT67.js +0 -124
  96. package/dist/chunk-TT4KNT67.js.map +0 -1
  97. package/dist/chunk-U5CHZ5M3.js +0 -357
  98. package/dist/chunk-U5CHZ5M3.js.map +0 -1
  99. package/dist/chunk-ULOKLHIQ.js +0 -1937
  100. package/dist/chunk-ULOKLHIQ.js.map +0 -1
  101. package/dist/chunk-VI2UW6B6.js +0 -162
  102. package/dist/chunk-VI2UW6B6.js.map +0 -1
  103. package/dist/chunk-VQMK5FMP.js +0 -247
  104. package/dist/chunk-VQMK5FMP.js.map +0 -1
  105. package/dist/chunk-VSMTAMNK.js +0 -53
  106. package/dist/chunk-VSMTAMNK.js.map +0 -1
  107. package/dist/chunk-VZSRQ272.js +0 -149
  108. package/dist/chunk-VZSRQ272.js.map +0 -1
  109. package/dist/chunk-WW2A73HW.js +0 -159
  110. package/dist/chunk-WW2A73HW.js.map +0 -1
  111. package/dist/chunk-X4UCIOTZ.js +0 -136
  112. package/dist/chunk-X4UCIOTZ.js.map +0 -1
  113. package/dist/chunk-XDIRG3TO.js +0 -1266
  114. package/dist/chunk-XDIRG3TO.js.map +0 -1
  115. package/dist/chunk-XJYR7XFV.js +0 -317
  116. package/dist/chunk-XJYR7XFV.js.map +0 -1
  117. package/dist/chunk-ZET2UAYW.js +0 -89
  118. package/dist/chunk-ZET2UAYW.js.map +0 -1
  119. package/dist/chunk-ZZUXHH3R.js +0 -99
  120. package/dist/chunk-ZZUXHH3R.js.map +0 -1
  121. package/dist/cli.d.ts +0 -1
  122. package/dist/cli.js +0 -112
  123. package/dist/cli.js.map +0 -1
  124. package/dist/contract/index.d.ts +0 -4969
  125. package/dist/contract/index.js +0 -1653
  126. package/dist/contract/index.js.map +0 -1
  127. package/dist/control.d.ts +0 -1013
  128. package/dist/control.js +0 -34
  129. package/dist/control.js.map +0 -1
  130. package/dist/fuzz.d.ts +0 -759
  131. package/dist/fuzz.js +0 -714
  132. package/dist/fuzz.js.map +0 -1
  133. package/dist/hosted/index.d.ts +0 -730
  134. package/dist/hosted/index.js +0 -14
  135. package/dist/hosted/index.js.map +0 -1
  136. package/dist/index.d.ts +0 -16780
  137. package/dist/index.js +0 -12168
  138. package/dist/index.js.map +0 -1
  139. package/dist/matrix/index.d.ts +0 -155
  140. package/dist/matrix/index.js +0 -8
  141. package/dist/matrix/index.js.map +0 -1
  142. package/dist/meta-eval/index.d.ts +0 -1030
  143. package/dist/meta-eval/index.js +0 -417
  144. package/dist/meta-eval/index.js.map +0 -1
  145. package/dist/multishot/index.d.ts +0 -579
  146. package/dist/multishot/index.js +0 -589
  147. package/dist/multishot/index.js.map +0 -1
  148. package/dist/openapi.json +0 -992
  149. package/dist/pipelines/index.d.ts +0 -567
  150. package/dist/pipelines/index.js +0 -515
  151. package/dist/pipelines/index.js.map +0 -1
  152. package/dist/reporting.d.ts +0 -1277
  153. package/dist/reporting.js +0 -48
  154. package/dist/reporting.js.map +0 -1
  155. package/dist/rl.d.ts +0 -4092
  156. package/dist/rl.js +0 -1724
  157. package/dist/rl.js.map +0 -1
  158. package/dist/run-campaign-HNFPJET4.js +0 -14
  159. package/dist/run-campaign-HNFPJET4.js.map +0 -1
  160. package/dist/storyboard/index.d.ts +0 -279
  161. package/dist/storyboard/index.js +0 -767
  162. package/dist/storyboard/index.js.map +0 -1
  163. package/dist/trace-attributes.d.ts +0 -52
  164. package/dist/trace-attributes.js +0 -62
  165. package/dist/trace-attributes.js.map +0 -1
  166. package/dist/traces.d.ts +0 -2343
  167. package/dist/traces.js +0 -249
  168. package/dist/traces.js.map +0 -1
  169. package/dist/wire/index.d.ts +0 -1252
  170. package/dist/wire/index.js +0 -81
  171. package/dist/wire/index.js.map +0 -1
@@ -1,730 +0,0 @@
1
- /**
2
- * Pass A substrate types — `runCampaign` is the one primitive every
3
- * eval flow composes from. Three contracts in this file:
4
- *
5
- * - `Scenario` input set
6
- * - `DispatchFn` how to run one scenario → artifact
7
- * - `CampaignResult` defined output schema (the contract downstream tools depend on)
8
- *
9
- * Three more lifted from earlier substrate work (re-exported):
10
- *
11
- * - `JudgeConfig` pluggable dimensional scorer (0.38)
12
- * - `Mutator` optimization-loop surface mutator
13
- * - `Gate` promotion gate (`HeldOutGate` and friends adapt to this)
14
- *
15
- * No new architecture vs 0.38 — Pass A formalizes the shapes so consumers
16
- * can build dashboards / CI gates / regression diffs against a stable schema.
17
- */
18
-
19
- /** A tier-4 code surface — a finalized candidate change to the agent's
20
- * IMPLEMENTATION, not its prompt. Produced by autoresearch (reads codebase +
21
- * trace findings → opens a worktree). `worktreeRef` locates the candidate;
22
- * the exact commits, tree, and binary-patch digest identify it. See the
23
- * improvement-tier table in `docs/design/loop-taxonomy.md`. */
24
- interface CodeSurface {
25
- readonly kind: 'code';
26
- /** Worktree path or git ref holding the candidate code change. This is a
27
- * mutable locator and is deliberately excluded from content hashes. */
28
- readonly worktreeRef: string;
29
- /** Human-readable ref the worktree was forked from. Not identity-bearing. */
30
- readonly baseRef: string;
31
- /** Exact commit the candidate was forked from. */
32
- readonly baseCommit: string;
33
- /** Exact tree object for `baseCommit`. */
34
- readonly baseTree: string;
35
- /** Exact finalized candidate commit. */
36
- readonly candidateCommit: string;
37
- /** Exact tree object for `candidateCommit`. */
38
- readonly candidateTree: string;
39
- /** Identity of the exact patch artifact. The deployable candidate bundle
40
- * carries the same descriptor plus its base64-encoded content. */
41
- readonly patch: {
42
- readonly format: 'git-diff-binary';
43
- readonly sha256: `sha256:${string}`;
44
- readonly byteLength: number;
45
- };
46
- /** Human summary of what changed — rendered into the auto-PR body. */
47
- readonly summary?: string;
48
- }
49
- /** The mutable surface a proposer changes. Tiers (see
50
- * `docs/design/loop-taxonomy.md`):
51
- * - `string` — tiers 1-2: system-prompt addendum / serialized tool
52
- * config. Cheap, reversible, text-diffable.
53
- * - `CodeSurface` — tier 4: an implementation change behind a worktree ref.
54
- * Tier 3 (knowledge) is owned by agent-knowledge and rides its own adapter,
55
- * not this type. */
56
- type MutableSurface = string | CodeSurface;
57
- /** Five-valued verdict taxonomy (MOSS-paper alignment). */
58
- type GateDecision = 'ship' | 'hold' | 'need_more_work' | 'model_ceiling' | 'arch_ceiling';
59
-
60
- /**
61
- * Reporting helpers — production summaries and paper-quality figures — sit alongside `reporter.ts` rather
62
- * than replacing it.
63
- *
64
- * Three artefacts:
65
- *
66
- * - `summaryTable` Markdown table of per-candidate means,
67
- * 95% bootstrap CIs, BH-adjusted Wilcoxon
68
- * p-values, and Cohen's d versus a
69
- * comparator candidate.
70
- * - `paretoChart` Abstract spec for a cost vs quality
71
- * scatter, with gate decisions overlaid.
72
- * Returns numbers + labels — caller
73
- * chooses the plotting library.
74
- * - `gainHistogram`
75
- * Per-item paired holdout deltas as a
76
- * histogram spec (bins + counts + median +
77
- * CI). Same "data, not images" contract.
78
- *
79
- * The figure types are PlotSpecs — JSON-friendly, library-agnostic.
80
- * They aren't React components and they aren't PNGs; they are
81
- * what you'd hand to vega-lite, plotly, matplotlib, or your own
82
- * Canvas renderer to draw the actual figure.
83
- */
84
-
85
- interface ParetoPoint {
86
- candidateId: string;
87
- /** Mean USD cost per run on the chosen split. */
88
- cost: number;
89
- /** Mean score on the chosen split. */
90
- quality: number;
91
- /** Number of runs that informed this point. */
92
- n: number;
93
- /** Whether this candidate is on the Pareto frontier — high
94
- * quality, low cost, no dominator. */
95
- onFrontier: boolean;
96
- /** Optional gate verdict for this candidate, if a `GateDecision`
97
- * for it was passed in. */
98
- gate?: 'promote' | 'reject_few_runs' | 'reject_negative_delta' | 'reject_overfit_gap' | null;
99
- }
100
- interface ParetoFigureSpec {
101
- kind: 'pareto-cost-quality';
102
- split: 'search' | 'holdout';
103
- points: ParetoPoint[];
104
- axes: {
105
- x: 'costUsd';
106
- y: 'score';
107
- };
108
- }
109
- interface GainDistributionBin {
110
- /** Inclusive lower edge. */
111
- lo: number;
112
- /** Exclusive upper edge (or inclusive if it's the last bin). */
113
- hi: number;
114
- /** Number of pairs whose delta lands in this bin. */
115
- count: number;
116
- }
117
-
118
- interface ContinuousAgreement {
119
- /** Cohen's κ_w with quadratic weights, computed on raw [0,1] scores. */
120
- weightedKappa: number;
121
- /** ICC(2,1): two-way random effects, absolute agreement, single rater. */
122
- icc: number;
123
- /** Pearson product-moment correlation (averaged over rater pairs if N>2). */
124
- pearson: number;
125
- /** Spearman rank correlation (averaged over rater pairs if N>2). */
126
- spearman: number;
127
- /** 95% bootstrap percentile CIs over items. */
128
- ci: {
129
- icc: [number, number];
130
- weightedKappa: [number, number];
131
- };
132
- /** Number of complete items (no NaN across raters). */
133
- n: number;
134
- /** Number of raters. */
135
- raters: number;
136
- }
137
-
138
- /**
139
- * # InsightReport — the rigorous decision packet for any set of agent runs.
140
- *
141
- * Returned by `analyzeRuns()` and embedded in `SelfImproveResult.insight` +
142
- * the hosted-tier `EvalRunEvent.insightReport`. One shape across two surfaces:
143
- *
144
- * - **Customer who has a closed loop** (`selfImprove`): the report ships
145
- * with the loop output. Their dashboard renders ship/hold + lift CI +
146
- * calibration + cluster + Pareto in one packet.
147
- * - **Customer who has observed runs but no loop** (`analyzeRuns` directly):
148
- * same packet from a `RunRecord[]` they already have — production traces,
149
- * approve/reject corpus, CSV gold set.
150
- *
151
- * Every field is optional except the distributional summary — fields are
152
- * populated when the input data supports them:
153
- *
154
- * - `lift` requires both baseline and candidate splits to be present.
155
- * - `interRater` requires multi-rater feedback (≥2 raters per run).
156
- * - `judges` populates per-judge stats only when the run records carry
157
- * `outcome.judgeScores`.
158
- * - `failureClusters` requires the optional `analystRegistry` to be wired.
159
- * - `contamination` requires canary scenarios to be passed in.
160
- * - `outcomeCorrelation` requires a downstream outcome signal.
161
- * - `sequential` requires the run set to be ordered (treats them as a
162
- * stream and emits an anytime-valid interim decision).
163
- *
164
- * Consumers read the `recommendations` array first — that's the
165
- * actionable layer, ranked by priority. The numeric sections back it up.
166
- */
167
-
168
- interface InsightReport {
169
- /** Number of runs analyzed. */
170
- n: number;
171
- /** Runtime facts carried by the run records. These describe execution,
172
- * not task quality: duration, queueing, token categories, models, and
173
- * explicitly recorded failures. */
174
- execution: ExecutionInsight;
175
- /** Composite-score distribution across all runs. Always present. */
176
- composite: ScalarDistribution;
177
- /** Per-dimension distributions for every dimension that appeared in any
178
- * run's judge scores. Empty when no judge scores were recorded. */
179
- perDimension: Record<string, ScalarDistribution>;
180
- /** Cost/quality distribution and Pareto frontier. */
181
- costQuality: {
182
- cost: ScalarDistribution;
183
- pareto: ParetoFigureSpec;
184
- /** Cost source coverage. `uncaptured` rows are excluded from the USD
185
- * distribution and Pareto chart; observed and estimated totals remain
186
- * separate so reports never present estimates as billed spend. */
187
- provenance?: CostProvenanceSummary;
188
- /** Set when the cost/quality view is degraded because the input data
189
- * doesn't fully support it — e.g. all `costUsd` were zero, or only a
190
- * single candidate appears (so the Pareto is a single point). The
191
- * named fields name the degraded sub-view, free-text the reason. */
192
- degraded?: {
193
- cost?: string;
194
- pareto?: string;
195
- };
196
- };
197
- /** Per-judge calibration + bias detection. Populated for every judge name
198
- * that appears in `outcome.judgeScores`. Bias fields require either a
199
- * gold reference or multi-rater data. */
200
- judges: Record<string, JudgeInsight>;
201
- /** Inter-rater agreement when multiple judges scored the same runs.
202
- * Includes pairwise kappa and the specific run ids where raters
203
- * disagree — the cases worth a human meeting. */
204
- interRater?: InterRaterInsight;
205
- /** Pairwise lift (baseline → candidate) with bootstrap CI. Present when
206
- * `RunRecord.splitTag` includes both `holdout` and search/dev splits,
207
- * or when caller passes an explicit baseline/candidate split. */
208
- lift?: LiftInsight;
209
- /** Failure clusters with exemplars. Populated when an AnalystRegistry
210
- * is wired in `analyzeRuns({ analyst })`. */
211
- failureClusters?: FailureClusterInsight;
212
- /** Canary leak count + holdout audit status. Populated when canary
213
- * scenarios are passed in. */
214
- contamination?: ContaminationInsight;
215
- /** Correlation between judge composite and a downstream outcome the
216
- * caller supplies (engagement, revenue, downstream pass rate, etc.).
217
- * When present, the optional reward model is the model that maps
218
- * judge scores → predicted outcome. */
219
- outcomeCorrelation?: OutcomeCorrelationInsight;
220
- /** Aggregate release-readiness summary. A consumer needing the full
221
- * substrate `ReleaseConfidenceScorecard` (SLO-axis evaluation,
222
- * ActionableSideInfo bag) calls `evaluateReleaseConfidence()` directly;
223
- * this summary captures the analyzeRuns-derived axes. */
224
- release: ReleaseSummary;
225
- /** Delta vs a prior period when `baselineRuns` is passed. Per-metric
226
- * current vs baseline with Welch CI + Cohen's d + significance flag.
227
- * Answers "did my last change help?" — the customer-conversion question.
228
- * Surfaced metrics: composite, cost, duration, tokenUsage, plus any
229
- * per-dimension judge metric present in both windows. */
230
- priorPeriodComparison?: PriorPeriodComparison;
231
- /** Model-free failure-mode breakdown from `RunRecord.failureMode`, ranked
232
- * by count descending. Present when any run carries a `failureMode`.
233
- * Complements `failureClusters` (LLM-semantic) with the structured tags
234
- * the harness already recorded — actionable with no analyst wired. */
235
- failureModes?: FailureModeTally[];
236
- /** Top-N actionable recommendations, ranked by priority. The packet's
237
- * human-readable layer; the numeric sections are the evidence. */
238
- recommendations: Recommendation[];
239
- }
240
- interface CostProvenanceSummary {
241
- observed: {
242
- n: number;
243
- totalUsd: number;
244
- };
245
- estimated: {
246
- n: number;
247
- totalUsd: number;
248
- };
249
- uncaptured: {
250
- n: number;
251
- };
252
- knownFraction: number;
253
- }
254
- interface ExecutionInsight {
255
- /** End-to-end wall time for every run. */
256
- durationMs: ScalarDistribution;
257
- /** Queue time for the subset of runs that recorded it. */
258
- queueMs: ScalarDistribution;
259
- /** Token distributions plus corpus totals. Optional token categories use
260
- * distribution `n` to disclose how many runs recorded that category. */
261
- tokenUsage: TokenUsageInsight;
262
- /** Usage reported only by orchestration or agent aggregate spans.
263
- * Kept separate because it may duplicate model-call telemetry in other traces. */
264
- aggregateUsage: {
265
- runs: number;
266
- tokenUsage: TokenUsageInsight;
267
- costUsd: ScalarDistribution;
268
- totalCostUsd: number;
269
- };
270
- /** Stable model counts, largest cohort first. */
271
- models: Array<{
272
- model: string;
273
- runs: number;
274
- }>;
275
- /** Model-call coverage. `events` is available only from producers that
276
- * record `outcome.raw.llm_span_count`; `runs` also recognizes non-zero
277
- * token usage from other producers. */
278
- modelCalls: {
279
- runs: number;
280
- events: number;
281
- reportingRuns: number;
282
- };
283
- /** Failure counts remain separate from outcome scores. `reportedErrorEvents`
284
- * sums `outcome.raw.error_span_count` only where a producer supplied it. */
285
- failures: {
286
- runs: number;
287
- fraction: number;
288
- reportedErrorEvents: number;
289
- reportingRuns: number;
290
- };
291
- }
292
- interface TokenUsageInsight {
293
- input: ScalarDistribution;
294
- output: ScalarDistribution;
295
- reasoning: ScalarDistribution;
296
- cached: ScalarDistribution;
297
- cacheWrite: ScalarDistribution;
298
- totals: {
299
- input: number;
300
- output: number;
301
- reasoning: number;
302
- cached: number;
303
- cacheWrite: number;
304
- };
305
- }
306
- /** Distributional summary of a scalar-valued metric. */
307
- interface ScalarDistribution {
308
- /** Sample count after dropping non-finite values. */
309
- n: number;
310
- mean: number;
311
- p50: number;
312
- p95: number;
313
- stddev: number;
314
- min: number;
315
- max: number;
316
- /** Histogram bins using `agent-eval`'s `gainHistogram` primitive. */
317
- histogram: GainDistributionBin[];
318
- /** Worst-N runs by score, ascending. Populated for the composite
319
- * distribution so the report names the runs a customer should
320
- * inspect first. Undefined when the distribution was computed from a
321
- * raw value list with no run identity (e.g. cost). */
322
- tailRuns?: Array<{
323
- runId: string;
324
- score: number;
325
- }>;
326
- }
327
- interface JudgeInsight {
328
- /** Number of times this judge scored a run. */
329
- n: number;
330
- /** Mean composite over this judge's runs. */
331
- meanScore: number;
332
- /** Calibration against a gold reference, when provided. Cohen's κ for
333
- * binary thresholding + continuous agreement metrics. */
334
- calibration?: ContinuousAgreement;
335
- /** Positional bias — when the judge sees options in different orders,
336
- * do its preferences track the content or the position? */
337
- positionalBias?: number;
338
- /** Self-preference — when the judge sees its own model's output vs a
339
- * competitor, does it over-pick its own? */
340
- selfPreference?: number;
341
- /** Verbosity bias — does the judge reward longer outputs regardless of
342
- * quality? */
343
- verbosityBias?: number;
344
- }
345
- interface InterRaterInsight {
346
- /** Number of raters whose scores were aggregated. */
347
- raters: number;
348
- /** Number of runs every rater scored. */
349
- jointlyRated: number;
350
- /** Cohen's κ averaged across rater pairs. */
351
- kappa: number;
352
- /** Pairwise κ per rater pair (key = `"raterA::raterB"`). */
353
- perPair: Record<string, number>;
354
- /** Run ids where raters disagree the most — the high-value triage list. */
355
- disagreementCases: Array<{
356
- runId: string;
357
- ratings: Array<{
358
- rater: string;
359
- score: number;
360
- }>;
361
- range: number;
362
- }>;
363
- }
364
- interface LiftInsight {
365
- baselineMean: number;
366
- candidateMean: number;
367
- /** Candidate − baseline. */
368
- delta: number;
369
- /** Lower / upper bound of bootstrap CI on the delta. */
370
- ci95: [number, number];
371
- /** Paired-t-test p-value. */
372
- pValue: number;
373
- /** Number of paired observations. */
374
- n: number;
375
- /** Cohen's d for the delta. */
376
- cohensD: number;
377
- /** Minimum detectable effect at current n, 80% power. */
378
- mde: number;
379
- /** Sample size needed to detect the observed delta at 80% power. */
380
- requiredN: number;
381
- }
382
- interface FailureClusterInsight {
383
- /** All clusters identified by the registry, ranked by share descending. */
384
- clusters: Array<{
385
- id: string;
386
- name: string;
387
- /** Fraction of failed runs in this cluster, 0..1. */
388
- share: number;
389
- /** Exemplar `runId`s (≤ 5) the consumer can drill into. */
390
- exemplars: string[];
391
- /** Short LLM-generated suggested fix when the registry supports it. */
392
- suggestedFix?: string;
393
- }>;
394
- totalFailures: number;
395
- }
396
- /** Model-free failure breakdown over the structured `RunRecord.failureMode`
397
- * enum. Unlike `failureClusters` (semantic, requires an LLM analyst), this
398
- * is computed directly from the tags the harness already recorded — so a
399
- * customer ingesting one batch with no judge/analyst still learns which
400
- * named failure dominates. */
401
- interface FailureModeTally {
402
- /** The `failureMode` tag. */
403
- mode: string;
404
- /** Number of runs carrying this tag. */
405
- count: number;
406
- /** Share of the whole corpus, 0..1. */
407
- share: number;
408
- }
409
- interface ContaminationInsight {
410
- /** Canary phrases that leaked into outputs. */
411
- leaks: number;
412
- /** Holdout audit verdict — did any holdout-tagged run end up in the
413
- * search/dev pool, or vice versa? */
414
- holdoutAuditPassed: boolean;
415
- details?: Array<{
416
- runId: string;
417
- canary: string;
418
- matched: string;
419
- }>;
420
- }
421
- interface OutcomeCorrelationInsight {
422
- /** What outcome the consumer is correlating against (e.g.
423
- * `'engagement_rate'`, `'approval_rate'`, `'downstream_pass'`). */
424
- metric: string;
425
- /** Number of (run, outcome) pairs used. */
426
- n: number;
427
- /** Pearson correlation between composite score and outcome. */
428
- pearson: number;
429
- /** Spearman rank correlation — robust to monotonic non-linearity. */
430
- spearman: number;
431
- /** When present, the simple linear reward model fit to the data. */
432
- rewardModel?: {
433
- intercept: number;
434
- slope: number;
435
- r2: number;
436
- };
437
- }
438
- interface ReleaseSummary {
439
- /** Overall verdict across axes — fail if any axis fails, else warn if any
440
- * warns, else pass. */
441
- status: 'pass' | 'warn' | 'fail';
442
- axes: Array<{
443
- name: 'quality-lift' | 'contamination' | 'composite-distribution';
444
- status: 'pass' | 'warn' | 'fail';
445
- detail: string;
446
- }>;
447
- /** Free-form issues surfaced beyond the standard axes. Empty by default;
448
- * consumers can post-process to populate. */
449
- issues: string[];
450
- }
451
- interface MetricDelta {
452
- /** Current-period mean. */
453
- current: number;
454
- /** Baseline-period mean. */
455
- baseline: number;
456
- /** current - baseline. Positive means improved (or, for cost/duration,
457
- * the consumer-side interpretation: "higher current" — semantic
458
- * direction depends on the metric). */
459
- delta: number;
460
- /** Welch 95% confidence interval on the delta. Two-sample, unpaired —
461
- * the baseline and current run sets may have different scenarios. */
462
- ci95: [number, number];
463
- /** Welch t-test p-value (two-sided). */
464
- pValue: number;
465
- /** Cohen's d (pooled stddev). Effect size, signed. */
466
- cohensD: number;
467
- /** Sample sizes. */
468
- baselineN: number;
469
- currentN: number;
470
- /** True when p < 0.05 AND |d| >= 0.2 (small-effect threshold). The
471
- * conjunction prevents large-effect-but-noisy and significant-but-
472
- * tiny from triggering recommendations. */
473
- significant: boolean;
474
- }
475
- interface PriorPeriodComparison {
476
- /** Sample counts. */
477
- baselineN: number;
478
- currentN: number;
479
- /** Optional human-readable label — "vs prior 7 days", "vs v3 release". */
480
- windowLabel?: string;
481
- /** Every metric we could compare. Keys: 'composite', 'cost', 'duration',
482
- * 'tokenUsage' for always-present ones; per-dimension keys when both
483
- * windows have judge scores on the same dimension. */
484
- metrics: Record<string, MetricDelta>;
485
- /** Metric names where current is significantly WORSE than baseline.
486
- * Direction-aware: for cost/duration, higher current = worse. */
487
- regressedMetrics: string[];
488
- /** Metric names where current is significantly BETTER than baseline. */
489
- improvedMetrics: string[];
490
- }
491
- interface Recommendation {
492
- priority: 'critical' | 'high' | 'medium' | 'low';
493
- kind: 'ship' | 'hold' | 'investigate' | 'fix' | 'recalibrate' | 'expand-corpus';
494
- title: string;
495
- detail: string;
496
- /** Optional pointer back into the report for the evidence. */
497
- evidencePath?: string;
498
- }
499
-
500
- /**
501
- * # Hosted-tier wire format — the schema that EVERY orchestrator (ours,
502
- * a partner's self-hosted one, a future open implementation) must accept.
503
- *
504
- * **Stability:** every type in this file is committed under semver. New
505
- * minors only ADD optional fields. Breaking changes mean a major bump
506
- * (`HostedWireVersion` literal increment).
507
- *
508
- * The wire format is two event streams in one transport:
509
- *
510
- * 1. **Eval-run events** (`POST /v1/ingest/eval-runs`). Posted when a
511
- * campaign / improvement-loop completes (or per-generation if
512
- * streaming). Carries the structured result + per-cell scores +
513
- * surface diffs the orchestrator stores for the dashboard.
514
- *
515
- * 2. **Trace spans** (`POST /v1/ingest/traces`). Standard OTLP-shaped
516
- * spans with a few additional attributes so the orchestrator can
517
- * pivot from eval-run → underlying execution. Compatible with any
518
- * OTel collector.
519
- *
520
- * Both endpoints are authenticated with a bearer token + a tenant id
521
- * header. Tenants isolate everything downstream of ingest; no tenant
522
- * ever sees another tenant's data.
523
- */
524
-
525
- declare const HOSTED_WIRE_VERSION: "2026-05-26.v1";
526
- type HostedWireVersion = typeof HOSTED_WIRE_VERSION;
527
- /** Every ingest request carries these. */
528
- interface HostedIngestHeaders {
529
- /** Bearer token. The orchestrator validates against the tenant key. */
530
- authorization: `Bearer ${string}`;
531
- /** Stable tenant id (the orchestrator-side primary key for the tenant). */
532
- 'x-tangle-tenant-id': string;
533
- /** Wire-version pin so the server can reject incompatible payloads. */
534
- 'x-tangle-wire-version': HostedWireVersion;
535
- /** Optional idempotency key for retry-safe ingest. */
536
- 'idempotency-key'?: string;
537
- }
538
- /** Lifecycle stages of an eval-run as the substrate reports them. */
539
- type EvalRunStatus = 'started' | 'baseline-complete' | 'generation-complete' | 'gate-decided' | 'finished' | 'errored';
540
- interface EvalRunCellScore {
541
- /** Stable scenario id from the consumer's scenario set. */
542
- scenarioId: string;
543
- /** Repetition index when reps > 1; 0 for the default. */
544
- rep: number;
545
- /** Composite score across all judges + dimensions for this cell. */
546
- compositeMean: number;
547
- /** Per-judge → per-dimension scores; null where the judge did not run. */
548
- dimensions: Record<string, Record<string, number>>;
549
- /** Per-cell error message if the dispatch threw. Null on success. */
550
- errorMessage?: string;
551
- }
552
- interface EvalRunGenerationSnapshot {
553
- /** Generation index. 0 is baseline. */
554
- index: number;
555
- /** Candidate surface fingerprint (stable hash) — pivot key into the
556
- * trace stream to fetch the underlying execution. */
557
- surfaceHash: string;
558
- /** The candidate surface itself. May be omitted to avoid PII when the
559
- * consumer prefers not to ship verbatim prompts. */
560
- surface?: MutableSurface;
561
- /** Per-cell scores for this generation. */
562
- cells: EvalRunCellScore[];
563
- /** Aggregate composite mean across all cells in this generation. */
564
- compositeMean: number;
565
- /** Total $ spent across this generation. */
566
- costUsd: number;
567
- /** Wall-clock duration of this generation. */
568
- durationMs: number;
569
- }
570
- /**
571
- * The top-level eval-run event. One ingest call per logical eval-run;
572
- * generations stream in incrementally via repeated calls with the same
573
- * `runId`. The orchestrator deduplicates by `(runId, generation.index)`.
574
- */
575
- interface EvalRunEvent {
576
- /** Stable run id (the substrate's `runId`). UUID or substrate-generated. */
577
- runId: string;
578
- /** Where this run was happening — derived from `RunCampaignOptions.runDir`. */
579
- runDir: string;
580
- /** ISO-8601 timestamp the substrate recorded the event. */
581
- timestamp: string;
582
- /** Lifecycle stage this event represents. */
583
- status: EvalRunStatus;
584
- /** Free-form consumer tags (env, branch, model id, etc.). Searchable. */
585
- labels: Record<string, string>;
586
- /** Baseline campaign snapshot. Present when status >= baseline-complete. */
587
- baseline?: EvalRunGenerationSnapshot;
588
- /** Per-generation snapshots. Streams in; orchestrator appends. */
589
- generations: EvalRunGenerationSnapshot[];
590
- /** Final gate decision. Present when status >= gate-decided. */
591
- gateDecision?: GateDecision;
592
- /** Held-out lift = winner-on-holdout - baseline-on-holdout. */
593
- holdoutLift?: number;
594
- /** Total $ spent across baseline + every generation. */
595
- totalCostUsd: number;
596
- /** Total wall-clock duration. */
597
- totalDurationMs: number;
598
- /** Error message if status === 'errored'. */
599
- errorMessage?: string;
600
- /** Rigor packet emitted alongside the run — distributional summary,
601
- * paired-bootstrap lift CI, judge stats, inter-rater agreement,
602
- * contamination check, failure clusters (when an analyst is wired),
603
- * outcome correlation (when downstream signal is supplied), and the
604
- * recommendations the dashboard surfaces verbatim. Additive; older
605
- * clients that don't know about this field continue to work. */
606
- insightReport?: InsightReport;
607
- }
608
- /**
609
- * OTel-shape span with a few additional attributes for eval-run pivoting.
610
- * Compatible with any OTLP collector — `name`, `traceId`, `spanId`,
611
- * `startTimeUnixNano`, `endTimeUnixNano`, `attributes` are stock OTel.
612
- */
613
- interface TraceSpanEvent {
614
- traceId: string;
615
- spanId: string;
616
- parentSpanId?: string;
617
- name: string;
618
- startTimeUnixNano: number;
619
- endTimeUnixNano: number;
620
- attributes: Record<string, string | number | boolean>;
621
- events?: Array<{
622
- timeUnixNano: number;
623
- name: string;
624
- attributes?: Record<string, string | number | boolean>;
625
- }>;
626
- status?: {
627
- code: 'OK' | 'ERROR' | 'UNSET';
628
- message?: string;
629
- };
630
- /** Pivot back into the eval-run stream. */
631
- 'tangle.runId'?: string;
632
- /** Pivot to the specific generation. */
633
- 'tangle.generation'?: number;
634
- /** Pivot to the specific cell. */
635
- 'tangle.cellId'?: string;
636
- /** Pivot to the specific scenario. */
637
- 'tangle.scenarioId'?: string;
638
- }
639
- interface IngestEvalRunsRequest {
640
- wireVersion: HostedWireVersion;
641
- events: EvalRunEvent[];
642
- }
643
- interface IngestTracesRequest {
644
- wireVersion: HostedWireVersion;
645
- spans: TraceSpanEvent[];
646
- }
647
- interface IngestResponse {
648
- /** Accepted events / spans count. */
649
- accepted: number;
650
- /** Rejected events with reasons (validation failures, dup idempotency key, etc.). */
651
- rejected: Array<{
652
- index: number;
653
- reason: string;
654
- }>;
655
- }
656
-
657
- /**
658
- * # Hosted-tier ingest client.
659
- *
660
- * Ships eval-run events + trace spans to any orchestrator (ours, a
661
- * partner's self-hosted one, or a future open implementation) that
662
- * speaks the wire format in `./types.ts`.
663
- *
664
- * Three modes:
665
- * - **Ours:** point at `https://orchestrator.tangle.tools` (the host root —
666
- * the client appends the versioned `/v1/ingest/...` path itself; a trailing
667
- * `/v1` on the endpoint is tolerated and normalized away). We handle ingest
668
- * + storage + dashboard.
669
- * - **Self-hosted:** point at whatever URL runs the reference receiver
670
- * from `examples/hosted-ingest-server/`.
671
- * - **Off (default):** when `hostedTenant` is unset, nothing is sent.
672
- * Everything stays local.
673
- */
674
-
675
- interface HostedTenant {
676
- /** Orchestrator endpoint base URL (no trailing slash). Required. */
677
- endpoint: string;
678
- /** Bearer token issued by the orchestrator. Required. */
679
- apiKey: string;
680
- /** Tenant id — the orchestrator's primary key for this consumer. Required. */
681
- tenantId: string;
682
- /** Optional `fetch` override (auth wrappers, custom agent, test mocks). */
683
- fetchImpl?: typeof fetch;
684
- /** Per-call timeout in ms. Default 30s. */
685
- timeoutMs?: number;
686
- /** Retries on 5xx / network errors. Default 2. */
687
- retries?: number;
688
- }
689
- interface HostedClient {
690
- ingestEvalRun(event: EvalRunEvent, idempotencyKey?: string): Promise<IngestResponse>;
691
- ingestEvalRuns(events: EvalRunEvent[], idempotencyKey?: string): Promise<IngestResponse>;
692
- ingestTraces(spans: TraceSpanEvent[], idempotencyKey?: string): Promise<IngestResponse>;
693
- readonly tenant: HostedTenant;
694
- readonly wireVersion: HostedWireVersion;
695
- }
696
- declare function createHostedClient(tenant: HostedTenant): HostedClient;
697
- /**
698
- * Build a `HostedClient` from environment, or `undefined` when ingest is not
699
- * configured — the canonical, fail-soft wiring every product uses so eval-run +
700
- * trace provenance lands in the Intelligence dashboard with ONE call:
701
- *
702
- * const hosted = hostedClientFromEnv()
703
- * // ...run the loop...
704
- * await emitLoopProvenance({ ..., hostedClient: hosted }) // no-op if undefined
705
- *
706
- * Returns `undefined` (NOT an error) when any of endpoint / apiKey / tenantId is
707
- * missing — so a product wires the ship call unconditionally and it stays a
708
- * no-op until the env is set. Env precedence:
709
- * - endpoint: `TANGLE_INGEST_URL` → `TANGLE_ORCHESTRATOR_URL`
710
- * - apiKey: `TANGLE_INGEST_API_KEY` → `TANGLE_API_KEY`
711
- * - tenantId: `TANGLE_TENANT_ID`
712
- * A trailing slash on the endpoint is stripped. Pass `overrides` to supply any
713
- * field directly (e.g. a fixed `tenantId` per product) — overrides win over env.
714
- */
715
- /**
716
- * Build a {@link HostedTenant} config from env — the input `selfImprove`'s
717
- * `hostedTenant` and `emitLoopProvenance` take. Same env precedence + overrides
718
- * as {@link hostedClientFromEnv}; returns `undefined` (not an error) when any of
719
- * endpoint / apiKey / tenantId is missing, so a product wires
720
- * `hostedTenant: hostedTenantFromEnv({ tenantId: 'my-agent' })` unconditionally
721
- * and it stays off until the env is set.
722
- */
723
- declare function hostedTenantFromEnv(overrides?: Partial<HostedTenant> & {
724
- env?: Record<string, string | undefined>;
725
- }): HostedTenant | undefined;
726
- declare function hostedClientFromEnv(overrides?: Partial<HostedTenant> & {
727
- env?: Record<string, string | undefined>;
728
- }): HostedClient | undefined;
729
-
730
- export { type EvalRunCellScore, type EvalRunEvent, type EvalRunGenerationSnapshot, type EvalRunStatus, HOSTED_WIRE_VERSION, type HostedClient, type HostedIngestHeaders, type HostedTenant, type HostedWireVersion, type IngestEvalRunsRequest, type IngestResponse, type IngestTracesRequest, type TraceSpanEvent, createHostedClient, hostedClientFromEnv, hostedTenantFromEnv };