@tangle-network/agent-eval 0.150.0 → 0.150.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (150) hide show
  1. package/CHANGELOG.md +43 -0
  2. package/dist/analyst/index.d.ts +13 -13
  3. package/dist/analyst/index.js +4 -4
  4. package/dist/{backend-integrity-DaeroaEs.d.ts → backend-integrity-DOCa_QrR.d.ts} +2 -2
  5. package/dist/{backend-integrity-DaeroaEs.d.ts.map → backend-integrity-DOCa_QrR.d.ts.map} +1 -1
  6. package/dist/{benchmark-A0E-JZ1H.d.ts → benchmark-BtAWA8nT.d.ts} +3 -3
  7. package/dist/{benchmark-A0E-JZ1H.d.ts.map → benchmark-BtAWA8nT.d.ts.map} +1 -1
  8. package/dist/{benchmark-command-CqGA0SJQ.js → benchmark-command-BU1Las59.js} +7 -7
  9. package/dist/{benchmark-command-CqGA0SJQ.js.map → benchmark-command-BU1Las59.js.map} +1 -1
  10. package/dist/benchmarks/index.d.ts +4 -4
  11. package/dist/benchmarks/index.js +3 -3
  12. package/dist/builder-eval/index.js +1 -1
  13. package/dist/campaign/index.d.ts +8 -8
  14. package/dist/campaign/index.js +5 -5
  15. package/dist/{campaign-CR6Yt5ZS.js → campaign-la-gEYNz.js} +18 -9
  16. package/dist/campaign-la-gEYNz.js.map +1 -0
  17. package/dist/{capture-fetch-Im0UN1i6.d.ts → capture-fetch-BBVFzhkk.d.ts} +2 -2
  18. package/dist/{capture-fetch-Im0UN1i6.d.ts.map → capture-fetch-BBVFzhkk.d.ts.map} +1 -1
  19. package/dist/{chat-client-sFiClniQ.js → chat-client-Bvmxedyv.js} +3 -3
  20. package/dist/{chat-client-sFiClniQ.js.map → chat-client-Bvmxedyv.js.map} +1 -1
  21. package/dist/cli.js +2 -2
  22. package/dist/{client-w6vVJhSp.d.ts → client-kPQYT_56.d.ts} +2 -2
  23. package/dist/{client-w6vVJhSp.d.ts.map → client-kPQYT_56.d.ts.map} +1 -1
  24. package/dist/contract/index.d.ts +10 -10
  25. package/dist/contract/index.js +5 -5
  26. package/dist/{default-registry-BhRNsP-D.d.ts → default-registry-Cw0Ohdoj.d.ts} +6 -6
  27. package/dist/{default-registry-BhRNsP-D.d.ts.map → default-registry-Cw0Ohdoj.d.ts.map} +1 -1
  28. package/dist/{define-agent-eval-ChrtErmg.js → define-agent-eval-C8V8sMqP.js} +7 -7
  29. package/dist/{define-agent-eval-ChrtErmg.js.map → define-agent-eval-C8V8sMqP.js.map} +1 -1
  30. package/dist/{define-agent-eval-Cw3mkB4-.d.ts → define-agent-eval-CEQWL9Hy.d.ts} +5 -5
  31. package/dist/{define-agent-eval-Cw3mkB4-.d.ts.map → define-agent-eval-CEQWL9Hy.d.ts.map} +1 -1
  32. package/dist/{descriptive-B5MwKfbf.js → descriptive-jDOuI6mz.js} +22 -2
  33. package/dist/descriptive-jDOuI6mz.js.map +1 -0
  34. package/dist/{dspy-rlm-engine-Csi3CUAT.js → dspy-rlm-engine-CBYlPvNy.js} +2 -2
  35. package/dist/{dspy-rlm-engine-Csi3CUAT.js.map → dspy-rlm-engine-CBYlPvNy.js.map} +1 -1
  36. package/dist/{engine-CgQdUh-r.d.ts → engine-BLzhNzoY.d.ts} +4 -4
  37. package/dist/{engine-CgQdUh-r.d.ts.map → engine-BLzhNzoY.d.ts.map} +1 -1
  38. package/dist/{eval-campaign-Czc2TcW5.js → eval-campaign-CQuZrLR_.js} +3 -3
  39. package/dist/{eval-campaign-Czc2TcW5.js.map → eval-campaign-CQuZrLR_.js.map} +1 -1
  40. package/dist/{exact-types-D9_T83Ol.d.ts → exact-types-ccQAyut1.d.ts} +2 -2
  41. package/dist/{exact-types-D9_T83Ol.d.ts.map → exact-types-ccQAyut1.d.ts.map} +1 -1
  42. package/dist/experiment/index.d.ts +3 -3
  43. package/dist/{experiment-tracker-Cp9ougHs.d.ts → experiment-tracker-0MhuPArU.d.ts} +2 -2
  44. package/dist/{experiment-tracker-Cp9ougHs.d.ts.map → experiment-tracker-0MhuPArU.d.ts.map} +1 -1
  45. package/dist/{external-optimizer-contracts-mbhR9Ldu.d.ts → external-optimizer-contracts-DbLsm4Po.d.ts} +2 -2
  46. package/dist/{external-optimizer-contracts-mbhR9Ldu.d.ts.map → external-optimizer-contracts-DbLsm4Po.d.ts.map} +1 -1
  47. package/dist/{external-optimizer-process-tDe72VRO.js → external-optimizer-process-x9oEXKsU.js} +2 -2
  48. package/dist/{external-optimizer-process-tDe72VRO.js.map → external-optimizer-process-x9oEXKsU.js.map} +1 -1
  49. package/dist/{external-optimizer-subprocess-BepyNwtj.js → external-optimizer-subprocess-CKNb42oM.js} +278 -43
  50. package/dist/external-optimizer-subprocess-CKNb42oM.js.map +1 -0
  51. package/dist/{feedback-trajectory-Cc9M2tNF.d.ts → feedback-trajectory-DpTTjo0q.d.ts} +3 -3
  52. package/dist/{feedback-trajectory-Cc9M2tNF.d.ts.map → feedback-trajectory-DpTTjo0q.d.ts.map} +1 -1
  53. package/dist/hosted/index.d.ts +2 -2
  54. package/dist/{index-DL4MJuJL.d.ts → index-Aj3WO3_a.d.ts} +8 -8
  55. package/dist/{index-DL4MJuJL.d.ts.map → index-Aj3WO3_a.d.ts.map} +1 -1
  56. package/dist/index-C1ravkGA.d.ts +1 -0
  57. package/dist/{index-dh0kp2GM.d.ts → index-IQccV3Ou.d.ts} +15 -10
  58. package/dist/index-IQccV3Ou.d.ts.map +1 -0
  59. package/dist/index.d.ts +21 -90
  60. package/dist/index.d.ts.map +1 -1
  61. package/dist/index.js +13 -13
  62. package/dist/{integrity-D2MAlC_K.d.ts → integrity-B0dZ96EO.d.ts} +2 -2
  63. package/dist/{integrity-D2MAlC_K.d.ts.map → integrity-B0dZ96EO.d.ts.map} +1 -1
  64. package/dist/{integrity-DysDBWDu.js → integrity-DL91tucI.js} +16 -3
  65. package/dist/integrity-DL91tucI.js.map +1 -0
  66. package/dist/{judge-calibration-DZkWrm5H.js → judge-calibration-zZjLz8hr.js} +2 -2
  67. package/dist/{judge-calibration-DZkWrm5H.js.map → judge-calibration-zZjLz8hr.js.map} +1 -1
  68. package/dist/{llm-client-CxyJ8Oft.js → llm-client-Bg32RW0j.js} +47 -2
  69. package/dist/llm-client-Bg32RW0j.js.map +1 -0
  70. package/dist/{llm-judge-D7UuMI2t.js → llm-judge-BqqMS8t7.js} +5 -5
  71. package/dist/{llm-judge-D7UuMI2t.js.map → llm-judge-BqqMS8t7.js.map} +1 -1
  72. package/dist/{matrix-D3hL68uV.d.ts → matrix-DrVnRp4G.d.ts} +2 -2
  73. package/dist/{matrix-D3hL68uV.d.ts.map → matrix-DrVnRp4G.d.ts.map} +1 -1
  74. package/dist/meta-eval/index.d.ts +1 -1
  75. package/dist/meta-eval/index.js +2 -2
  76. package/dist/multishot/golden/index.d.ts +1 -1
  77. package/dist/multishot/index.d.ts +2 -2
  78. package/dist/openapi.json +1 -1
  79. package/dist/pipelines/index.js +1 -1
  80. package/dist/{produced-state-WJs_enbZ.js → produced-state-Bnq4FaDO.js} +3 -3
  81. package/dist/{produced-state-WJs_enbZ.js.map → produced-state-Bnq4FaDO.js.map} +1 -1
  82. package/dist/{promotion-policy-DYRzXdTV.d.ts → promotion-policy-DLOUkYhI.d.ts} +2 -2
  83. package/dist/{promotion-policy-DYRzXdTV.d.ts.map → promotion-policy-DLOUkYhI.d.ts.map} +1 -1
  84. package/dist/{provenance-Bh9k3sjl.d.ts → provenance-oA4-zUqm.d.ts} +6 -6
  85. package/dist/{provenance-Bh9k3sjl.d.ts.map → provenance-oA4-zUqm.d.ts.map} +1 -1
  86. package/dist/{registry-DhpX9Bsg.d.ts → registry-BQwrSYpC.d.ts} +4 -4
  87. package/dist/{registry-DhpX9Bsg.d.ts.map → registry-BQwrSYpC.d.ts.map} +1 -1
  88. package/dist/reporting.js +2 -2
  89. package/dist/{researcher-CbCK-ppF.d.ts → researcher-DJnoUE8c.d.ts} +3 -3
  90. package/dist/{researcher-CbCK-ppF.d.ts.map → researcher-DJnoUE8c.d.ts.map} +1 -1
  91. package/dist/{reward-hacking-DFo2FU5J.js → reward-hacking-C0x0xihA.js} +20 -2
  92. package/dist/reward-hacking-C0x0xihA.js.map +1 -0
  93. package/dist/rl.d.ts +2 -2
  94. package/dist/rl.js +3 -3
  95. package/dist/{rubric-predictive-validity-Cwwyd7ah.js → rubric-predictive-validity-CzxLoZge.js} +2 -2
  96. package/dist/{rubric-predictive-validity-Cwwyd7ah.js.map → rubric-predictive-validity-CzxLoZge.js.map} +1 -1
  97. package/dist/{semantic-concept-judge-B7KPD57W.js → semantic-concept-judge-laMCnTLn.js} +2 -2
  98. package/dist/{semantic-concept-judge-B7KPD57W.js.map → semantic-concept-judge-laMCnTLn.js.map} +1 -1
  99. package/dist/{series-convergence-VhTad5TM.d.ts → series-convergence-BxKEgBwA.d.ts} +2 -2
  100. package/dist/{series-convergence-VhTad5TM.d.ts.map → series-convergence-BxKEgBwA.d.ts.map} +1 -1
  101. package/dist/{server-DNiUVirV.js → server-dIWwF3j_.js} +2 -2
  102. package/dist/{server-DNiUVirV.js.map → server-dIWwF3j_.js.map} +1 -1
  103. package/dist/{skillopt-optimization-method-BtgL6PMH.d.ts → skillopt-optimization-method-BDD_o1xE.d.ts} +5 -5
  104. package/dist/{skillopt-optimization-method-BtgL6PMH.d.ts.map → skillopt-optimization-method-BDD_o1xE.d.ts.map} +1 -1
  105. package/dist/{skillopt-optimization-method-L8vi4h4m.js → skillopt-optimization-method-CPBlTcj5.js} +5 -5
  106. package/dist/{skillopt-optimization-method-L8vi4h4m.js.map → skillopt-optimization-method-CPBlTcj5.js.map} +1 -1
  107. package/dist/{statistical-heldout-Caxd30ZF.d.ts → statistical-heldout-_woZ9q9j.d.ts} +2 -2
  108. package/dist/{statistical-heldout-Caxd30ZF.d.ts.map → statistical-heldout-_woZ9q9j.d.ts.map} +1 -1
  109. package/dist/{store-tool-spans-BGg2QFk1.d.ts → store-tool-spans-Br2_IUhm.d.ts} +4 -4
  110. package/dist/{store-tool-spans-BGg2QFk1.d.ts.map → store-tool-spans-Br2_IUhm.d.ts.map} +1 -1
  111. package/dist/{summary-report-Blysd6Z2.js → summary-report-DW2bEpdB.js} +2 -2
  112. package/dist/{summary-report-Blysd6Z2.js.map → summary-report-DW2bEpdB.js.map} +1 -1
  113. package/dist/supervisor-run/index.d.ts +26 -4
  114. package/dist/supervisor-run/index.d.ts.map +1 -1
  115. package/dist/supervisor-run/index.js +102 -24
  116. package/dist/supervisor-run/index.js.map +1 -1
  117. package/dist/{tool-groups-BuhFx38z.d.ts → tool-groups-B4tqh8jB.d.ts} +3 -3
  118. package/dist/tool-groups-B4tqh8jB.d.ts.map +1 -0
  119. package/dist/{tool-waste-BDdBZG1F.js → tool-waste-C-VHSRwF.js} +3 -3
  120. package/dist/{tool-waste-BDdBZG1F.js.map → tool-waste-C-VHSRwF.js.map} +1 -1
  121. package/dist/trace-repair/index.d.ts +1 -1
  122. package/dist/traces.d.ts +7 -7
  123. package/dist/{types-BNbhD_Vj.d.ts → types-B2NsbrNy.d.ts} +2 -2
  124. package/dist/{types-BNbhD_Vj.d.ts.map → types-B2NsbrNy.d.ts.map} +1 -1
  125. package/dist/{types-B61iI2ZG.d.ts → types-CLAwnY-L.d.ts} +2 -2
  126. package/dist/{types-B61iI2ZG.d.ts.map → types-CLAwnY-L.d.ts.map} +1 -1
  127. package/dist/{types-COEjZMLY.d.ts → types-DdFNuyxQ.d.ts} +3 -3
  128. package/dist/{types-COEjZMLY.d.ts.map → types-DdFNuyxQ.d.ts.map} +1 -1
  129. package/dist/{types-yLK8gXE9.d.ts → types-I5WwVzQ7.d.ts} +159 -9
  130. package/dist/types-I5WwVzQ7.d.ts.map +1 -0
  131. package/dist/{types-COAJy8Ku.d.ts → types-jUBXJ7Iz.d.ts} +36 -3
  132. package/dist/types-jUBXJ7Iz.d.ts.map +1 -0
  133. package/dist/wire/index.d.ts +2 -2
  134. package/dist/wire/index.js +1 -1
  135. package/docs/adapters-observability.md +9 -23
  136. package/docs/campaign-proposers.md +18 -3
  137. package/docs/concepts.md +3 -4
  138. package/docs/wire-protocol.md +1 -1
  139. package/package.json +1 -1
  140. package/dist/campaign-CR6Yt5ZS.js.map +0 -1
  141. package/dist/descriptive-B5MwKfbf.js.map +0 -1
  142. package/dist/external-optimizer-subprocess-BepyNwtj.js.map +0 -1
  143. package/dist/index-DjpxROrT.d.ts +0 -1
  144. package/dist/index-dh0kp2GM.d.ts.map +0 -1
  145. package/dist/integrity-DysDBWDu.js.map +0 -1
  146. package/dist/llm-client-CxyJ8Oft.js.map +0 -1
  147. package/dist/reward-hacking-DFo2FU5J.js.map +0 -1
  148. package/dist/tool-groups-BuhFx38z.d.ts.map +0 -1
  149. package/dist/types-COAJy8Ku.d.ts.map +0 -1
  150. package/dist/types-yLK8gXE9.d.ts.map +0 -1
@@ -1,4 +1,91 @@
1
1
  import { h as RolloutLine } from "./schema-Cef2cFmb.js";
2
+ //#region src/statistics/descriptive.d.ts
3
+ /**
4
+ * Descriptive statistics: means, bootstrap spread, correlation, and the
5
+ * weighted judge-dimension composite. Nothing here is a significance test.
6
+ */
7
+ /** Weighted mean — falls back to uniform weights when omitted */
8
+ declare function weightedMean(scores: {
9
+ score: number;
10
+ weight?: number;
11
+ }[]): number;
12
+ /**
13
+ * Percentile bootstrap confidence interval on the mean of `scores`.
14
+ *
15
+ * Descriptive spread. It is not a significance test, and at small n its bounds
16
+ * are anti-conservative in the same way {@link pairedBootstrap}'s are — see
17
+ * {@link BOOTSTRAP_GATE_MIN_N}. With no `seed` the resampling is seeded from
18
+ * the scores themselves, so the interval is reproducible either way.
19
+ */
20
+ declare function confidenceInterval(scores: number[], confidence?: number, opts?: {
21
+ seed?: number;
22
+ resamples?: number;
23
+ }): {
24
+ mean: number;
25
+ lower: number;
26
+ upper: number;
27
+ };
28
+ /** Partial credit: returns 0-1 ratio of current toward target */
29
+ declare function partialCredit(current: number, target: number): number;
30
+ /** Distribution summary of a number series: count, extremes, quantiles, sum. */
31
+ interface SeriesDistribution {
32
+ readonly n: number;
33
+ readonly min: number;
34
+ readonly p50: number;
35
+ readonly p90: number;
36
+ readonly max: number;
37
+ readonly sum: number;
38
+ }
39
+ /**
40
+ * Fold a number series into its distribution summary. Quantiles use the
41
+ * nearest-rank definition — the `ceil(q·n)`-th order statistic — so every
42
+ * reported quantile is a value from the series. Returns `null` for an empty
43
+ * series: an empty series has no distribution, and a zero-filled summary
44
+ * would read as a measured all-zero series.
45
+ */
46
+ declare function summarizeNumberSeries(values: readonly number[]): SeriesDistribution | null;
47
+ /**
48
+ * Average-rank-with-ties transform (1-indexed). Tied values receive the mean
49
+ * of the ranks they span, the standard correction for Spearman's ρ.
50
+ */
51
+ declare function ranks(xs: number[]): number[];
52
+ /**
53
+ * Pearson product-moment correlation coefficient r ∈ [-1, 1] between two
54
+ * equal-length series. See the edge-case contract above: NaN for n < 2 or
55
+ * unequal lengths, 1 when both series are constant, 0 when exactly one is.
56
+ */
57
+ declare function pearsonR(a: number[], b: number[]): number;
58
+ /**
59
+ * Spearman's rank correlation ρ — Pearson over the average-rank-with-ties
60
+ * transform of each series. Same edge-case contract as {@link pearsonR}.
61
+ */
62
+ declare function spearmanR(a: number[], b: number[]): number;
63
+ interface WeightedCompositeInput {
64
+ /** Per-dimension scores (typically 0..1). */
65
+ dims: Record<string, number>;
66
+ /** Weight per dimension. Every weighted dimension MUST be present in
67
+ * `dims` — a weight for an absent dimension is a config error and throws,
68
+ * because silently dropping it would renormalise the composite onto a
69
+ * different denominator than intended. */
70
+ weights: Record<string, number>;
71
+ /** Optional pass threshold; when set, the result reports `pass`. */
72
+ threshold?: number;
73
+ }
74
+ interface WeightedCompositeResult {
75
+ composite: number;
76
+ pass?: boolean;
77
+ }
78
+ /**
79
+ * Weighted composite over judge dimensions: `Σ(score_d · w_d) / Σ(w_d)` across
80
+ * the weighted dimensions. The canonical replacement for the per-consumer
81
+ * hand-rolled composite math (tax/legal/creative/gtm each ship a copy).
82
+ *
83
+ * Fail-loud: throws if a weighted dimension is missing from `dims`, if any
84
+ * weight is negative, or if the weights sum to 0 — none of which can produce
85
+ * a meaningful composite.
86
+ */
87
+ declare function weightedComposite(input: WeightedCompositeInput): WeightedCompositeResult;
88
+ //#endregion
2
89
  //#region src/supervisor-run/types.d.ts
3
90
  /** A metric that could not be computed, with the reason its artifact was missing. */
4
91
  interface Unavailable {
@@ -88,6 +175,12 @@ interface SupervisorRunSources {
88
175
  * settled `verdict` may be a legacy string or `{ valid, score, ... }`.
89
176
  */
90
177
  readonly journal: string | null;
178
+ /**
179
+ * Source-specific reason `journal` is null. The analyzer uses it verbatim as
180
+ * the `unavailable` reason on every journal-dependent metric, so a non-loops
181
+ * layout names its own journal file instead of inheriting the loops paths.
182
+ */
183
+ readonly journalMissingReason?: string;
91
184
  /** Per-brain-call tap (JSONL): finish_reason, completion tokens, requested max tokens. */
92
185
  readonly brainLog: string | null;
93
186
  /** Source-specific reason `brainLog` is absent. */
@@ -189,6 +282,15 @@ interface OrchestrationMetrics {
189
282
  readonly delegationDepth: Measured<number>;
190
283
  readonly timeToFirstSpawnMs: Measured<number>;
191
284
  readonly supervisorWallMs: Measured<number>;
285
+ /**
286
+ * Which measurement `supervisorWallMs` holds — never a silent substitution.
287
+ * `stamps`: explicit start and completion stamps. `journal-span`: the start
288
+ * stamp (or first stamped event) to the last stamped journal event — a
289
+ * lower bound, derived when the store wrote no completion stamp. `idleMs`,
290
+ * `idlePct`, and `workerUtilization` cover the same span. Unavailable
291
+ * exactly when `supervisorWallMs` is, with the same reason.
292
+ */
293
+ readonly supervisorWallSource: Measured<'stamps' | 'journal-span'>;
192
294
  /** Wall time inside the supervisor run with ZERO live workers. */
193
295
  readonly idleMs: Measured<number>;
194
296
  readonly idlePct: Measured<number>;
@@ -251,13 +353,33 @@ interface PerWorkerRow {
251
353
  /** Numeric verdict score exactly as recorded; null means no score was recorded. */
252
354
  readonly score: number | null;
253
355
  }
254
- interface WallDistribution {
255
- readonly n: number;
256
- readonly min: number;
257
- readonly p50: number;
258
- readonly p90: number;
259
- readonly max: number;
260
- readonly sum: number;
356
+ /**
357
+ * `SeriesDistribution` (from `../statistics`) over per-worker wall
358
+ * milliseconds. The fold itself is `summarizeNumberSeries`, exported for any
359
+ * series — fleet wall medians, tokens-per-claim spreads — not only wall.
360
+ */
361
+ type WallDistribution = SeriesDistribution;
362
+ /** One spend measurement and the number of source records behind it. */
363
+ interface SpendMeasurement {
364
+ readonly usd: Measured<number>;
365
+ /** Source records folded into `usd`; 0 when the measurement is unavailable. */
366
+ readonly records: number;
367
+ }
368
+ /**
369
+ * The run's total inference spend, measured two ways.
370
+ *
371
+ * `closeRecord` is the spend the store recorded as settled when the run
372
+ * closed (loops `state.json` `result.spentUsd`; Runtime `result.json`
373
+ * `spentTotal.usd`) — the billing-shaped answer. `journalDerived` is the
374
+ * spend execution observably consumed (journal `metered` + `settled` rows) —
375
+ * the execution-accounting answer. Neither is canonical for the other's
376
+ * question. The two cover different records at different moments, so
377
+ * divergence between them is itself a signal (a dropped settlement, a
378
+ * double meter, spend after the close) — read it, never average it away.
379
+ */
380
+ interface SpendMeasurements {
381
+ readonly journalDerived: SpendMeasurement;
382
+ readonly closeRecord: SpendMeasurement;
261
383
  }
262
384
  interface EconomicsMetrics {
263
385
  /** Driver/brain inference — journal `metered` events. */
@@ -272,6 +394,14 @@ interface EconomicsMetrics {
272
394
  readonly brainTruncations: Measured<number>;
273
395
  /** Worker inference — journal `settled` spend plus the harness session join. */
274
396
  readonly workers: RoleSpend;
397
+ /** Both total-spend measurements, each with its own record count. */
398
+ readonly spend: SpendMeasurements;
399
+ /**
400
+ * One collapsed number kept for existing consumers: the close record when
401
+ * the store wrote one, else the journal-derived sum. `totalUsdSource` names
402
+ * the pick. Prefer `spend` — the collapse hides which accounting question
403
+ * the number answers.
404
+ */
275
405
  readonly totalUsd: Measured<number>;
276
406
  /**
277
407
  * Where `totalUsd` came from. CLI-backend workers never price their own inference into
@@ -343,7 +473,27 @@ interface SupervisorRunRollup {
343
473
  readonly idlePctMean: Measured<number>;
344
474
  readonly workersSpawnedTotal: Measured<number>;
345
475
  readonly acceptedTotal: Measured<number>;
476
+ /**
477
+ * Sum of the per-run collapsed `totalUsd`. Prefer `spendUsd`: this total
478
+ * mixes close-record and journal-derived cells without saying which.
479
+ */
346
480
  readonly usdTotal: Measured<number>;
481
+ /**
482
+ * Fleet spend measured two ways. `runs` is each measurement's own
483
+ * denominator — the cells where that measurement was available. The two
484
+ * sums cover different run sets, so comparing the values without their
485
+ * denominators manufactures a phantom divergence.
486
+ */
487
+ readonly spendUsd: {
488
+ readonly journalDerived: {
489
+ readonly value: Measured<number>;
490
+ readonly runs: number;
491
+ };
492
+ readonly closeRecord: {
493
+ readonly value: Measured<number>;
494
+ readonly runs: number;
495
+ };
496
+ };
347
497
  readonly resolvedCount: Measured<number>;
348
498
  readonly perCell: readonly RollupCellRow[];
349
499
  }
@@ -368,5 +518,5 @@ interface SupervisorRunTreeGap {
368
518
  readonly count?: number;
369
519
  }
370
520
  //#endregion
371
- export { Unavailable as C, showMeasured as D, isUnavailable as E, unavailable as O, SupervisorRunTreeGapCode as S, WorkerLogSource as T, SupervisorRunReport as _, OrchestrationMetrics as a, SupervisorRunTree as b, PerWorkerRow as c, SUPERVISOR_RUN_ROLLUP_SCHEMA as d, SUPERVISOR_RUN_SCHEMA as f, SupervisorRunReader as g, SupervisorRunNodeRole as h, NO_SOURCE_LIMITS as i, RoleSpend as l, SteerBreakdown as m, EconomicsMetrics as n, OutcomeMetrics as o, SourceLimits as p, Measured as r, PatchStats as s, DecisionMetrics as t, RollupCellRow as u, SupervisorRunRollup as v, WallDistribution as w, SupervisorRunTreeGap as x, SupervisorRunSources as y };
372
- //# sourceMappingURL=types-yLK8gXE9.d.ts.map
521
+ export { unavailable as A, weightedComposite as B, SupervisorRunTreeGap as C, WorkerLogSource as D, WallDistribution as E, partialCredit as F, pearsonR as I, ranks as L, WeightedCompositeInput as M, WeightedCompositeResult as N, isUnavailable as O, confidenceInterval as P, spearmanR as R, SupervisorRunTree as S, Unavailable as T, weightedMean as V, SupervisorRunNodeRole as _, OrchestrationMetrics as a, SupervisorRunRollup as b, PerWorkerRow as c, SUPERVISOR_RUN_ROLLUP_SCHEMA as d, SUPERVISOR_RUN_SCHEMA as f, SteerBreakdown as g, SpendMeasurements as h, NO_SOURCE_LIMITS as i, SeriesDistribution as j, showMeasured as k, RoleSpend as l, SpendMeasurement as m, EconomicsMetrics as n, OutcomeMetrics as o, SourceLimits as p, Measured as r, PatchStats as s, DecisionMetrics as t, RollupCellRow as u, SupervisorRunReader as v, SupervisorRunTreeGapCode as w, SupervisorRunSources as x, SupervisorRunReport as y, summarizeNumberSeries as z };
522
+ //# sourceMappingURL=types-I5WwVzQ7.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"types-I5WwVzQ7.d.ts","names":[],"sources":["../src/statistics/descriptive.ts","../src/supervisor-run/types.ts"],"mappings":";;;;;;;iBAQgB,aAAa;EAAU;EAAe;;;;;;;;;;iBAoBtC,mBACd,kBACA,qBACA;EAAQ;EAAe;;EACpB;EAAc;EAAe;;;iBAiClB,cAAc,iBAAiB;;UAM9B;WACN;WACA;WACA;WACA;WACA;WACA;;;;;;;;;iBAUK,sBAAsB,4BAA4B;;;;;iBA8BlD,MAAM;;;;;;iBAmBN,SAAS,aAAa;;;;;iBAuBtB,UAAU,aAAa;UAKtB;;EAEf,MAAM;;;;;EAKN,SAAS;;EAET;;UAGe;EACf;EACA;;;;;;;;;;;iBAYc,kBAAkB,OAAO,yBAAyB;;;;UC5JjD;WACN;;;KAIC,SAAS,KAAK,IAAI;iBAEd,YAAY,iBAAiB;iBAI7B,cAAc,aAAa,KAAK;;iBAKhC,aAAa,GAAG;;KAWpB;;;;;UAMK;;;;;WAKN;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA;WACA;WACA;WACA;;;;;;;;;;;;;UAcM;;WAEN;;WAEA;;WAEA;;WAEA;;WAEA;;;cAIE,kBAAkB;;;;;;;;;;UAiBd;;WAEN;WACA;;WAEA;;WAEA;;;;;;WAMA;;;;;;WAMA;;WAEA;;WAEA;;WAEA;;WAEA;;WAEA,kBAAkB;;WAElB;;WAEA;;;;;;;WAOA;;WAEA;;WAEA;;WAEA;;;;;WAKA;IACP;IACA;IACA;IACA;;IAEA;IACA;;WAEO;;WAEA,QAAQ;;;;;;WAMR;;;;;WAKA;;;;;;UAOM;;WAEN;EACT,QAAQ,QAAQ;;cAOL;cACA;UAEI;;WAEN;WACA;;WAEA;;WAEA;;UAGM;WACN,gBAAgB;WAChB,gBAAgB;WAChB,kBAAkB;;WAElB,QAAQ;WACR,iBAAiB;WACjB,gBAAgB,kBAAkB;;WAElC,kBAAkB;;;;;;WAMlB,OAAO;WACP,WAAW;WACX,gBAAgB;;WAEhB,UAAU;;WAEV,gBAAgB;;WAEhB,iBAAiB;WACjB,oBAAoB;WACpB,kBAAkB;;;;;;;;;WASlB,sBAAsB;;WAEtB,QAAQ;WACR,SAAS;;WAET,mBAAmB;;UAGb;WACN,iBAAiB,SAAS;WAC1B,iBAAiB,SAAS;;WAE1B,UAAU;;WAEV,UAAU;;WAEV,WAAW;;WAEX,oBAAoB;;WAEpB,wBAAwB;;WAExB,eAAe;WACf,qBAAqB;;UAGf;WACN,UAAU;WACV,WAAW;;;;;;WAMX,WAAW;WACX,YAAY;WACZ,KAAK;WACL;;UAGM;;WAEN;WACA;;WAEA,MAAM;;WAEN;;WAEA;;WAEA;;WAEA;;WAEA;WACA;;WAEA;WACA;WACA;WACA;WACA;;WAEA;;;;;;;KAQC,mBAAmB;;UAGd;WACN,KAAK;;WAEL;;;;;;;;;;;;;;UAeM;WACN,gBAAgB;WAChB,aAAa;;UAGP;;WAEN,OAAO;;;;;;;;WAQP,kBAAkB;;WAElB,SAAS;;WAET,OAAO;;;;;;;WAOP,UAAU;;;;;;WAMV;WACA,yBAAyB;WACzB,0BAA0B,SAAS;WACnC,WAAW,kBAAkB;;UAGvB;WACN;WACA;WACA;WACA;;UAGM;WACN,WAAW;WACX,YAAY;WACZ,WAAW;WACX,eAAe;WACf,YAAY;WACZ,aAAa;WACb,YAAY;WACZ,YAAY;WACZ,UAAU;WACV,OAAO,SAAS;;WAEhB;;UAGM;WACN,eAAe;;WAEf;WACA;WACA;WACA,cAAc;WACd,yBAAyB;WACzB;WACA,eAAe;WACf,UAAU;WACV,WAAW;WACX,SAAS;;WAET;;WAEA;;UAGM;WACN;WACA;WACA,QAAQ;WACR,OAAO;WACP,aAAa;WACb,SAAS;WACT,UAAU;WACV,KAAK;;UAGC;WACN,eAAe;WACf;WACA,aAAa;WACb,iBAAiB;WACjB;WACA,WAAW;WACX,mBAAmB;WACnB,iBAAiB;WACjB,aAAa;WACb,qBAAqB;WACrB,eAAe;;;;;WAKf,UAAU;;;;;;;WAOV;aACE;eAA2B,OAAO;eAA2B;;aAC7D;eAAwB,OAAO;eAA2B;;;WAE5D,eAAe;WACf,kBAAkB;;;;;;;;UASZ;WACN;WACA,gBAAgB;;WAEhB,eAAe;;;KAId;UASK;WACN,MAAM;WACN;WACA;WACA"}
@@ -289,7 +289,7 @@ declare function assertCrossFamilyServed(pairs: ReadonlyArray<{
289
289
  //#endregion
290
290
  //#region src/llm-client.d.ts
291
291
  interface LlmMessage {
292
- role: 'system' | 'user' | 'assistant';
292
+ role: 'system' | 'user' | 'assistant' | 'tool';
293
293
  /**
294
294
  * Either a plain text content string OR a multimodal content array
295
295
  * (text + image_url parts) for vision-capable models.
@@ -304,8 +304,35 @@ interface LlmMessage {
304
304
  detail?: 'auto' | 'low' | 'high';
305
305
  };
306
306
  }>;
307
+ /** Tool invocations made by an `assistant` message earlier in the turn. */
308
+ toolCalls?: LlmToolCall[];
309
+ /** The invocation a `tool` message answers. Required when role is `tool`. */
310
+ toolCallId?: string;
307
311
  }
308
312
  type LlmThinkingMode = 'enabled' | 'disabled';
313
+ /** Canonical function-tool definition offered to the model. */
314
+ interface LlmToolDefinition {
315
+ type: 'function';
316
+ function: {
317
+ name: string;
318
+ description?: string;
319
+ parameters: Record<string, unknown>;
320
+ };
321
+ }
322
+ /** Canonical tool-choice policy; meaningful only when `tools` is present. */
323
+ type LlmToolChoice = 'auto' | 'none' | 'required' | {
324
+ type: 'function';
325
+ function: {
326
+ name: string;
327
+ };
328
+ };
329
+ /** One tool invocation on a response or an assistant history message. */
330
+ interface LlmToolCall {
331
+ id: string;
332
+ name: string;
333
+ /** JSON-encoded arguments, exactly as the provider produced them. */
334
+ argumentsJson: string;
335
+ }
309
336
  interface LlmCallRequest {
310
337
  model: string;
311
338
  messages: LlmMessage[];
@@ -316,6 +343,10 @@ interface LlmCallRequest {
316
343
  name: string;
317
344
  schema: Record<string, unknown>;
318
345
  };
346
+ /** Function tools offered to the model for this call. */
347
+ tools?: LlmToolDefinition[];
348
+ /** Tool-choice policy for `tools`. */
349
+ toolChoice?: LlmToolChoice;
319
350
  temperature?: number;
320
351
  maxTokens?: number;
321
352
  /** OpenAI-compatible reasoning mode. Omitted when the provider default should apply. */
@@ -327,7 +358,7 @@ interface LlmCallRequest {
327
358
  * Returns undefined when output or multimodal input is not bounded, causing a
328
359
  * capped CostLedger to reject the call before execution. Pass
329
360
  * `customTokenPricing` when package pricing does not cover the model or endpoint. */
330
- declare function maximumChargeForLlmRequest(request: Pick<LlmCallRequest, 'model' | 'messages' | 'jsonSchema' | 'maxTokens' | 'thinking'>, options?: LlmClientOptions): MaximumCharge | undefined;
361
+ declare function maximumChargeForLlmRequest(request: Pick<LlmCallRequest, 'model' | 'messages' | 'jsonSchema' | 'tools' | 'toolChoice' | 'maxTokens' | 'thinking'>, options?: LlmClientOptions): MaximumCharge | undefined;
331
362
  interface LlmUsage {
332
363
  promptTokens: number;
333
364
  completionTokens: number;
@@ -342,6 +373,8 @@ interface LlmUsage {
342
373
  interface LlmCallResult {
343
374
  /** The text content of the first choice. Empty string if none. */
344
375
  content: string;
376
+ /** Tool invocations from the first choice, when the model called tools. */
377
+ toolCalls?: LlmToolCall[];
345
378
  usage: LlmUsage;
346
379
  /**
347
380
  * Cost in USD. Uses the provider's reported cost when present, otherwise
@@ -848,4 +881,4 @@ interface CheckResult {
848
881
  }
849
882
  //#endregion
850
883
  export { assertServedModels as $, LlmClientOptions as A, isTransientLlmError as B, SandboxSdkTransportOpts as C, LlmCallRequest as D, LlmCallMetadata as E, assertLlmRoute as F, AssertServedModelOptions as G, probeLlm as H, callLlm as I, ServedModelCheck as J, ModelSubstitutionError as K, callLlmJson as L, LlmResponseError as M, LlmRouteRequirements as N, LlmCallResult as O, LlmUsage as P, assertServedModel as Q, costReceiptFromLlm as R, RouterTransportOpts as S, LlmCallError as T, stripFencedJson as U, maximumChargeForLlmRequest as V, AssertCrossFamilyServedOptions as W, ServedModelVerdict as X, ServedModelPolicy as Y, assertCrossFamilyServed as Z, CliBridgeTransportOpts as _, defaultProviderRedactor as _t, JudgeInput as a, assertCrossFamily as at, DirectProviderTransportOpts as b, PersonaConfig as c, FileSystemRawProviderSinkOptions as ct, Scenario as d, NoopRawProviderSink as dt, checkServedModel as et, ChatCallOpts as f, ProviderRedactor as ft, ChatTransport as g, RawProviderSinkFilter as gt, ChatResponse as h, RawProviderSink as ht, JudgeFn as i, JudgeFamily as it, LlmMessage as j, LlmClient as k, ProductClientConfig as l, InMemoryRawProviderSink as lt, ChatRequest as m, RawProviderEvent as mt, CompletionCriterion as n, AssertCrossFamilyOptions as nt, JudgeRubric as o, judgeFamily as ot, ChatClient as p, RawProviderDirection as pt, ServedCrossFamilyError as q, DriverState as r, CrossFamilyError as rt, JudgeScore as s, FileSystemRawProviderSink as st, CheckResult as t, servedModelAcceptable as tt, RouteMap as u, InMemoryRawProviderSinkOptions as ut, CreateChatClientOpts as v, providerFromBaseUrl as vt, createChatClient as w, MockTransportOpts as x, CustomTransportOpts as y, costReceiptFromLlmError as z };
851
- //# sourceMappingURL=types-COAJy8Ku.d.ts.map
884
+ //# sourceMappingURL=types-jUBXJ7Iz.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"types-jUBXJ7Iz.d.ts","names":[],"sources":["../src/trace/raw-provider-sink.ts","../src/judge-families.ts","../src/integrity/served-model.ts","../src/llm-client.ts","../src/analyst/chat-client.ts","../src/types.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;KA8BY;UAEK;;EAEf;;EAEA;EACA;;;;;;EAMA;EACA;;EAEA;;EAEA;;EAEA;EACA,WAAW;;EAEX;;EAEA;EACA;EACA,iBAAiB;EACjB;EACA,kBAAkB;EAClB;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA,YAAY;EACZ;;UAGe;EACf,OAAO,OAAO,mBAAmB;;EAEjC,MAAM,SAAS,wBAAwB,QAAQ;;EAE/C,UAAU;;KAGA,oBAAoB,OAAO,qBAAqB;;;;;;iBAmB5C,wBAAwB,OAAO,mBAAmB;UA8CjD;EACf,WAAW;;cAGA,mCAAmC;UACtC;UACA;EAER,YAAY,OAAM;EAIZ,OAAO,OAAO,mBAAmB;EAIjC,KAAK,SAAQ,wBAA6B,QAAQ;EAUxD;;cAKW,+BAA+B;EACpC,UAAU;;;;;;;EASV,QAAQ,QAAQ;;UAOP;;EAEf;;EAEA;;EAEA;EACA,WAAW;;cAGA,qCAAqC;UACxC;UACA;UACA;UACA;UACA;UACA;UACA;EAER,YAAY,MAAM;UAOJ;UAON;EAKF,OAAO,OAAO,mBAAmB;EAYjC,KAAK,SAAQ,wBAA6B,QAAQ;;;;;;iBAkC1C,oBAAoB;;;;;;;;;;;;;KC5QxB;;;;;;;iBAkEI,YAAY,kBAAkB;UAc7B;;EAEf;;;;EAIA;;cAGW,yBAAyB;WAGlB,UAAU;WACV;EAHlB,YACE,iBACgB,UAAU,eACV;;;;;;;;;;;;;;;iBAoBJ,kBACd,kBACA,OAAM,2BACL;;;;KCnFS;;;;;;;;;;;UAYK;;EAEf;;EAEA;EACA,iBAAiB;;EAEjB,cAAc;EACd,SAAS;;EAET;;;;;;;;;;;;iBA4Cc,iBACd,mBACA,oCACC;cA4CU,+BAA+B;WAGxB,QAAQ,cAAc;EAFxC,YACE,iBACgB,QAAQ,cAAc;;;;;;;;;KAc9B;UAWK;;;;;;EAMf;;;;;EAKA;;EAEA;;;;;;;iBAQc,sBACd,OAAO,kBACP,OAAM;;;;;;iBA6BQ,kBACd,mBACA,mCACA,OAAM,2BACL;;;;;iBAea,mBACd,OAAO;EAAgB;EAAmB;IAC1C,OAAM,2BACL;UAec,uCAAuC;;EAEtD;;EAEA;;cAGW,+BAA+B;WAGxB,UAAU;WACV,QAAQ,cAAc;EAHxC,YACE,iBACgB,UAAU,eACV,QAAQ,cAAc;;;;;;;;;;;iBAgB1B,wBACd,OAAO;EAAgB;EAAmB;IAC1C,OAAM,iCACL;;;UCnQc;EACf;;;;;EAKA,kBAEI;IACM;IAAc;;IACd;IAAmB;MAAa;MAAa;;;;EAGvD,YAAY;;EAEZ;;KAGU;;UAGK;EACf;EACA;IACE;IACA;IACA,YAAY;;;;KAKJ;EAIN;EAAkB;IAAY;;;;UAGnB;EACf;EACA;;EAEA;;UAGe;EACf;EACA,UAAU;;EAEV;;EAEA;IAAe;IAAc,QAAQ;;;EAErC,QAAQ;;EAER,aAAa;EACb;EACA;;EAEA,WAAW;;EAEX;;;;;;iBAOc,2BACd,SAAS,KACP,0GAGF,UAAS,mBACR;UAgCc;EACf;EACA;EACA;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf;;EAEA,YAAY;EACZ,OAAO;;;;;EAKP;;;;;;;;;EASA;;;;;;;;;;;;EAYA;;EAEA;;;;;;;;;EASA;;;;;;;EAOA;;EAEA,KAAK;;KAGK,kBAAkB,KAAK;;;;;;iBAOnB,mBACd,QAAQ,eACR,qBAAqB,qBACpB;;iBA6Ba,wBACd,OAAO,OACP,qBAAqB,qBACpB;cAMU,qBAAqB;WAGd;WACA;WACA;EAJlB,YACE,iBACgB,gBACA,cACA;;;;;cASP,yBAAyB;WAGlB,QAAQ;EAF1B,YACE,iBACgB,QAAQ,eACxB;IAAY;;;UAMC;;EAEf;;EAEA;EACA;;EAEA;IAAe;IAAc;;;EAE7B;;EAEA;;;;;;;EAOA,SAAS;;;;;;;;EAQT;;EAEA;;EAEA,qBAAqB;;;;;;;EAOrB;;;;;;EAMA;;EAEA,WAAW;;EAEX,eAAe;;;;;;;;EAQf,UAAU;;;;;EAKV;;EAEA;IAAiB;IAAgB;;;EAEjC,WAAW;;;;;;;;;;EAUX,8BAA8B;;;;;;;;;;;;;iBAyEhB,oBAAoB;;;;;;iBAqMpB,gBAAgB;;;;;;iBA6EV,QACpB,KAAK,gBACL,OAAM,mBACL,QAAQ;;;;;;;iBAyVW,YAAY,aAChC,KAAK,gBACL,OAAM,mBACL;EAAU,OAAO;EAAG,QAAQ;;UA6Ed;;;;;;;EAOf;;;;;EAKA,kBAAkB,eAAe;;EAEjC,kBAAkB,eAAe;;EAEjC;;;;;EAKA;;;;;;;;;;;iBAYc,eAAe,MAAM,kBAAkB,MAAK;;;;;;;;;;;;;;;;;;iBA6EtC,SACpB,eACA,OAAM;EAAqB;IAC1B;EACD;EACA;EACA;;EAEA;;EAEA;;;;;;;cAoCW;WACF;mBACQ;EAEjB,YAAY,OAAM;EAKlB,KAAK,KAAK,gBAAgB,MAAM,mBAAmB,QAAQ;EAK3D,SAAS,aACP,KAAK,gBACL,MAAM,mBACL;IAAU,OAAO;IAAG,QAAQ;;;;;;;;UC7wChB;;WAEN,WAAW;;WAEX;;WAEA;;EAGT,KAAK,KAAK,aAAa,OAAO,eAAe,QAAQ;;KAG3C;UAQK,oBAAoB,KAAK;;EAExC;;KAGU,eAAe;UAEV;;EAEf,SAAS;;EAET;;EAEA;;EAEA;;KAKU,uBACR,sBACA,yBACA,8BACA,0BACA,sBACA;UAEM;EACR;;EAEA;;UAGe,4BAA4B;EAC3C;EACA;EACA;;UAGe,+BAA+B;EAC9C;EACA;EACA;;UAGe,oCAAoC;EACnD;EACA;EACA;;;;;;UAOe,gCAAgC;EAC/C;EACA,OAAO,KAAK,aAAa,OAAO,iBAAiB,QAAQ;;;UAI1C,4BAA4B;EAC3C;EACA,OAAO,KAAK,aAAa,OAAO,iBAAiB,QAAQ;;;;;;UAO1C,0BAA0B;EACzC;EACA,UAAU,KAAK,aAAa,OAAO,iBAAiB,QAAQ;;;;;;iBAO9C,iBAAiB,MAAM,uBAAuB;;;UCjH7C;EACf;EACA;EACA;EACA;EACA;EACA,OAAO;EACP,gBAAgB;EAChB;;UAGe;EACf;EACA;EACA;EACA;;UAKe;EACf;EAQA;EACA;EACA;EACA;;UAKe;EACf;EACA;EACA,YAAY;;UAGG;EACf;EACA;EACA;EACA;EACA;;UAmBe;EACf;EACA;EACA;EACA;EACA;IAAmB;IAAc;;EACjC;EACA;;UASe;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;IAAc;IAAc;;EAC5B;IAAmB;IAAc,QAAQ;;EACzC;IAAc;IAAkB;;EAChC;;UAKe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;GACC;;UAGc;EACf;EACA,QAAQ;;EAER;;UAyBe;EACf;EACA,QAAQ,OAAO;EACf,YAAY,OAAO;;;;;;;;;;KAWT;UAEK;EACf;EACA;EACA;EACA,oBAAoB;EACpB;;EAEA,QAAQ;;;;;;;EAOR;;;;;;;EAOA;;;;;;EAMA;;UAGe;EACf;EACA;EACA;IAAa;IAAiB;IAAkB;;EAChD;EACA;EACA;;UAGe;EACf,UAAU;EACV,OAAO;EACP,WAAW;;EAEX,aAAa;EACb;EACA,WAAW;EACX,SAAS;;KAGC,WAAW,MAAM,YAAY,OAAO,eAAe,QAAQ;UAYtD;EACf;EACA;EACA;EACA"}
@@ -1,7 +1,7 @@
1
1
  import { c as CostLedgerHandle } from "../cost-ledger-DbQdN3nO.js";
2
- import { A as LlmClientOptions, N as LlmRouteRequirements } from "../types-COAJy8Ku.js";
2
+ import { A as LlmClientOptions, N as LlmRouteRequirements } from "../types-jUBXJ7Iz.js";
3
3
  import { s as TraceStore } from "../store-CT9YIIve.js";
4
- import { m as FeedbackTrajectoryStore } from "../feedback-trajectory-Cc9M2tNF.js";
4
+ import { m as FeedbackTrajectoryStore } from "../feedback-trajectory-DpTTjo0q.js";
5
5
  import { z } from "zod";
6
6
  import { ServerType } from "@hono/node-server";
7
7
  import { Hono } from "hono";
@@ -1,2 +1,2 @@
1
- import { A as TraceEventSchema, C as HealthResponseSchema, D as RubricDimensionSchema, E as ListRubricsResponseSchema, F as hashRubric, M as TracesIngestResponseSchema, N as VersionResponseSchema, O as RubricInfoSchema, P as WIRE_VERSION, S as FeedbackTrajectorySchema, T as JudgeResultSchema, _ as ErrorResponseSchema, a as runRpcBatch, b as FeedbackIngestResponseSchema, c as WireError, d as handleListRubrics, f as handleTracesIngest, g as listBuiltinRubrics, h as getBuiltinRubric, i as dispatchRpc, j as TracesIngestRequestSchema, k as RubricSchema, l as handleFeedbackIngest, m as BUILTIN_RUBRICS, n as startServer, o as runRpcOnce, p as handleVersion, r as startServerAsync, s as buildOpenApi, t as createApp, u as handleJudge, v as FailureModeSchema, w as JudgeRequestSchema, x as FeedbackLabelSchema, y as FeedbackAttemptSchema } from "../server-DNiUVirV.js";
1
+ import { A as TraceEventSchema, C as HealthResponseSchema, D as RubricDimensionSchema, E as ListRubricsResponseSchema, F as hashRubric, M as TracesIngestResponseSchema, N as VersionResponseSchema, O as RubricInfoSchema, P as WIRE_VERSION, S as FeedbackTrajectorySchema, T as JudgeResultSchema, _ as ErrorResponseSchema, a as runRpcBatch, b as FeedbackIngestResponseSchema, c as WireError, d as handleListRubrics, f as handleTracesIngest, g as listBuiltinRubrics, h as getBuiltinRubric, i as dispatchRpc, j as TracesIngestRequestSchema, k as RubricSchema, l as handleFeedbackIngest, m as BUILTIN_RUBRICS, n as startServer, o as runRpcOnce, p as handleVersion, r as startServerAsync, s as buildOpenApi, t as createApp, u as handleJudge, v as FailureModeSchema, w as JudgeRequestSchema, x as FeedbackLabelSchema, y as FeedbackAttemptSchema } from "../server-dIWwF3j_.js";
2
2
  export { BUILTIN_RUBRICS, ErrorResponseSchema, FailureModeSchema, FeedbackAttemptSchema, FeedbackIngestResponseSchema, FeedbackLabelSchema, FeedbackTrajectorySchema, HealthResponseSchema, JudgeRequestSchema, JudgeResultSchema, ListRubricsResponseSchema, RubricDimensionSchema, RubricInfoSchema, RubricSchema, TraceEventSchema, TracesIngestRequestSchema, TracesIngestResponseSchema, VersionResponseSchema, WIRE_VERSION, WireError, buildOpenApi, createApp, dispatchRpc, getBuiltinRubric, handleFeedbackIngest, handleJudge, handleListRubrics, handleTracesIngest, handleVersion, hashRubric, listBuiltinRubrics, runRpcBatch, runRpcOnce, startServer, startServerAsync };
@@ -41,26 +41,13 @@ it*. Unified at the trace level, you see both as one timeline per cell.
41
41
  - Compose: register TraceAI's instrumentations on the global tracer
42
42
  provider, then either point both at your OTLP collector or at
43
43
  TraceAI's hosted backend if you want their UI.
44
- - **A bridge exists in source but is not published:** `createOtelBridge`
45
- (`src/adapters/otel.ts`) converts finished OTel spans (`ReadableSpan`
46
- shape) and forwards them into the hosted-tier ingest, lifting
47
- `tangle.runId` / `tangle.scenarioId` / `tangle.cellId` /
48
- `tangle.generation` to first-class wire fields so the dashboard pivots
49
- correctly. As of 0.140.x there is no `./adapters/otel` entry in
50
- `package.json` `exports`, so it is not importable from the published
51
- package — the snippet below documents the design; it will throw
52
- `ERR_PACKAGE_PATH_NOT_EXPORTED` against `@tangle-network/agent-eval`
53
- installed from npm.
54
- ```ts
55
- // (not currently published — see status note above)
56
- import { createHostedClient } from '@tangle-network/agent-eval/hosted'
57
- import { createOtelBridge } from '@tangle-network/agent-eval/adapters/otel'
58
-
59
- const client = createHostedClient({ endpoint, apiKey, tenantId })
60
- const bridge = createOtelBridge({ client, defaultRunId: substrateRunId })
61
- processor.onEnd = (span) => { void bridge.ingest([span]) }
62
- // ...or call `bridge.ingest(batch)` from a SpanProcessor.onShutdown.
63
- ```
44
+ - **No OTel-span-to-hosted-ingest bridge ships.** To land finished OTel
45
+ spans in the hosted tier, write your own mapping from the span shape to
46
+ `TraceEvent` rows and post them through `createHostedClient` from
47
+ `@tangle-network/agent-eval/hosted` or the `/v1/traces/ingest` wire
48
+ route ([wire-protocol.md](./wire-protocol.md#tracesingest-batch-ingest-production-trace-events)).
49
+ For run records that already exist, `fromOtelSpans` from `/contract`
50
+ converts collector output into `RunRecord[]` for `analyzeRuns()`.
64
51
 
65
52
  ### Langfuse SDK
66
53
 
@@ -131,9 +118,8 @@ OTel-protocol composition:
131
118
  1. **Cost-aware judging.** Your observability tool's auto-instrumented
132
119
  spans carry token counts + cost. A custom `JudgeConfig` can read
133
120
  them via the OTel context and refuse to score artifacts that
134
- exceeded a per-call budget. Easy to write yourself; we'll ship a
135
- reference helper (`costAwareJudgeFromOtel`) when a partner pulls on
136
- this.
121
+ exceeded a per-call budget. Easy to write yourself; no reference
122
+ helper ships today.
137
123
  2. **Tool-aware judging.** Your instrumentation captures the tool-call
138
124
  sequence (`langchain.tool.invoked`, `openai.function.called`, etc.).
139
125
  A judge that scores "did the agent use the right tool" reads those
@@ -332,16 +332,28 @@ The credential stays inside the owner closure; the proxy still enforces every bu
332
332
  ### Metered agent CLI engines
333
333
 
334
334
  The `autoresearch` and `meta_harness` engines drive a `claude` CLI subprocess.
335
+ They ship only in the tested official source revision, not in the published `gepa` package (see [Install Official GEPA](#install-official-gepa)).
335
336
  Set `optimizer.anthropicEndpoint: true` to admit them in proxied mode.
337
+ This path is measured live: a real `claude` CLI session completes with every tool call translated and every call metered.
336
338
  The loopback proxy then also serves `POST /v1/messages` (Anthropic Messages API) and the bridge child receives `ANTHROPIC_BASE_URL`, an ephemeral `ANTHROPIC_AUTH_TOKEN`, and `ANTHROPIC_MODEL` in its environment.
337
339
  Every CLI call becomes one canonical execution-owner call with the same reservation, receipt, and budget pipeline as reflection traffic; the run fails if the receipt count differs from the admitted call count.
338
340
  Each agent engine run must set `engineConfig.model` to `optimizer.model`, because the engines pass `--model` and that flag beats the injected environment.
339
- The endpoint translates text conversations only; it refuses tool use, thinking, images, `top_p`, `top_k`, and `stop_sequences` with a loud Anthropic error envelope.
341
+ The endpoint translates text and tool-use conversations.
342
+ Anthropic `tools`, `tool_choice`, `tool_use`, and `tool_result` map onto the canonical execution-owner contract, and a tool-calling response is synthesized back as the Anthropic stream shape the CLI expects.
343
+ System text translates from both slots the CLI uses: the top-level `system` field and system-role turns injected inside `messages`.
344
+ Claude-specific control fields (`thinking`, `context_management`, `output_config`) carry no token-billing semantics on the owner wire; the shim strips them and records the names in the ledger tag `strippedFields`.
345
+ It still refuses images, server tools, `top_p`, `top_k`, and `stop_sequences` with a loud Anthropic error envelope.
340
346
  A budget refusal surfaces to the CLI as HTTP 402, which the CLI treats as terminal instead of retrying.
341
347
  Agent sessions are chatty: size `budget.maxRequests` for tens of calls per engine run.
342
348
  Without the flag, agent engines stay rejected in proxied mode.
343
349
 
344
- Other official engines can still receive their own settings:
350
+ **The `-inf` trap.**
351
+ GEPA scores an agent-engine candidate as one aggregate evaluation over the whole train set.
352
+ That one registering evaluation costs `trainSet.length` callback evaluations against `maxEvaluations`.
353
+ When `maxEvaluations` is below the train-set size, the callback rejects mid-aggregate and GEPA records the candidate score as `-inf`.
354
+ Set `maxEvaluations` to at least the train-set size for every registering evaluation you expect.
355
+
356
+ Without `optimizer`, an engine runs unproxied and can receive its own settings:
345
357
 
346
358
  ```ts
347
359
  recipe: {
@@ -357,7 +369,7 @@ recipe: {
357
369
  }
358
370
  ```
359
371
 
360
- Their external model spend remains incomplete unless that engine reports it.
372
+ Its external model spend remains incomplete unless that engine reports it.
361
373
  Supply provider API keys only through `runner.env`.
362
374
  An exported shell variable never reaches the bridge child.
363
375
  The spawn builds the child environment from a fixed allowlist of benign variables (PATH, HOME, locale, `PYTHONPATH`) plus `runner.env`, so the parent environment is stripped by construction.
@@ -387,6 +399,9 @@ The table names the failure so you can set the knob before the run dies mid-spen
387
399
  | `reflection_lm_kwargs.num_retries` | litellm default (3) | Each failed reflection request retries 3 times inside litellm, so the proxy meters 4 request attempts per logical call and `budget.maxRequests` exhausts 4x early. Set `num_retries: 0`; the proxy already accounts each attempt. | `recipe.run.engineConfig.reflection.reflection_lm_kwargs` |
388
400
  | `reflection_lm_kwargs.max_tokens` | `budget.maxOutputTokensPerRequest` | Every reflection request ships the full budget cap as `max_tokens`. A provider family with a lower completion cap rejects every call. A reasoning model also needs headroom for hidden reasoning tokens. Set a value at or below the smallest family cap; it must not exceed `budget.maxOutputTokensPerRequest`. | `recipe.run.engineConfig.reflection.reflection_lm_kwargs` |
389
401
  | `maxProposerCostUsd` | unset | Without it, one engine stage can spend up to `optimizer.budget.maxCostUsd` or the campaign `costCeiling` before any limit fires. Supply it only when the execution owner can enforce billed USD. | `recipe.run.maxProposerCostUsd` |
402
+ | `maxEvaluations` (agent engines) | required, no default | An agent engine registers one aggregate evaluation that costs the full train set of callback evaluations. A value below the train-set size rejects mid-aggregate and GEPA records the candidate as `-inf`. | `recipe.run.maxEvaluations` |
403
+ | `budget.maxRequests` (agent engines) | required, no default | An agent CLI session makes tens of calls per engine run. A text-campaign-sized limit exhausts mid-run, and the CLI sees a terminal 402. | `optimizer.budget.maxRequests` |
404
+ | `expectUsage` | `'assert'` | A deterministic evaluator that makes no LLM calls records zero usage, so `'assert'` fails the run as a stub. Set `'off'` only for an evaluator with no paid calls. | `selfImprove({ expectUsage })` |
390
405
 
391
406
  ## Install Official SkillOpt
392
407
 
package/docs/concepts.md CHANGED
@@ -275,10 +275,9 @@ That drop is the signal.
275
275
 
276
276
  Every reported interval is a bootstrap 95 % interval: the statistic is recomputed on many resamples of the data, and the middle 95 % of those values is the interval.
277
277
 
278
- Three bias probes cover three separate failure modes.
279
- `positionalBias` finds a judge that scores by position.
280
- `verbosityBias` finds one that scores by length.
281
- `selfPreference` finds one that prefers output from its own model family.
278
+ `verbosityBias` is the one exported bias probe: it finds a judge that rewards length regardless of quality.
279
+ The `JudgeInsight` report shape also carries optional `positionalBias` and `selfPreference` fields for caller-computed probes.
280
+ No built-in computes those two fields.
282
281
 
283
282
  ## Trace Model
284
283
 
@@ -100,7 +100,7 @@ GET /v1/version
100
100
  ```json
101
101
  {
102
102
  "package": "@tangle-network/agent-eval",
103
- "version": "0.140.1",
103
+ "version": "0.150.1",
104
104
  "wireVersion": "1.0.0",
105
105
  "apiSurface": ["judge", "listRubrics", "version", "feedback.ingest", "traces.ingest"]
106
106
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-eval",
3
- "version": "0.150.0",
3
+ "version": "0.150.2",
4
4
  "description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
5
5
  "homepage": "https://github.com/tangle-network/agent-eval#readme",
6
6
  "repository": {