@tangle-network/agent-eval 0.133.2 → 0.134.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/CHANGELOG.md +218 -0
  2. package/dist/analyst/index.d.ts +11 -35
  3. package/dist/analyst/index.d.ts.map +1 -1
  4. package/dist/analyst/index.js +4 -53
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/{analyze-runs-BmX-h_yn.d.ts → analyze-runs-DMo3Lb_y.d.ts} +4 -4
  7. package/dist/{analyze-runs-BmX-h_yn.d.ts.map → analyze-runs-DMo3Lb_y.d.ts.map} +1 -1
  8. package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
  9. package/dist/analyze-runs-qk8op0tN.js.map +1 -0
  10. package/dist/baseline-BaPxoROc.js +149 -0
  11. package/dist/baseline-BaPxoROc.js.map +1 -0
  12. package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
  13. package/dist/baseline-D_fT6277.d.ts.map +1 -0
  14. package/dist/benchmarks/index.d.ts +1 -1
  15. package/dist/benchmarks/index.js +1 -1
  16. package/dist/{benchmarks-CJr1H1_a.js → benchmarks-v5piCeDl.js} +3 -3
  17. package/dist/{benchmarks-CJr1H1_a.js.map → benchmarks-v5piCeDl.js.map} +1 -1
  18. package/dist/builder-eval/index.js +1 -1
  19. package/dist/campaign/index.d.ts +5 -4
  20. package/dist/campaign/index.js +4 -3
  21. package/dist/{campaign-BJjn1rhw.js → campaign-DEC_7DLn.js} +12 -6
  22. package/dist/{campaign-BJjn1rhw.js.map → campaign-DEC_7DLn.js.map} +1 -1
  23. package/dist/{client-COvaLoQG.d.ts → client-BIyh1RCr.d.ts} +29 -15
  24. package/dist/client-BIyh1RCr.d.ts.map +1 -0
  25. package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
  26. package/dist/client-LIuo-KPv.js.map +1 -0
  27. package/dist/contract/index.d.ts +12 -11
  28. package/dist/contract/index.d.ts.map +1 -1
  29. package/dist/contract/index.js +12 -18
  30. package/dist/contract/index.js.map +1 -1
  31. package/dist/{default-registry-Cl3pHo4n.d.ts → default-registry-Brxr728w.d.ts} +4 -268
  32. package/dist/default-registry-Brxr728w.d.ts.map +1 -0
  33. package/dist/{default-registry-D3T9XbuY.js → default-registry-IjYs7T8l.js} +4 -61
  34. package/dist/default-registry-IjYs7T8l.js.map +1 -0
  35. package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
  36. package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
  37. package/dist/hosted/index.d.ts +2 -2
  38. package/dist/hosted/index.d.ts.map +1 -1
  39. package/dist/hosted/index.js +1 -1
  40. package/dist/{index-C7Wue8R6.d.ts → index-BoJNQR6n.d.ts} +29 -11
  41. package/dist/index-BoJNQR6n.d.ts.map +1 -0
  42. package/dist/{index-BREtv3ZZ.d.ts → index-C21xKtxu.d.ts} +4 -4
  43. package/dist/{index-BREtv3ZZ.d.ts.map → index-C21xKtxu.d.ts.map} +1 -1
  44. package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
  45. package/dist/index-DSC51roc2.d.ts.map +1 -0
  46. package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
  47. package/dist/index-DuhJaaiH.d.ts.map +1 -0
  48. package/dist/index.d.ts +60 -13
  49. package/dist/index.d.ts.map +1 -1
  50. package/dist/index.js +134 -27
  51. package/dist/index.js.map +1 -1
  52. package/dist/ledger-core/index.d.ts +2 -2
  53. package/dist/ledger-core/index.js +2 -2
  54. package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
  55. package/dist/ledger-core-DAKFKRzi.js.map +1 -0
  56. package/dist/matrix/index.d.ts +1 -1
  57. package/dist/meta-eval/index.d.ts +1 -1
  58. package/dist/meta-eval/index.js +2 -2
  59. package/dist/multishot/index.d.ts +2 -2
  60. package/dist/openapi.json +1 -1
  61. package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
  62. package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
  63. package/dist/pipelines/index.d.ts +1 -1
  64. package/dist/pipelines/index.js +3 -2
  65. package/dist/pipelines/index.js.map +1 -1
  66. package/dist/proposal-findings-DCawte-y.js +164 -0
  67. package/dist/proposal-findings-DCawte-y.js.map +1 -0
  68. package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
  69. package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
  70. package/dist/{release-report-CjHWa8Ia.d.ts → release-report-CuULWKyk.d.ts} +2 -2
  71. package/dist/{release-report-CjHWa8Ia.d.ts.map → release-report-CuULWKyk.d.ts.map} +1 -1
  72. package/dist/reporting.d.ts +3 -3
  73. package/dist/reporting.js +4 -4
  74. package/dist/{researcher-CbSKhK8z.d.ts → researcher-DVtruQ9U.d.ts} +2 -2
  75. package/dist/{researcher-CbSKhK8z.d.ts.map → researcher-DVtruQ9U.d.ts.map} +1 -1
  76. package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
  77. package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
  78. package/dist/rl.d.ts +2 -2
  79. package/dist/rl.d.ts.map +1 -1
  80. package/dist/rl.js +18 -5
  81. package/dist/rl.js.map +1 -1
  82. package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
  83. package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
  84. package/dist/{semantic-concept-judge-BypLt6Fw.js → semantic-concept-judge-C0P1VTXD.js} +2 -3
  85. package/dist/{semantic-concept-judge-BypLt6Fw.js.map → semantic-concept-judge-C0P1VTXD.js.map} +1 -1
  86. package/dist/{skill-usage-BaaxFSJR.d.ts → skill-usage-BDQVPIG1.d.ts} +3 -2
  87. package/dist/skill-usage-BDQVPIG1.d.ts.map +1 -0
  88. package/dist/{skillopt-optimization-method-CF6a327Q.js → skillopt-optimization-method-BY6vKLJB.js} +169 -53
  89. package/dist/skillopt-optimization-method-BY6vKLJB.js.map +1 -0
  90. package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts → skillopt-optimization-method-DJ3l4w8W.d.ts} +20 -18
  91. package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +1 -0
  92. package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
  93. package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
  94. package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
  95. package/dist/statistics-RwRNu2__.js.map +1 -0
  96. package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
  97. package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
  98. package/dist/{summary-report-CFnQgNfg.d.ts → summary-report-DGp0-_XO.d.ts} +61 -4
  99. package/dist/summary-report-DGp0-_XO.d.ts.map +1 -0
  100. package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
  101. package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
  102. package/dist/types-DVjczBM9.d.ts +276 -0
  103. package/dist/types-DVjczBM9.d.ts.map +1 -0
  104. package/dist/{types-BokuXvOG.d.ts → types-DiWLru6Z.d.ts} +20 -37
  105. package/dist/types-DiWLru6Z.d.ts.map +1 -0
  106. package/docs/campaign-proposers.md +5 -0
  107. package/docs/design/statistics-decisions.md +271 -0
  108. package/docs/design.md +1 -0
  109. package/docs/insight-report.md +1 -1
  110. package/docs/research-report-methodology.md +4 -1
  111. package/package.json +2 -1
  112. package/dist/analyze-runs-B-afTpCv.js.map +0 -1
  113. package/dist/baseline-DcX5hQDv.js.map +0 -1
  114. package/dist/baseline-hG3K85h4.d.ts.map +0 -1
  115. package/dist/client-COvaLoQG.d.ts.map +0 -1
  116. package/dist/client-CYzbdJOZ.js.map +0 -1
  117. package/dist/default-registry-Cl3pHo4n.d.ts.map +0 -1
  118. package/dist/default-registry-D3T9XbuY.js.map +0 -1
  119. package/dist/index-C7Wue8R6.d.ts.map +0 -1
  120. package/dist/index-DSC51roc.d.ts.map +0 -1
  121. package/dist/index-nhIYz9hn.d.ts.map +0 -1
  122. package/dist/ledger-core-CPZfcrC2.js.map +0 -1
  123. package/dist/run-score-iEEAWiBY.js +0 -41
  124. package/dist/run-score-iEEAWiBY.js.map +0 -1
  125. package/dist/skill-usage-BaaxFSJR.d.ts.map +0 -1
  126. package/dist/skillopt-optimization-method-CF6a327Q.js.map +0 -1
  127. package/dist/skillopt-optimization-method-wHF5xsUv.d.ts.map +0 -1
  128. package/dist/statistics-DWM_AyLe.js.map +0 -1
  129. package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
  130. package/dist/summary-report-CFnQgNfg.d.ts.map +0 -1
  131. package/dist/types-BokuXvOG.d.ts.map +0 -1
package/CHANGELOG.md CHANGED
@@ -4,6 +4,224 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
4
4
 
5
5
  ---
6
6
 
7
+ ## [0.134.0] - 2026-07-28 - isolated proposal inputs
8
+
9
+ ### Changed
10
+
11
+ - `runOptimization()` now accepts only findings labeled from search runs or observed production behavior.
12
+ - `ProposalFinding` carries the required `proposal_origin`.
13
+ - Removed opaque `report` and capture-store access from `ProposeContext`; final evaluation data has no internal path into candidate generation.
14
+ - Removed `assertNoJudgeVerdict`, `isJudgeVerdict`, and `isTraceObservable`; use validated `ProposalFinding` inputs instead.
15
+ - `runOptimization()` snapshots its baseline and candidate outputs before measurement.
16
+
17
+ ## [0.133.3] - 2026-07-27 - trustworthy statistical decisions
18
+
19
+ ### Consumer notice — reported p-values were too small in every release from 0.1.0 to 0.133.0
20
+
21
+ 0.133.1 corrected the standard-normal CDF. This release carries the rest, and
22
+ restates the notice because the re-check bands and the affected version range are
23
+ what a consumer actually needs.
24
+
25
+ A standard-normal CDF mixed the arguments of the Abramowitz–Stegun error-function
26
+ approximation, giving up to `3.7189e-2` absolute CDF error where a correct
27
+ implementation is bounded by `7.5e-8`. Every p-value routed through it was too
28
+ small by 26–36 % relative, so the module's real type-I error rate was **6.53 % at
29
+ a nominal 5 %** and **1.34 % at a nominal 1 %**. The defect entered at the initial
30
+ commit (`7d5032b`, 2026-04-20) and shipped in every release through 0.133.0.
31
+
32
+ **Affected:** `mannWhitneyU`, `wilcoxonSignedRank`, `pairedTTest`, `welchsTTest`,
33
+ `compareToBaseline`, `mcnemarPower`.
34
+
35
+ **Unaffected** (verified numerically identical before and after, because they route
36
+ through the inverse normal rather than the forward CDF): `requiredSampleSize`,
37
+ `requiredPairedSampleSize`, `pairedMde`, `mcnemarRequiredN`, `mcnemar`,
38
+ `pairedSignTest`, `wilson`, `passAtK`, `corpusInterRaterAgreement`, `eProcess`,
39
+ `holm`, `ranks`, `pearsonR`, `spearmanR`, `cliffsDelta`, `pairedCohensDz`.
40
+
41
+ **How to re-check a decision you already made.** The error is monotone in `|z|`, so
42
+ the affected band is exact and narrow:
43
+
44
+ - Any recorded p in `[0.038053, 0.050000)` crossed a 5 % gate it should not have.
45
+ - At `α = 0.01` the band is `[0.007443, 0.010000)`; at `α = 0.10`, `[0.077398, 0.100000)`.
46
+ - A recorded p below `0.038053` was significant either way; at or above `0.05`, not
47
+ significant either way. Neither needs re-checking.
48
+
49
+ Three further cautions, independent of the CDF:
50
+
51
+ - A `wilcoxonSignedRank` leg that reported `p = 1` on fewer than six non-zero
52
+ differences measured nothing — the function hard-returned `p = 1` there with no
53
+ flag. A clean 5-of-5 shift reported `1.0` where the exact answer is `0.0625`.
54
+ Exact ties are dropped before ranking, so ten pairs with five tied deltas also
55
+ fell into that branch. Re-run those on this release.
56
+ - A promotion that turned on a `pairedBootstrap` `low > 0` check below 20 pairs was
57
+ never valid at the stated confidence: measured false-positive rate is 13.53 % at
58
+ `n = 3` against a nominal 2.5 %. `gateEligible` now reports this.
59
+ - A bootstrap interval recorded through `analyze-runs.ts` at or before 0.133.0 is
60
+ not reproducible — that call site passed no seed and `makeRng` fell back to
61
+ `Math.random`.
62
+
63
+ Full evidence, per-statistic verdicts, and the dependency argument:
64
+ [`docs/design/statistics-decisions.md`](./docs/design/statistics-decisions.md).
65
+
66
+ ### Consumer notice — `HeldOutGate` promoted candidates that scored nothing on most held-out items
67
+
68
+ Every release through 0.133.2 decided a promotion over the items where BOTH arms
69
+ produced a finite score, and dropped the rest without a word. An item the candidate
70
+ crashed on, timed out on, or wrote no row for simply left the comparison.
71
+
72
+ **Measured, deterministic fixture, no model calls:** 26 held-out items, a candidate
73
+ that produced no score at all on 20 of them and 0.95 on the 6 it answered against a
74
+ 0.60 baseline. Published 0.133.2 and `origin/main` PROMOTE it at every threshold
75
+ from −0.05 to +0.30 — `productiveRuns: 6`, `unpairedBaselineRuns: 20` sitting in the
76
+ evidence, read by nothing. The control, the same 20 failures scored as the 0 they
77
+ earned, is correctly refused at a mean paired delta of −0.3808. An agent that failed
78
+ 77 % of its tasks was promoted, and the paired-delta, overfit-gap and cost gates all
79
+ sat behind that filter.
80
+
81
+ A second shape, same cause: a crashed first attempt plus a scored retry at the same
82
+ `(experimentId, scenarioId, seed)` PROMOTED on 0.133.2, even though the gate's own
83
+ docstring says duplicate identities throw — the crashed row was filtered out before
84
+ the duplicate could be seen. It now throws, exactly as two scored rows already did.
85
+
86
+ **How to re-check a decision you already made.** Read `unpairedBaselineRuns` and
87
+ `unpairedCandidateRuns` on any recorded `GateDecision`: a nonzero value means the
88
+ verdict was computed over a subset. `productiveRuns` below the number of held-out
89
+ items you dealt means the same thing. Those promotions are not valid at the stated
90
+ threshold and should be re-run on this release.
91
+
92
+ ### Fixed
93
+
94
+ - `HeldOutGate`'s cost median is taken over the rows that DECIDED the verdict — the
95
+ matched pairs on both splits — instead of over every row the caller passed. The old
96
+ population was a denominator nobody measured and was trivially movable: 48 rows tagged
97
+ `dev` at \$0.0001, which the gate never scores, drag a real \$5.00/task candidate to a
98
+ reported \$0.0001 and clear a \$1.00 `costPerTaskCeiling`. Measured on `origin/main`
99
+ (2789970): 12 fully-covered items at \$5.00/task, ceiling \$1.00 — 0 pad rows rejects
100
+ with `cost_ceiling`, 24 pad rows reports \$2.50005, 48 pad rows reports \$0.0001 and
101
+ PROMOTES. The population is derived from the pairing rather than from a list of split
102
+ tags, so there is no tag that sits outside the rule. On a comparison with no rows
103
+ outside the two decided splits the reported number is unchanged.
104
+ - `HeldOutGate` requires COVERAGE before it decides anything: on both the search and
105
+ the holdout split, `answered / dealt` must be at least the new `minCoverage`
106
+ (default **1** — every item the comparison was dealt carries a real score on both
107
+ arms), else it refuses with the new `incomplete_coverage` rejection code. The
108
+ denominator is measured, not declared: it is what `pairRunRecords` reports when
109
+ given every row of a split rather than only the scored ones, so an item counts as
110
+ dealt because a row for it exists on at least one arm. The gate does not impute a
111
+ value for a missing score — it does not know the failure value of the caller's
112
+ metric, and a caller who does knows to write it onto the record before calling.
113
+ `GateEvidence` gains `holdoutCoverage` and `searchCoverage` (`SplitCoverage`:
114
+ `dealt`, `answered`, `unscoredPairs`, `candidateOnly`, `baselineOnly`, `coverage`),
115
+ so a shrunken denominator can never be read without seeing it.
116
+ Verified monotone against `origin/main` over 6000 randomised comparisons: on
117
+ complete inputs the verdict, rejection code, CI and n are identical in 3000/3000
118
+ cases; across both sweeps there are 0 cases where the new gate promotes something
119
+ the old one refused.
120
+ - `regularizedIncompleteBeta` takes the mandatory symmetry branch `I_x(a,b) = 1 − I_{1−x}(b,a)`.
121
+ `studentTCdf(0.005, 100)` returned `0.89152130` against a true `0.50198972`; a
122
+ perfectly null paired result reported `p < 0.05`. This survived 0.133.1, which
123
+ corrected the normal CDF but not the beta continued fraction beneath it, and
124
+ 0.133.2, which changed no statistics math.
125
+ - `mannWhitneyU` and `wilcoxonSignedRank` compute an EXACT conditional p by default
126
+ inside the enumeration thresholds, conditioning on the observed tie pattern.
127
+ - `mannWhitneyU` chooses exact computation from bounded state and work estimates,
128
+ so imbalanced designs such as 1 v 24 remain exact without admitting expensive
129
+ balanced designs. Its automatic permutation seed is invariant to observation
130
+ order and to swapping the two groups.
131
+ - Deleted `wilcoxonSignedRank`'s `n < 6` hard return of `{w: 0, p: 1}`.
132
+ - The asymptotic rank-test path applies the tie correction and the continuity
133
+ correction; both were missing.
134
+ - `mannWhitneyU` and `wilcoxonSignedRank` reject non-finite input. A single `NaN`
135
+ previously spun forever: the tie-grouping loop advanced on `===`, and
136
+ `NaN === NaN` is false.
137
+ - `bonferroni` and `benjaminiHochberg` reject at the inclusive boundary (`p ≤ α`,
138
+ `q ≤ fdr`), matching `holm`, and validate `alpha`/`fdr` and the p-value range.
139
+ `bonferroni([0.0125]×4, 0.05)` returned all-false where `holm` returned all-true.
140
+ BH q-values use R's `(n/rank)·p` form; `(p·n)/rank` lands one ULP above the
141
+ boundary at `p = 0.05, n = 3`.
142
+ - `interRaterReliability` groups by (dimension, item) across judges. It was
143
+ bucketing consecutive scores from the SAME judge, so it measured within-judge
144
+ spread: two identical judges returned `−0.5` where the true α is `+1.0`.
145
+ - `mulberry32(0)` is its own stream. `seed | 0 || 0x9e3779b9` collapsed seed 0 onto
146
+ the golden-ratio constant, so two runs a caller believed were independent
147
+ replicates were the same run. Non-finite seeds now throw.
148
+ - Unseeded bootstraps derive their seed from the data instead of `Math.random`, so
149
+ an interval is reproducible whether or not the caller passes a seed.
150
+ - `studentTCdf` uses the regularized incomplete beta for every finite degree of
151
+ freedom. The deleted normal shortcut changed `df = 102, t = 1.98` from the true
152
+ two-sided `p = 0.050398` to `0.047703`.
153
+ - At the default 95% confidence, campaign promotion decisions use an exact
154
+ one-sided sign test from 6 through 19 paired observations and the bootstrap
155
+ interval from 20 onward. Samples too small to attain the requested confidence
156
+ remain inconclusive.
157
+ - Prior-period reports use the shared Welch implementation. Zero-variance and
158
+ under-sized comparisons carry an explicit status and null inferential fields
159
+ instead of fabricated `p = 1`, `d = 0`, and a zero-width interval.
160
+
161
+ ### Changed — BREAKING
162
+
163
+ - `mannWhitneyU(a, b, opts?)` returns `{ u, uA, p, method, pFloor }`. `p` is now the
164
+ exact conditional p inside the threshold: `mannWhitneyU([1,2,3],[4,5,6]).p` moves
165
+ from `0.03769147` to `0.10000000`, which is the smallest p attainable at 3 v 3.
166
+ `uA` carries the direction that `u = min(u₁,u₂)` discards.
167
+ - `wilcoxonSignedRank(before, after, opts?)` returns
168
+ `{ w, p, method, pFloor, nNonZero }`.
169
+ - Both take `method: 'auto' | 'exact' | 'asymptotic'`, default `'auto'`. `'auto'`
170
+ never selects `'asymptotic'`. Requesting `'asymptotic'` where an exact answer
171
+ exists THROWS a `ValidationError` naming the attainable floor, and requesting
172
+ `'exact'` above the threshold throws rather than enumerating an unbounded
173
+ distribution.
174
+ - `pairedTTest` returns `{ t: number | null, df, p: number | null }`. A non-zero
175
+ constant delta returned `{t: Infinity, p: 0}` — absolute certainty from three
176
+ observations — and now returns null, matching `pairedCohensDz`. An all-zero delta
177
+ is still `{t: 0, p: 1}`.
178
+ - `cohensD` returns `number | null`. It returned a silent `0` for a maximal
179
+ zero-variance separation and for under-sized groups.
180
+ - `MetricVerdict.cohensD` and `LiftInsight.pValue` are nullable accordingly.
181
+ - `PairedBootstrapResult` carries `gateEligible`, false below
182
+ `BOOTSTRAP_GATE_MIN_N = 20`.
183
+ - `welchsTTest` returns a status-tagged full result with means, delta, standard
184
+ error, degrees of freedom, Student-t interval, p-value, and Cohen's d.
185
+ - Prior-period `MetricDelta` values carry the same status. Invalid inference has
186
+ null `ci95`, `pValue`, and `cohensD`, and the comparison lists those metric
187
+ names in `inconclusiveMetrics`.
188
+
189
+ ### Added
190
+
191
+ - `scripts/generate-statistics-oracle.py` + `tests/fixtures/statistics-oracle.json`:
192
+ 154 scipy/statsmodels-generated golden values across 22 statistics, asserted by
193
+ `tests/statistics-oracle.test.ts`. scipy is a CI oracle and is never a runtime
194
+ dependency.
195
+ - `tests/statistics-library-crosscheck.test.ts` cross-checks untied exact null
196
+ distributions against `lib-r-math.js`. Multiple-comparison functions are pinned
197
+ to statsmodels-generated fixture values; `@stdlib/stats-padjust` was removed.
198
+ - First test coverage for `welchsTTest` / `compareToBaseline`, which gated
199
+ improved / regressed / stable verdicts with nothing asserting their numbers.
200
+ - Exported `normalCdf`, `studentTCdf`, `BOOTSTRAP_GATE_MIN_N`,
201
+ `MANN_WHITNEY_EXACT_MAX_STATES`, `MANN_WHITNEY_EXACT_MAX_WORK`,
202
+ `WILCOXON_EXACT_MAX_N`, `DEFAULT_PERMUTATIONS`, and `pairedDeltaTest`.
203
+
204
+ ### Trusted-head recovery
205
+
206
+ #### Fixed
207
+
208
+ - A trusted-head pin write that fails after its journal row is already durable no longer leaves that row permanently unpinned.
209
+ Retrying the same `eventId` moves the pin up to the acknowledged entry when the entry is ahead of the pin, so the last row of a ledger — the promotion decision — cannot be truncated away undetected after a full disk or a read-only mount.
210
+ - A pin whose journal was deleted or rebuilt is recoverable instead of refusing every later append and replay forever.
211
+ The refusal still stands, since a missing journal beside a live pin is the deletion the pin exists to catch, but it now names the sidecar file and the operation that resolves it.
212
+ - Trusted-head read and write faults are reported through the journal codec's error taxonomy instead of escaping as raw Node filesystem errors.
213
+
214
+ #### Added
215
+
216
+ - `clearTrustedHeadFile`, exposed as `FileLedgerJournal.clearTrustedHead()` and `SearchLedger.clearTrustedHead()`, discards a pin and returns the guarantee it gave up.
217
+ - `SearchLedger.pinTrustedHead()`, the campaign-level route to adopting a pin for a ledger that has none.
218
+ - `readTrustedHeadFile` is exported from `@tangle-network/agent-eval/ledger-core`, so a consumer of `verifyEntriesAgainstTrustedHead` has a shape-validating way to load a pin.
219
+
220
+ #### Changed
221
+
222
+ - **Breaking:** `verifyEntriesAgainstTrustedHead` takes `{ subject, trustedHeadPath }` as its fourth argument instead of a bare `subject` string, so every refusal can name the sidecar file. Passing the old string throws a `TypeError`.
223
+ - **Breaking:** `SearchLedger` declares `pinTrustedHead` and `clearTrustedHead`; an external implementation of the interface must supply them.
224
+
7
225
  ## [0.133.2] - 2026-07-27 - protect final evaluation data
8
226
 
9
227
  ### Fixed
@@ -1,10 +1,11 @@
1
1
  import { a as MultiLayerVerifier, l as VerifyOptions, o as Severity } from "../multi-layer-verifier-BHY1gWAc.js";
2
- import { C as FindingSubjectKind, D as parseFindingSubject, E as findingSubjectGrammarPromptFor, F as createAnalystAi, H as SemanticConceptJudgeInput, O as renderFindingSubject, P as CreateAnalystAiConfig, S as FindingSubject, T as KIND_EXPECTED_SUBJECTS, U as SemanticConceptJudgeOptions, Y as RunTrace, _ as defaultIsMaterial, a as SkillUsageScanConfig, b as FINDING_SUBJECT_KINDS, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as PersistedFinding, h as FindingsStore, i as SkillUsageReport, k as BehavioralMetrics, l as KNOWLEDGE_POISONING_KIND_SPEC, m as FindingsDiff, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as DiffPolicy, q as RunCritic, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as diffFindings, w as FindingSubjectStringSchema, x as FINDING_SUBJECT_SYNTAX, y as FINDING_SUBJECT_GRAMMAR_PROMPT } from "../skill-usage-BaaxFSJR.js";
2
+ import { C as FindingSubjectKind, D as parseFindingSubject, E as findingSubjectGrammarPromptFor, F as createAnalystAi, H as SemanticConceptJudgeInput, O as renderFindingSubject, P as CreateAnalystAiConfig, S as FindingSubject, T as KIND_EXPECTED_SUBJECTS, U as SemanticConceptJudgeOptions, Y as RunTrace, _ as defaultIsMaterial, a as SkillUsageScanConfig, b as FINDING_SUBJECT_KINDS, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as PersistedFinding, h as FindingsStore, i as SkillUsageReport, k as BehavioralMetrics, l as KNOWLEDGE_POISONING_KIND_SPEC, m as FindingsDiff, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as DiffPolicy, q as RunCritic, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as diffFindings, w as FindingSubjectStringSchema, x as FINDING_SUBJECT_SYNTAX, y as FINDING_SUBJECT_GRAMMAR_PROMPT } from "../skill-usage-BDQVPIG1.js";
3
3
  import { c as CostLedgerHandle } from "../cost-ledger-fGS_u_O1.js";
4
4
  import { o as LlmClientOptions } from "../llm-client-BiK4HW0u.js";
5
5
  import { A as ChatClient, B as SandboxSdkTransportOpts, F as CreateChatClientOpts, I as CustomTransportOpts, L as DirectProviderTransportOpts, M as ChatResponse, N as ChatTransport, P as CliBridgeTransportOpts, R as MockTransportOpts, V as createChatClient, j as ChatRequest, k as ChatCallOpts, m as JudgeInput, p as JudgeFn, z as RouterTransportOpts } from "../types-Cc3qbqzj.js";
6
6
  import { t as TraceAnalysisStore } from "../store-CxJry_cs.js";
7
- import { A as AnalystRunResult, C as AnalystContext, D as AnalystRequirements, E as AnalystInputKind, F as computeFindingId, I as makeFinding, M as AnalystSeverity, N as AnalystUsageReceipt, O as AnalystRunEvent, P as EvidenceRef, S as Analyst, T as AnalystFinding, _ as RawAnalystEvidenceSchema, a as AnalystRegistryOptions, b as evidenceRefsFromRawFinding, c as CreateTraceAnalystKindOpts, d as createTraceAnalystKind, f as renderPriorFindings, g as RawAnalystEvidence, h as RAW_FINDING_SCHEMA_PROMPT, i as AnalystRegistry, j as AnalystRunSummary, k as AnalystRunInputs, l as TraceAnalystGolden, m as ANALYST_SEVERITIES, n as buildDefaultAnalystRegistry, o as BudgetPolicy, p as renderUpstreamFindings, r as AnalystHooks, s as RegistryRunOpts, t as DefaultAnalystRegistryOptions, u as TraceAnalystKindSpec, v as RawAnalystFinding, w as AnalystCost, x as parseRawFinding, y as RawAnalystFindingSchema } from "../default-registry-Cl3pHo4n.js";
7
+ import { _ as makeFinding, a as AnalystInputKind, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as computeFindingId, h as ProposalFindingOrigin, i as AnalystFinding, l as AnalystRunResult, m as ProposalFinding, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as makeProposalFinding } from "../types-DVjczBM9.js";
8
+ import { _ as RawAnalystEvidenceSchema, a as AnalystRegistryOptions, b as evidenceRefsFromRawFinding, c as CreateTraceAnalystKindOpts, d as createTraceAnalystKind, f as renderPriorFindings, g as RawAnalystEvidence, h as RAW_FINDING_SCHEMA_PROMPT, i as AnalystRegistry, l as TraceAnalystGolden, m as ANALYST_SEVERITIES, n as buildDefaultAnalystRegistry, o as BudgetPolicy, p as renderUpstreamFindings, r as AnalystHooks, s as RegistryRunOpts, t as DefaultAnalystRegistryOptions, u as TraceAnalystKindSpec, v as RawAnalystFinding, x as parseRawFinding, y as RawAnalystFindingSchema } from "../default-registry-Brxr728w.js";
8
9
  import { AxFunction } from "@ax-llm/ax";
9
10
  //#region src/analyst/adapters.d.ts
10
11
  declare function liftSeverity(s: Severity): AnalystSeverity;
@@ -88,40 +89,15 @@ declare function coerceJson(text: string): unknown;
88
89
  */
89
90
  declare function coerceToFindingRows(raw: unknown): unknown[];
90
91
  //#endregion
91
- //#region src/analyst/steer-firewall.d.ts
92
- /** DESCRIPTIVE predicate: does the finding cite at least one observable
93
- * (span/event/artifact) evidence ref. Useful for ranking evidence quality or
94
- * rendering — it is NOT the steer gate. Evidence presence is the WRONG
95
- * discriminator for steering: a legitimate trace-analyst observation may cite
96
- * nothing (it would be wrongly rejected), and a judge verdict may cite an
97
- * artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate
98
- * steering; use this only where "is this grounded in observable evidence" is the
99
- * literal question. */
100
- declare function isTraceObservable(finding: AnalystFinding): boolean;
101
- /** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a
102
- * finding), identified by provenance set at the lift site — independent of
103
- * whatever evidence it cites. */
104
- declare function isJudgeVerdict(finding: AnalystFinding): boolean;
92
+ //#region src/analyst/proposal-findings.d.ts
93
+ /** True when a finding names a source candidate generation may learn from. */
94
+ declare function isProposalFinding(finding: unknown): finding is ProposalFinding;
105
95
  /**
106
- * THE steer firewall. Fail-loud guard for any path that admits analyst findings
107
- * as STEERING input (the `f(trace)` role): rejects naming the offenders — any
108
- * finding whose provenance is a judge verdict, rather than let `J` leak into the
109
- * loop. Returns the findings unchanged for chaining.
110
- *
111
- * Call this at the chokepoint where a detector that ALSO scores/gates has its
112
- * findings turned into a steer (the judge-and-steer dual-role case). It keys on
113
- * provenance, so it correctly admits evidence-less trace-analyst observations and
114
- * correctly rejects an artifact-citing judge verdict — the cases an evidence
115
- * check gets backwards.
116
- *
117
- * It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge
118
- * whose output is laundered through a hand-built finding with no provenance flag
119
- * is out of its reach — provenance must be honestly set at every judge→finding
120
- * lift (today: createJudgeAdapter). That is why the integrity rule lives at the
121
- * lift site, and why ProposeContext.judgeScores?: never is the complementary
122
- * compile-time tripwire on the obvious direct channel.
96
+ * Reject findings whose source has not been explicitly admitted for candidate
97
+ * generation. Search feedback and observed production behavior are allowed;
98
+ * final evaluation data has no allowed origin.
123
99
  */
124
- declare function assertNoJudgeVerdict(findings: ReadonlyArray<AnalystFinding>, context?: string): ReadonlyArray<AnalystFinding>;
100
+ declare function assertProposalFindings(findings: unknown, context?: string): ReadonlyArray<ProposalFinding>;
125
101
  //#endregion
126
102
  //#region src/analyst/structure-findings.d.ts
127
103
  interface StructureFindingsOptions {
@@ -176,5 +152,5 @@ type TraceToolGroupName =
176
152
  */
177
153
  declare function buildTraceToolsForGroup(group: TraceToolGroupName, store: TraceAnalysisStore): AxFunction[];
178
154
  //#endregion
179
- export { ANALYST_SEVERITIES, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type BudgetPolicy, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateTraceAnalystKindOpts, type CustomTransportOpts, DEFAULT_TRACE_ANALYST_KINDS, type DefaultAnalystRegistryOptions, type DiffPolicy, type DirectProviderTransportOpts, type EvidenceRef, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type MockTransportOpts, type PersistedFinding, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StructureFindingsOptions, type StructureFindingsResult, type TraceAnalystGolden, type TraceAnalystKindSpec, type TraceToolGroupName, type VerifierAdapterOpts, assertNoJudgeVerdict, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isJudgeVerdict, isTraceObservable, liftSeverity, makeFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
155
+ export { ANALYST_SEVERITIES, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type BudgetPolicy, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateTraceAnalystKindOpts, type CustomTransportOpts, DEFAULT_TRACE_ANALYST_KINDS, type DefaultAnalystRegistryOptions, type DiffPolicy, type DirectProviderTransportOpts, type EvidenceRef, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type MockTransportOpts, type PersistedFinding, type ProposalFinding, type ProposalFindingOrigin, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StructureFindingsOptions, type StructureFindingsResult, type TraceAnalystGolden, type TraceAnalystKindSpec, type TraceToolGroupName, type VerifierAdapterOpts, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, makeFinding, makeProposalFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
180
156
  //# sourceMappingURL=index.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"index.d.ts","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/behavioral-analyst.ts","../../src/analyst/parse-tolerant.ts","../../src/analyst/steer-firewall.ts","../../src/analyst/structure-findings.ts","../../src/analyst/tool-groups.ts"],"mappings":";;;;;;;;;iBA4CgB,aAAa,GAAG,WAAgB;UAe/B,oBAAoB;EACnC;EACA;EACA,UAAU,mBAAmB;;;;;EAK7B,UAAU,KAAK,cAAc;;iBAGf,sBAAsB,KAAK,MAAM,oBAAoB,OAAO,QAAQ;UAwEnE;EACf;EACA;EACA,SAAS;;EAET;;iBAGc,uBAAuB,OAAM,uBAA4B,QAAQ;UAiEhE;EACf;EACA;EACA,OAAO;;EAEP,MAAM;;EAEN,OAAO;;EAEP;;iBAGc,mBAAmB,MAAM,mBAAmB,QAAQ;UAmDnD;EACf;EACA;;EAEA,UAAU,KAAK;;EAEf;;iBAGc,kCACd,OAAM,kCACL,QAAQ;;;;;;;iBC9OK,yBACd,SAAS,mBACT;EAAQ;EAAoB;IAC3B;;iBAkCa,qBAAqB,QAAQ;;;;;;;;;;;;;;;iBC3E7B,gBAAgB;;;;;iBAgBhB,WAAW;;;;;;;iBAeX,oBAAoB;;;;;;;;;;;iBCNpB,kBAAkB,SAAS;;;;iBAO3B,eAAe,SAAS;;;;;;;;;;;;;;;;;;;;iBAuBxB,qBACd,UAAU,cAAc,iBACxB,mBACC,cAAc;;;UCrCA;;EAEf;EACA;;EAEA;EACA;EACA;EACA;;EAEA,aAAa;EACb;EACA,WAAW;EACX;EACA,SAAS;;EAET;;EAEA,cAAc,KAAK,sBAAsB;;EAEzC,kBAAkB;;EAElB,YAAY;;UAGG;EACf,UAAU;EACV;;iBAuCoB,kBACpB,MAAM,2BACL,QAAQ;;;;KCrFC;;;;;;;;;;;;;;;;;;iBAuCI,wBACd,OAAO,oBACP,OAAO,qBACN"}
1
+ {"version":3,"file":"index.d.ts","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/behavioral-analyst.ts","../../src/analyst/parse-tolerant.ts","../../src/analyst/proposal-findings.ts","../../src/analyst/structure-findings.ts","../../src/analyst/tool-groups.ts"],"mappings":";;;;;;;;;;iBA4CgB,aAAa,GAAG,WAAgB;UAe/B,oBAAoB;EACnC;EACA;EACA,UAAU,mBAAmB;;;;;EAK7B,UAAU,KAAK,cAAc;;iBAGf,sBAAsB,KAAK,MAAM,oBAAoB,OAAO,QAAQ;UAwEnE;EACf;EACA;EACA,SAAS;;EAET;;iBAGc,uBAAuB,OAAM,uBAA4B,QAAQ;UAiEhE;EACf;EACA;EACA,OAAO;;EAEP,MAAM;;EAEN,OAAO;;EAEP;;iBAGc,mBAAmB,MAAM,mBAAmB,QAAQ;UAiDnD;EACf;EACA;;EAEA,UAAU,KAAK;;EAEf;;iBAGc,kCACd,OAAM,kCACL,QAAQ;;;;;;;iBC5OK,yBACd,SAAS,mBACT;EAAQ;EAAoB;IAC3B;;iBAkCa,qBAAqB,QAAQ;;;;;;;;;;;;;;;iBC3E7B,gBAAgB;;;;;iBAgBhB,WAAW;;;;;;;iBAeX,oBAAoB;;;;iBCXpB,kBAAkB,mBAAmB,WAAW;;;;;;iBAShD,uBACd,mBACA,mBACC,cAAc;;;UCXA;;EAEf;EACA;;EAEA;EACA;EACA;EACA;;EAEA,aAAa;EACb;EACA,WAAW;EACX;EACA,SAAS;;EAET;;EAEA,cAAc,KAAK,sBAAsB;;EAEzC,kBAAkB;;EAElB,YAAY;;UAGG;EACf,UAAU;EACV;;iBAuCoB,kBACpB,MAAM,2BACL,QAAQ;;;;KCrFC;;;;;;;;;;;;;;;;;;iBAuCI,wBACd,OAAO,oBACP,OAAO,qBACN"}
@@ -1,6 +1,7 @@
1
- import { A as parseFindingSubject, C as stripCodeFences, D as FindingSubjectStringSchema, E as FINDING_SUBJECT_SYNTAX, F as makeFinding, L as createChatClient, M as behavioralAnalyst, N as deriveEfficiencyFindings, O as KIND_EXPECTED_SUBJECTS, P as computeFindingId, R as createAnalystAi, S as coerceToFindingRows, T as FINDING_SUBJECT_KINDS, _ as RawAnalystEvidenceSchema, a as KNOWLEDGE_GAP_KIND_SPEC, b as parseRawFinding, c as buildTraceToolsForGroup, d as renderUpstreamFindings, f as settleUsageReceiptFromCostLedger, g as RAW_FINDING_SCHEMA_PROMPT, h as ANALYST_SEVERITIES, i as KNOWLEDGE_POISONING_KIND_SPEC, j as renderFindingSubject, k as findingSubjectGrammarPromptFor, l as createTraceAnalystKind, m as structureFindings, n as AnalystRegistry, o as IMPROVEMENT_KIND_SPEC, p as validateUsageSettlementTimeout, r as DEFAULT_TRACE_ANALYST_KINDS, s as FAILURE_MODE_KIND_SPEC, t as buildDefaultAnalystRegistry, u as renderPriorFindings, v as RawAnalystFindingSchema, w as FINDING_SUBJECT_GRAMMAR_PROMPT, x as coerceJson, y as evidenceRefsFromRawFinding } from "../default-registry-D3T9XbuY.js";
1
+ import { A as parseFindingSubject, C as stripCodeFences, D as FindingSubjectStringSchema, E as FINDING_SUBJECT_SYNTAX, F as createChatClient, I as createAnalystAi, M as behavioralAnalyst, N as deriveEfficiencyFindings, O as KIND_EXPECTED_SUBJECTS, S as coerceToFindingRows, T as FINDING_SUBJECT_KINDS, _ as RawAnalystEvidenceSchema, a as KNOWLEDGE_GAP_KIND_SPEC, b as parseRawFinding, c as buildTraceToolsForGroup, d as renderUpstreamFindings, f as settleUsageReceiptFromCostLedger, g as RAW_FINDING_SCHEMA_PROMPT, h as ANALYST_SEVERITIES, i as KNOWLEDGE_POISONING_KIND_SPEC, j as renderFindingSubject, k as findingSubjectGrammarPromptFor, l as createTraceAnalystKind, m as structureFindings, n as AnalystRegistry, o as IMPROVEMENT_KIND_SPEC, p as validateUsageSettlementTimeout, r as DEFAULT_TRACE_ANALYST_KINDS, s as FAILURE_MODE_KIND_SPEC, t as buildDefaultAnalystRegistry, u as renderPriorFindings, v as RawAnalystFindingSchema, w as FINDING_SUBJECT_GRAMMAR_PROMPT, x as coerceJson, y as evidenceRefsFromRawFinding } from "../default-registry-IjYs7T8l.js";
2
2
  import { i as CostLedger } from "../cost-ledger-BrJxbrMy.js";
3
- import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, l as emitSkillUsageFindings, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-BypLt6Fw.js";
3
+ import { c as makeFinding, l as makeProposalFinding, n as isProposalFinding, s as computeFindingId, t as assertProposalFindings } from "../proposal-findings-DCawte-y.js";
4
+ import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, l as emitSkillUsageFindings, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-C0P1VTXD.js";
4
5
  //#region src/analyst/adapters.ts
5
6
  /**
6
7
  * Adapter factories — lift each existing agent-eval primitive into the
@@ -288,56 +289,6 @@ function createSemanticConceptJudgeAdapter(opts = {}) {
288
289
  };
289
290
  }
290
291
  //#endregion
291
- //#region src/analyst/steer-firewall.ts
292
- /** Evidence grounded in the agent's OWN execution: OTLP trace elements
293
- * (`span`/`event`) or the artifact it produced (`artifact`). */
294
- const OBSERVABLE_KINDS = /* @__PURE__ */ new Set([
295
- "span",
296
- "event",
297
- "artifact"
298
- ]);
299
- /** DESCRIPTIVE predicate: does the finding cite at least one observable
300
- * (span/event/artifact) evidence ref. Useful for ranking evidence quality or
301
- * rendering — it is NOT the steer gate. Evidence presence is the WRONG
302
- * discriminator for steering: a legitimate trace-analyst observation may cite
303
- * nothing (it would be wrongly rejected), and a judge verdict may cite an
304
- * artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate
305
- * steering; use this only where "is this grounded in observable evidence" is the
306
- * literal question. */
307
- function isTraceObservable(finding) {
308
- return finding.evidence_refs.some((ref) => OBSERVABLE_KINDS.has(ref.kind));
309
- }
310
- /** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a
311
- * finding), identified by provenance set at the lift site — independent of
312
- * whatever evidence it cites. */
313
- function isJudgeVerdict(finding) {
314
- return finding.derived_from_judge === true;
315
- }
316
- /**
317
- * THE steer firewall. Fail-loud guard for any path that admits analyst findings
318
- * as STEERING input (the `f(trace)` role): rejects — naming the offenders — any
319
- * finding whose provenance is a judge verdict, rather than let `J` leak into the
320
- * loop. Returns the findings unchanged for chaining.
321
- *
322
- * Call this at the chokepoint where a detector that ALSO scores/gates has its
323
- * findings turned into a steer (the judge-and-steer dual-role case). It keys on
324
- * provenance, so it correctly admits evidence-less trace-analyst observations and
325
- * correctly rejects an artifact-citing judge verdict — the cases an evidence
326
- * check gets backwards.
327
- *
328
- * It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge
329
- * whose output is laundered through a hand-built finding with no provenance flag
330
- * is out of its reach — provenance must be honestly set at every judge→finding
331
- * lift (today: createJudgeAdapter). That is why the integrity rule lives at the
332
- * lift site, and why ProposeContext.judgeScores?: never is the complementary
333
- * compile-time tripwire on the obvious direct channel.
334
- */
335
- function assertNoJudgeVerdict(findings, context = "steer") {
336
- const leaks = findings.filter(isJudgeVerdict);
337
- if (leaks.length > 0) throw new Error(`${context}: a judge verdict cannot be admitted as steering input — that is the held-out judge leaking into the loop. Offending judge-derived findings: [${leaks.map((f) => f.finding_id).join(", ")}]. Steering consumes observations of behavior, never acceptance verdicts.`);
338
- return findings;
339
- }
340
- //#endregion
341
- export { ANALYST_SEVERITIES, AnalystRegistry, DEFAULT_TRACE_ANALYST_KINDS, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, FindingSubjectStringSchema, FindingsStore, IMPROVEMENT_KIND_SPEC, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, RAW_FINDING_SCHEMA_PROMPT, RawAnalystEvidenceSchema, RawAnalystFindingSchema, SKILL_USAGE_ANALYST, SkillUsageAnalyst, assertNoJudgeVerdict, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isJudgeVerdict, isTraceObservable, liftSeverity, makeFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
292
+ export { ANALYST_SEVERITIES, AnalystRegistry, DEFAULT_TRACE_ANALYST_KINDS, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, FindingSubjectStringSchema, FindingsStore, IMPROVEMENT_KIND_SPEC, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, RAW_FINDING_SCHEMA_PROMPT, RawAnalystEvidenceSchema, RawAnalystFindingSchema, SKILL_USAGE_ANALYST, SkillUsageAnalyst, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, makeFinding, makeProposalFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
342
293
 
343
294
  //# sourceMappingURL=index.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/steer-firewall.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport { CostLedger } from '../cost-ledger'\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n type SemanticConceptJudgeResult,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore } from '../types'\nimport type { ChatClient } from './chat-client'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\nimport { settleUsageReceiptFromCostLedger, validateUsageSettlementTimeout } from './usage-receipt'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** Chat client passed to the JudgeFn. */\n chat: ChatClient\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.chat, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Provenance: this finding IS a judge verdict (an acceptance score), not an\n // observation of behavior. The steer firewall (assertNoJudgeVerdict) rejects\n // it from steering — even when it cites an artifact above — because letting a\n // verdict steer the next attempt is the held-out judge leaking into the loop.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n /** Registry context owns cancellation and the per-analyst cost ledger. */\n options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>\n /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */\n settlementTimeoutMs?: number\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs)\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: {\n kind: 'llm',\n models: opts.options?.model ? [opts.options.model] : undefined,\n settlement_timeout_ms: settlementTimeoutMs,\n },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input, ctx) {\n const costLedger = new CostLedger(ctx.budgetUsd)\n let result: SemanticConceptJudgeResult\n try {\n result = await runSemanticConceptJudge(input, {\n ...opts.options,\n costLedger,\n signal: ctx.signal,\n })\n } finally {\n const usage = await settleUsageReceiptFromCostLedger(costLedger, {\n channel: 'judge',\n timeoutMs: settlementTimeoutMs,\n })\n if (!usage.settled) {\n ctx.log?.('semantic-concept judge provider settlement timed out', {\n pending_calls: usage.pendingCalls,\n timeout_ms: settlementTimeoutMs,\n })\n }\n ctx.recordUsage?.(usage.receipt)\n }\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n },\n }),\n )\n }\n return out\n },\n }\n}\n","// The realness-oracle firewall (docs/learning-flywheel.md, \"The steer is f(trace)\").\n//\n// A realness/authenticity signal has TWO legitimate roles that must stay\n// separated by a firewall:\n// (a) anchor judge J — write-only: scores the chosen output, gates promotion,\n// NEVER seen by the worker/optimizer mid-run (else the loop games it).\n// (b) steer f(trace) — an analyst observes the agent's OWN behavior in the\n// trace (\"imported a stub\", \"used a non-crypto PRNG where encryption was\n// required\") and steers the next attempt. Legitimate, because it is derived\n// from OBSERVABLE BEHAVIOR, not from J's held-out verdict.\n//\n// The correct discriminator is PROVENANCE, not evidence presence. A judge verdict\n// lifted into a finding (createJudgeAdapter → liftJudgeScore) is a verdict even\n// when it cites an artifact; an evidence-less trace-analyst bullet is an\n// observation even though it cites nothing. So the firewall keys on\n// `AnalystFinding.derived_from_judge` (set at the judge lift site), NOT on whether\n// evidence_refs is populated. The instant a verdict steers the next attempt it is\n// a back-channel for J and the loop Goodharts realness exactly as it would\n// Goodhart pass-rate.\n\nimport type { AnalystFinding, EvidenceRef } from './types'\n\n/** Evidence grounded in the agent's OWN execution: OTLP trace elements\n * (`span`/`event`) or the artifact it produced (`artifact`). */\nconst OBSERVABLE_KINDS: ReadonlySet<EvidenceRef['kind']> = new Set<EvidenceRef['kind']>([\n 'span',\n 'event',\n 'artifact',\n])\n\n/** DESCRIPTIVE predicate: does the finding cite at least one observable\n * (span/event/artifact) evidence ref. Useful for ranking evidence quality or\n * rendering — it is NOT the steer gate. Evidence presence is the WRONG\n * discriminator for steering: a legitimate trace-analyst observation may cite\n * nothing (it would be wrongly rejected), and a judge verdict may cite an\n * artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate\n * steering; use this only where \"is this grounded in observable evidence\" is the\n * literal question. */\nexport function isTraceObservable(finding: AnalystFinding): boolean {\n return finding.evidence_refs.some((ref) => OBSERVABLE_KINDS.has(ref.kind))\n}\n\n/** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a\n * finding), identified by provenance set at the lift site — independent of\n * whatever evidence it cites. */\nexport function isJudgeVerdict(finding: AnalystFinding): boolean {\n return finding.derived_from_judge === true\n}\n\n/**\n * THE steer firewall. Fail-loud guard for any path that admits analyst findings\n * as STEERING input (the `f(trace)` role): rejects — naming the offenders — any\n * finding whose provenance is a judge verdict, rather than let `J` leak into the\n * loop. Returns the findings unchanged for chaining.\n *\n * Call this at the chokepoint where a detector that ALSO scores/gates has its\n * findings turned into a steer (the judge-and-steer dual-role case). It keys on\n * provenance, so it correctly admits evidence-less trace-analyst observations and\n * correctly rejects an artifact-citing judge verdict — the cases an evidence\n * check gets backwards.\n *\n * It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge\n * whose output is laundered through a hand-built finding with no provenance flag\n * is out of its reach — provenance must be honestly set at every judge→finding\n * lift (today: createJudgeAdapter). That is why the integrity rule lives at the\n * lift site, and why ProposeContext.judgeScores?: never is the complementary\n * compile-time tripwire on the obvious direct channel.\n */\nexport function assertNoJudgeVerdict(\n findings: ReadonlyArray<AnalystFinding>,\n context = 'steer',\n): ReadonlyArray<AnalystFinding> {\n const leaks = findings.filter(isJudgeVerdict)\n if (leaks.length > 0) {\n throw new Error(\n `${context}: a judge verdict cannot be admitted as steering input — that is the ` +\n `held-out judge leaking into the loop. Offending judge-derived findings: [${leaks\n .map((f) => f.finding_id)\n .join(', ')}]. Steering consumes observations of behavior, never acceptance verdicts.`,\n )\n }\n return findings\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;AAwCA,MAAM,cAAc;AAIpB,SAAgB,aAAa,GAAmC;CAC9D,QAAQ,GAAR;EACE,KAAK,YACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,QACH,OAAO;CACX;AACF;AAeA,SAAgB,sBAA2B,MAA8C;CACvF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,YAAY;EACrB,MAAM,QAAQ,KAAK,KAAK;GACtB,MAAM,SAAS,MAAM,KAAK,SAAS,IAAI;IAAE;IAAK,GAAG,KAAK;GAAQ,CAAC;GAC/D,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,SAAS,OAAO,QAAQ;IACjC,KAAK,MAAM,WAAW,MAAM,UAC1B,IAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;IAI3D,IAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAC1E,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,MAAM;KACf,OAAO,UAAU,MAAM,MAAM,IAAI,MAAM,OAAO,IAAI,MAAM,UAAU;KAClE,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;KAC9E,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MACR,cAAc,MAAM;MACpB,aAAa,MAAM;MACnB,OAAO,MAAM;MACb,aAAa,MAAM;KACrB;IACF,CAAC,CACH;GAEJ;GACA,IAAI,MAAM,qBAAqB;IAC7B,QAAQ,OAAO,OAAO;IACtB,SAAS,OAAO;IAChB,UAAU,OAAO;GACnB,CAAC;GACD,OAAO;EACT;CACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;CAChB,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE,SAAS;EACpB,OAAO,EAAE;EACT,UAAU,aAAa,EAAE,QAAQ;EACjC,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EACL,UAAU,EAAE;CACd,CAAC;AACH;AAYA,SAAgB,uBAAuB,OAA6B,CAAC,GAAsB;CACzF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,SAAS,KAAK,UAAU,IAAI,UAAU;CAC5C,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,cAAc;EACvB,MAAM,QAAQ,OAAO;GACnB,MAAM,QAAQ,OAAO,WAAW,KAAK;GACrC,MAAM,MAAwB,CAAC;GAU/B,KAAK,MAAM,CAAC,KAAK,KAAK,QAAQ;IAR5B;KAAC;KAAW;KAAY;IAAmC;IAC3D;KAAC;KAAgB;KAAQ;IAAsB;IAC/C;KAAC;KAAoB;KAAQ;IAA6C;IAC1E;KAAC;KAAkB;KAAU;IAAyB;IACtD;KAAC;KAAgB;KAAU;IAA6B;IACxD;KAAC;KAAe;KAAQ;IAA6B;IACrD;KAAC;KAAa;KAAY;IAAwB;GAEnB,GAAG;IAClC,MAAM,QAAQ,MAAM;IACpB,IAAI,OAAO,UAAU,YAAY,QAAQ,WACvC,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS;KACT,OAAO;KACP,WAAW,GAAG,IAAI,GAAG,MAAM,QAAQ,CAAC,EAAE,mBAAmB;KACzD,UAAU;KACV,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MAAE,WAAW;MAAK;MAAO;MAAW,QAAQ,MAAM,IAAI;KAAM;IACxE,CAAC,CACH;GAEJ;GAEA,IAAI,MAAM,eAAe,IAAI,WAC3B,IAAI,KACF,YAAY;IACV,YAAY;IACZ;IACA,SAAS;IACT,OAAO;IACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC;IACvD,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU;KAAE,eAAe,MAAM;KAAc,OAAO,MAAM;IAAM;GACpE,CAAC,CACH;GAEF,OAAO;EACT;CACF;AACF;AAgBA,SAAgB,mBAAmB,MAA6C;CAC9E,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;EACjC,SAAS,SAAS;EAClB,MAAM,QAAQ,OAAO;GAEnB,QAAO,MADc,KAAK,MAAM,KAAK,MAAM,KAAK,EAAA,CAE7C,QAAQ,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,CAAC,CAC/C,KAAK,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;EAC3C;CACF;AACF;AAEA,SAAS,YAAY,GAAmB;CAEtC,OAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;CACvF,MAAM,UAAU,YAAY,EAAE,KAAK;CACnC,MAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;CAC7E,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE;EACX,OAAO,GAAG,EAAE,UAAU,GAAG,EAAE,UAAU,UAAU,QAAQ,QAAQ,CAAC,EAAE;EAClE,WAAW,EAAE;EACb;EACA,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EAKL,oBAAoB;EACpB,UAAU;GAAE,YAAY,EAAE;GAAW,WAAW,EAAE;GAAW,UAAU;EAAQ;CACjF,CAAC;AACH;AAaA,SAAgB,kCACd,OAAwC,CAAC,GACL;CACpC,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,sBAAsB,+BAA+B,KAAK,mBAAmB;CACnF,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM;GACJ,MAAM;GACN,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,KAAA;GACrD,uBAAuB;EACzB;EACA,SAAS,GAAG,+BAA+B,WAAW;EACtD,MAAM,QAAQ,OAAO,KAAK;GACxB,MAAM,aAAa,IAAI,WAAW,IAAI,SAAS;GAC/C,IAAI;GACJ,IAAI;IACF,SAAS,MAAM,wBAAwB,OAAO;KAC5C,GAAG,KAAK;KACR;KACA,QAAQ,IAAI;IACd,CAAC;GACH,UAAU;IACR,MAAM,QAAQ,MAAM,iCAAiC,YAAY;KAC/D,SAAS;KACT,WAAW;IACb,CAAC;IACD,IAAI,CAAC,MAAM,SACT,IAAI,MAAM,wDAAwD;KAChE,eAAe,MAAM;KACrB,YAAY;IACd,CAAC;IAEH,IAAI,cAAc,MAAM,OAAO;GACjC;GACA,IAAI,CAAC,OAAO,WACV,OAAO,CACL,YAAY;IACV,YAAY;IACZ;IACA,OAAO;IACP,WAAW,OAAO;IAClB,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;GACnC,CAAC,CACH;GAEF,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,KAAK,OAAO,UAAU;IAG/B,IAAI,EAAE,WAAW,EAAE,SAAS,GAAG;IAC/B,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,EAAE;KACX,OAAO,EAAE,UACL,YAAY,EAAE,QAAQ,aAAa,EAAE,MAAM,QAC3C,YAAY,EAAE,QAAQ;KAC1B,WAAW,EAAE;KACb,UAAU,aAAa,EAAE,QAAQ;KACjC,YAAY;KACZ,eAAe,CAAC;MAAE,MAAM;MAAY,KAAK;MAAmB,SAAS,EAAE;KAAS,CAAC;KACjF,UAAU;MACR,SAAS,EAAE;MACX,SAAS,EAAE;MACX,UAAU,EAAE;KACd;IACF,CAAC,CACH;GACF;GACA,OAAO;EACT;CACF;AACF;;;;;ACxVA,MAAM,mCAAqD,IAAI,IAAyB;CACtF;CACA;CACA;AACF,CAAC;;;;;;;;;AAUD,SAAgB,kBAAkB,SAAkC;CAClE,OAAO,QAAQ,cAAc,MAAM,QAAQ,iBAAiB,IAAI,IAAI,IAAI,CAAC;AAC3E;;;;AAKA,SAAgB,eAAe,SAAkC;CAC/D,OAAO,QAAQ,uBAAuB;AACxC;;;;;;;;;;;;;;;;;;;;AAqBA,SAAgB,qBACd,UACA,UAAU,SACqB;CAC/B,MAAM,QAAQ,SAAS,OAAO,cAAc;CAC5C,IAAI,MAAM,SAAS,GACjB,MAAM,IAAI,MACR,GAAG,QAAQ,gJACmE,MACzE,KAAK,MAAM,EAAE,UAAU,CAAC,CACxB,KAAK,IAAI,EAAE,0EAClB;CAEF,OAAO;AACT"}
1
+ {"version":3,"file":"index.js","names":[],"sources":["../../src/analyst/adapters.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport { CostLedger } from '../cost-ledger'\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n type SemanticConceptJudgeResult,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore } from '../types'\nimport type { ChatClient } from './chat-client'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\nimport { settleUsageReceiptFromCostLedger, validateUsageSettlementTimeout } from './usage-receipt'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** Chat client passed to the JudgeFn. */\n chat: ChatClient\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.chat, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Descriptive origin only. A caller may admit search feedback into candidate\n // generation, but final evaluation findings have no allowed proposal origin.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n /** Registry context owns cancellation and the per-analyst cost ledger. */\n options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>\n /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */\n settlementTimeoutMs?: number\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs)\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: {\n kind: 'llm',\n models: opts.options?.model ? [opts.options.model] : undefined,\n settlement_timeout_ms: settlementTimeoutMs,\n },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input, ctx) {\n const costLedger = new CostLedger(ctx.budgetUsd)\n let result: SemanticConceptJudgeResult\n try {\n result = await runSemanticConceptJudge(input, {\n ...opts.options,\n costLedger,\n signal: ctx.signal,\n })\n } finally {\n const usage = await settleUsageReceiptFromCostLedger(costLedger, {\n channel: 'judge',\n timeoutMs: settlementTimeoutMs,\n })\n if (!usage.settled) {\n ctx.log?.('semantic-concept judge provider settlement timed out', {\n pending_calls: usage.pendingCalls,\n timeout_ms: settlementTimeoutMs,\n })\n }\n ctx.recordUsage?.(usage.receipt)\n }\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n },\n }),\n )\n }\n return out\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;AAwCA,MAAM,cAAc;AAIpB,SAAgB,aAAa,GAAmC;CAC9D,QAAQ,GAAR;EACE,KAAK,YACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,QACH,OAAO;CACX;AACF;AAeA,SAAgB,sBAA2B,MAA8C;CACvF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,YAAY;EACrB,MAAM,QAAQ,KAAK,KAAK;GACtB,MAAM,SAAS,MAAM,KAAK,SAAS,IAAI;IAAE;IAAK,GAAG,KAAK;GAAQ,CAAC;GAC/D,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,SAAS,OAAO,QAAQ;IACjC,KAAK,MAAM,WAAW,MAAM,UAC1B,IAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;IAI3D,IAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAC1E,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,MAAM;KACf,OAAO,UAAU,MAAM,MAAM,IAAI,MAAM,OAAO,IAAI,MAAM,UAAU;KAClE,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;KAC9E,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MACR,cAAc,MAAM;MACpB,aAAa,MAAM;MACnB,OAAO,MAAM;MACb,aAAa,MAAM;KACrB;IACF,CAAC,CACH;GAEJ;GACA,IAAI,MAAM,qBAAqB;IAC7B,QAAQ,OAAO,OAAO;IACtB,SAAS,OAAO;IAChB,UAAU,OAAO;GACnB,CAAC;GACD,OAAO;EACT;CACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;CAChB,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE,SAAS;EACpB,OAAO,EAAE;EACT,UAAU,aAAa,EAAE,QAAQ;EACjC,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EACL,UAAU,EAAE;CACd,CAAC;AACH;AAYA,SAAgB,uBAAuB,OAA6B,CAAC,GAAsB;CACzF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,SAAS,KAAK,UAAU,IAAI,UAAU;CAC5C,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,cAAc;EACvB,MAAM,QAAQ,OAAO;GACnB,MAAM,QAAQ,OAAO,WAAW,KAAK;GACrC,MAAM,MAAwB,CAAC;GAU/B,KAAK,MAAM,CAAC,KAAK,KAAK,QAAQ;IAR5B;KAAC;KAAW;KAAY;IAAmC;IAC3D;KAAC;KAAgB;KAAQ;IAAsB;IAC/C;KAAC;KAAoB;KAAQ;IAA6C;IAC1E;KAAC;KAAkB;KAAU;IAAyB;IACtD;KAAC;KAAgB;KAAU;IAA6B;IACxD;KAAC;KAAe;KAAQ;IAA6B;IACrD;KAAC;KAAa;KAAY;IAAwB;GAEnB,GAAG;IAClC,MAAM,QAAQ,MAAM;IACpB,IAAI,OAAO,UAAU,YAAY,QAAQ,WACvC,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS;KACT,OAAO;KACP,WAAW,GAAG,IAAI,GAAG,MAAM,QAAQ,CAAC,EAAE,mBAAmB;KACzD,UAAU;KACV,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MAAE,WAAW;MAAK;MAAO;MAAW,QAAQ,MAAM,IAAI;KAAM;IACxE,CAAC,CACH;GAEJ;GAEA,IAAI,MAAM,eAAe,IAAI,WAC3B,IAAI,KACF,YAAY;IACV,YAAY;IACZ;IACA,SAAS;IACT,OAAO;IACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC;IACvD,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU;KAAE,eAAe,MAAM;KAAc,OAAO,MAAM;IAAM;GACpE,CAAC,CACH;GAEF,OAAO;EACT;CACF;AACF;AAgBA,SAAgB,mBAAmB,MAA6C;CAC9E,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;EACjC,SAAS,SAAS;EAClB,MAAM,QAAQ,OAAO;GAEnB,QAAO,MADc,KAAK,MAAM,KAAK,MAAM,KAAK,EAAA,CAE7C,QAAQ,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,CAAC,CAC/C,KAAK,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;EAC3C;CACF;AACF;AAEA,SAAS,YAAY,GAAmB;CAEtC,OAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;CACvF,MAAM,UAAU,YAAY,EAAE,KAAK;CACnC,MAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;CAC7E,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE;EACX,OAAO,GAAG,EAAE,UAAU,GAAG,EAAE,UAAU,UAAU,QAAQ,QAAQ,CAAC,EAAE;EAClE,WAAW,EAAE;EACb;EACA,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EAGL,oBAAoB;EACpB,UAAU;GAAE,YAAY,EAAE;GAAW,WAAW,EAAE;GAAW,UAAU;EAAQ;CACjF,CAAC;AACH;AAaA,SAAgB,kCACd,OAAwC,CAAC,GACL;CACpC,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,sBAAsB,+BAA+B,KAAK,mBAAmB;CACnF,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM;GACJ,MAAM;GACN,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,KAAA;GACrD,uBAAuB;EACzB;EACA,SAAS,GAAG,+BAA+B,WAAW;EACtD,MAAM,QAAQ,OAAO,KAAK;GACxB,MAAM,aAAa,IAAI,WAAW,IAAI,SAAS;GAC/C,IAAI;GACJ,IAAI;IACF,SAAS,MAAM,wBAAwB,OAAO;KAC5C,GAAG,KAAK;KACR;KACA,QAAQ,IAAI;IACd,CAAC;GACH,UAAU;IACR,MAAM,QAAQ,MAAM,iCAAiC,YAAY;KAC/D,SAAS;KACT,WAAW;IACb,CAAC;IACD,IAAI,CAAC,MAAM,SACT,IAAI,MAAM,wDAAwD;KAChE,eAAe,MAAM;KACrB,YAAY;IACd,CAAC;IAEH,IAAI,cAAc,MAAM,OAAO;GACjC;GACA,IAAI,CAAC,OAAO,WACV,OAAO,CACL,YAAY;IACV,YAAY;IACZ;IACA,OAAO;IACP,WAAW,OAAO;IAClB,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;GACnC,CAAC,CACH;GAEF,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,KAAK,OAAO,UAAU;IAG/B,IAAI,EAAE,WAAW,EAAE,SAAS,GAAG;IAC/B,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,EAAE;KACX,OAAO,EAAE,UACL,YAAY,EAAE,QAAQ,aAAa,EAAE,MAAM,QAC3C,YAAY,EAAE,QAAQ;KAC1B,WAAW,EAAE;KACb,UAAU,aAAa,EAAE,QAAQ;KACjC,YAAY;KACZ,eAAe,CAAC;MAAE,MAAM;MAAY,KAAK;MAAmB,SAAS,EAAE;KAAS,CAAC;KACjF,UAAU;MACR,SAAS,EAAE;MACX,SAAS,EAAE;MACX,UAAU,EAAE;KACd;IACF,CAAC,CACH;GACF;GACA,OAAO;EACT;CACF;AACF"}
@@ -1,8 +1,8 @@
1
1
  import { a as RunRecord } from "./run-record-DcObtIGh.js";
2
- import { i as AnalystRegistry } from "./default-registry-Cl3pHo4n.js";
2
+ import { i as AnalystRegistry } from "./default-registry-Brxr728w.js";
3
3
  import { a as DatasetScenario } from "./dataset-BvtnC8Dc.js";
4
- import "./summary-report-CFnQgNfg.js";
5
- import { C as InsightReport, b as ExecutionInsight, v as CostProvenanceSummary } from "./client-COvaLoQG.js";
4
+ import "./summary-report-DGp0-_XO.js";
5
+ import { C as InsightReport, b as ExecutionInsight, v as CostProvenanceSummary } from "./client-BIyh1RCr.js";
6
6
  //#region src/contract/analyze-runs.d.ts
7
7
  interface AnalyzeRunsOptions {
8
8
  /** The runs to analyze. */
@@ -69,4 +69,4 @@ declare function summarizeExecution(opts: SummarizeExecutionOptions): ExecutionR
69
69
  declare function analyzeRuns(opts: AnalyzeRunsOptions): Promise<InsightReport>;
70
70
  //#endregion
71
71
  export { summarizeExecution as a, analyzeRuns as i, ExecutionReport as n, SummarizeExecutionOptions as r, AnalyzeRunsOptions as t };
72
- //# sourceMappingURL=analyze-runs-BmX-h_yn.d.ts.map
72
+ //# sourceMappingURL=analyze-runs-DMo3Lb_y.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"analyze-runs-BmX-h_yn.d.ts","names":[],"sources":["../src/contract/analyze-runs.ts"],"mappings":";;;;;;UAmEiB;;EAEf,MAAM;;;EAGN;;;;;EAKA;EACA;;;EAGA,kBAAkB;;;EAGlB,UAAU;;;;EAIV;IACE;IACA,cAAc;;;;;EAKhB,cAAc;IAAQ;IAAe;IAAe;;;EAEpD;;;;EAIA;;;;;;;;;EASA,eAAe;;;EAGf;;UAGe;EACf,MAAM;EACN;;UAGe;EACf,WAAW;EACX,gBAAgB;;;iBAIF,mBAAmB,MAAM,4BAA4B;iBAS/C,YAAY,MAAM,qBAAqB,QAAQ"}
1
+ {"version":3,"file":"analyze-runs-DMo3Lb_y.d.ts","names":[],"sources":["../src/contract/analyze-runs.ts"],"mappings":";;;;;;UAoEiB;;EAEf,MAAM;;;EAGN;;;;;EAKA;EACA;;;EAGA,kBAAkB;;;EAGlB,UAAU;;;;EAIV;IACE;IACA,cAAc;;;;;EAKhB,cAAc;IAAQ;IAAe;IAAe;;;EAEpD;;;;EAIA;;;;;;;;;EASA,eAAe;;;EAGf;;UAGe;EACf,MAAM;EACN;;UAGe;EACf,WAAW;EACX,gBAAgB;;;iBAIF,mBAAmB,MAAM,4BAA4B;iBAS/C,YAAY,MAAM,qBAAqB,QAAQ"}