@tangle-network/agent-eval 0.133.2 → 0.134.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +218 -0
- package/dist/analyst/index.d.ts +11 -35
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +4 -53
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-BmX-h_yn.d.ts → analyze-runs-DMo3Lb_y.d.ts} +4 -4
- package/dist/{analyze-runs-BmX-h_yn.d.ts.map → analyze-runs-DMo3Lb_y.d.ts.map} +1 -1
- package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
- package/dist/analyze-runs-qk8op0tN.js.map +1 -0
- package/dist/baseline-BaPxoROc.js +149 -0
- package/dist/baseline-BaPxoROc.js.map +1 -0
- package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
- package/dist/baseline-D_fT6277.d.ts.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CJr1H1_a.js → benchmarks-v5piCeDl.js} +3 -3
- package/dist/{benchmarks-CJr1H1_a.js.map → benchmarks-v5piCeDl.js.map} +1 -1
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +5 -4
- package/dist/campaign/index.js +4 -3
- package/dist/{campaign-BJjn1rhw.js → campaign-DEC_7DLn.js} +12 -6
- package/dist/{campaign-BJjn1rhw.js.map → campaign-DEC_7DLn.js.map} +1 -1
- package/dist/{client-COvaLoQG.d.ts → client-BIyh1RCr.d.ts} +29 -15
- package/dist/client-BIyh1RCr.d.ts.map +1 -0
- package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
- package/dist/client-LIuo-KPv.js.map +1 -0
- package/dist/contract/index.d.ts +12 -11
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +12 -18
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-Cl3pHo4n.d.ts → default-registry-Brxr728w.d.ts} +4 -268
- package/dist/default-registry-Brxr728w.d.ts.map +1 -0
- package/dist/{default-registry-D3T9XbuY.js → default-registry-IjYs7T8l.js} +4 -61
- package/dist/default-registry-IjYs7T8l.js.map +1 -0
- package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
- package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-C7Wue8R6.d.ts → index-BoJNQR6n.d.ts} +29 -11
- package/dist/index-BoJNQR6n.d.ts.map +1 -0
- package/dist/{index-BREtv3ZZ.d.ts → index-C21xKtxu.d.ts} +4 -4
- package/dist/{index-BREtv3ZZ.d.ts.map → index-C21xKtxu.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
- package/dist/index-DSC51roc2.d.ts.map +1 -0
- package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
- package/dist/index-DuhJaaiH.d.ts.map +1 -0
- package/dist/index.d.ts +60 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +134 -27
- package/dist/index.js.map +1 -1
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
- package/dist/ledger-core-DAKFKRzi.js.map +1 -0
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
- package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
- package/dist/pipelines/index.d.ts +1 -1
- package/dist/pipelines/index.js +3 -2
- package/dist/pipelines/index.js.map +1 -1
- package/dist/proposal-findings-DCawte-y.js +164 -0
- package/dist/proposal-findings-DCawte-y.js.map +1 -0
- package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
- package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
- package/dist/{release-report-CjHWa8Ia.d.ts → release-report-CuULWKyk.d.ts} +2 -2
- package/dist/{release-report-CjHWa8Ia.d.ts.map → release-report-CuULWKyk.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +4 -4
- package/dist/{researcher-CbSKhK8z.d.ts → researcher-DVtruQ9U.d.ts} +2 -2
- package/dist/{researcher-CbSKhK8z.d.ts.map → researcher-DVtruQ9U.d.ts.map} +1 -1
- package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
- package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
- package/dist/rl.d.ts +2 -2
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +18 -5
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
- package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
- package/dist/{semantic-concept-judge-BypLt6Fw.js → semantic-concept-judge-C0P1VTXD.js} +2 -3
- package/dist/{semantic-concept-judge-BypLt6Fw.js.map → semantic-concept-judge-C0P1VTXD.js.map} +1 -1
- package/dist/{skill-usage-BaaxFSJR.d.ts → skill-usage-BDQVPIG1.d.ts} +3 -2
- package/dist/skill-usage-BDQVPIG1.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-CF6a327Q.js → skillopt-optimization-method-BY6vKLJB.js} +169 -53
- package/dist/skillopt-optimization-method-BY6vKLJB.js.map +1 -0
- package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts → skillopt-optimization-method-DJ3l4w8W.d.ts} +20 -18
- package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +1 -0
- package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
- package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
- package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
- package/dist/statistics-RwRNu2__.js.map +1 -0
- package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
- package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
- package/dist/{summary-report-CFnQgNfg.d.ts → summary-report-DGp0-_XO.d.ts} +61 -4
- package/dist/summary-report-DGp0-_XO.d.ts.map +1 -0
- package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
- package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
- package/dist/types-DVjczBM9.d.ts +276 -0
- package/dist/types-DVjczBM9.d.ts.map +1 -0
- package/dist/{types-BokuXvOG.d.ts → types-DiWLru6Z.d.ts} +20 -37
- package/dist/types-DiWLru6Z.d.ts.map +1 -0
- package/docs/campaign-proposers.md +5 -0
- package/docs/design/statistics-decisions.md +271 -0
- package/docs/design.md +1 -0
- package/docs/insight-report.md +1 -1
- package/docs/research-report-methodology.md +4 -1
- package/package.json +2 -1
- package/dist/analyze-runs-B-afTpCv.js.map +0 -1
- package/dist/baseline-DcX5hQDv.js.map +0 -1
- package/dist/baseline-hG3K85h4.d.ts.map +0 -1
- package/dist/client-COvaLoQG.d.ts.map +0 -1
- package/dist/client-CYzbdJOZ.js.map +0 -1
- package/dist/default-registry-Cl3pHo4n.d.ts.map +0 -1
- package/dist/default-registry-D3T9XbuY.js.map +0 -1
- package/dist/index-C7Wue8R6.d.ts.map +0 -1
- package/dist/index-DSC51roc.d.ts.map +0 -1
- package/dist/index-nhIYz9hn.d.ts.map +0 -1
- package/dist/ledger-core-CPZfcrC2.js.map +0 -1
- package/dist/run-score-iEEAWiBY.js +0 -41
- package/dist/run-score-iEEAWiBY.js.map +0 -1
- package/dist/skill-usage-BaaxFSJR.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CF6a327Q.js.map +0 -1
- package/dist/skillopt-optimization-method-wHF5xsUv.d.ts.map +0 -1
- package/dist/statistics-DWM_AyLe.js.map +0 -1
- package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
- package/dist/summary-report-CFnQgNfg.d.ts.map +0 -1
- package/dist/types-BokuXvOG.d.ts.map +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,224 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.134.0] - 2026-07-28 - isolated proposal inputs
|
|
8
|
+
|
|
9
|
+
### Changed
|
|
10
|
+
|
|
11
|
+
- `runOptimization()` now accepts only findings labeled from search runs or observed production behavior.
|
|
12
|
+
- `ProposalFinding` carries the required `proposal_origin`.
|
|
13
|
+
- Removed opaque `report` and capture-store access from `ProposeContext`; final evaluation data has no internal path into candidate generation.
|
|
14
|
+
- Removed `assertNoJudgeVerdict`, `isJudgeVerdict`, and `isTraceObservable`; use validated `ProposalFinding` inputs instead.
|
|
15
|
+
- `runOptimization()` snapshots its baseline and candidate outputs before measurement.
|
|
16
|
+
|
|
17
|
+
## [0.133.3] - 2026-07-27 - trustworthy statistical decisions
|
|
18
|
+
|
|
19
|
+
### Consumer notice — reported p-values were too small in every release from 0.1.0 to 0.133.0
|
|
20
|
+
|
|
21
|
+
0.133.1 corrected the standard-normal CDF. This release carries the rest, and
|
|
22
|
+
restates the notice because the re-check bands and the affected version range are
|
|
23
|
+
what a consumer actually needs.
|
|
24
|
+
|
|
25
|
+
A standard-normal CDF mixed the arguments of the Abramowitz–Stegun error-function
|
|
26
|
+
approximation, giving up to `3.7189e-2` absolute CDF error where a correct
|
|
27
|
+
implementation is bounded by `7.5e-8`. Every p-value routed through it was too
|
|
28
|
+
small by 26–36 % relative, so the module's real type-I error rate was **6.53 % at
|
|
29
|
+
a nominal 5 %** and **1.34 % at a nominal 1 %**. The defect entered at the initial
|
|
30
|
+
commit (`7d5032b`, 2026-04-20) and shipped in every release through 0.133.0.
|
|
31
|
+
|
|
32
|
+
**Affected:** `mannWhitneyU`, `wilcoxonSignedRank`, `pairedTTest`, `welchsTTest`,
|
|
33
|
+
`compareToBaseline`, `mcnemarPower`.
|
|
34
|
+
|
|
35
|
+
**Unaffected** (verified numerically identical before and after, because they route
|
|
36
|
+
through the inverse normal rather than the forward CDF): `requiredSampleSize`,
|
|
37
|
+
`requiredPairedSampleSize`, `pairedMde`, `mcnemarRequiredN`, `mcnemar`,
|
|
38
|
+
`pairedSignTest`, `wilson`, `passAtK`, `corpusInterRaterAgreement`, `eProcess`,
|
|
39
|
+
`holm`, `ranks`, `pearsonR`, `spearmanR`, `cliffsDelta`, `pairedCohensDz`.
|
|
40
|
+
|
|
41
|
+
**How to re-check a decision you already made.** The error is monotone in `|z|`, so
|
|
42
|
+
the affected band is exact and narrow:
|
|
43
|
+
|
|
44
|
+
- Any recorded p in `[0.038053, 0.050000)` crossed a 5 % gate it should not have.
|
|
45
|
+
- At `α = 0.01` the band is `[0.007443, 0.010000)`; at `α = 0.10`, `[0.077398, 0.100000)`.
|
|
46
|
+
- A recorded p below `0.038053` was significant either way; at or above `0.05`, not
|
|
47
|
+
significant either way. Neither needs re-checking.
|
|
48
|
+
|
|
49
|
+
Three further cautions, independent of the CDF:
|
|
50
|
+
|
|
51
|
+
- A `wilcoxonSignedRank` leg that reported `p = 1` on fewer than six non-zero
|
|
52
|
+
differences measured nothing — the function hard-returned `p = 1` there with no
|
|
53
|
+
flag. A clean 5-of-5 shift reported `1.0` where the exact answer is `0.0625`.
|
|
54
|
+
Exact ties are dropped before ranking, so ten pairs with five tied deltas also
|
|
55
|
+
fell into that branch. Re-run those on this release.
|
|
56
|
+
- A promotion that turned on a `pairedBootstrap` `low > 0` check below 20 pairs was
|
|
57
|
+
never valid at the stated confidence: measured false-positive rate is 13.53 % at
|
|
58
|
+
`n = 3` against a nominal 2.5 %. `gateEligible` now reports this.
|
|
59
|
+
- A bootstrap interval recorded through `analyze-runs.ts` at or before 0.133.0 is
|
|
60
|
+
not reproducible — that call site passed no seed and `makeRng` fell back to
|
|
61
|
+
`Math.random`.
|
|
62
|
+
|
|
63
|
+
Full evidence, per-statistic verdicts, and the dependency argument:
|
|
64
|
+
[`docs/design/statistics-decisions.md`](./docs/design/statistics-decisions.md).
|
|
65
|
+
|
|
66
|
+
### Consumer notice — `HeldOutGate` promoted candidates that scored nothing on most held-out items
|
|
67
|
+
|
|
68
|
+
Every release through 0.133.2 decided a promotion over the items where BOTH arms
|
|
69
|
+
produced a finite score, and dropped the rest without a word. An item the candidate
|
|
70
|
+
crashed on, timed out on, or wrote no row for simply left the comparison.
|
|
71
|
+
|
|
72
|
+
**Measured, deterministic fixture, no model calls:** 26 held-out items, a candidate
|
|
73
|
+
that produced no score at all on 20 of them and 0.95 on the 6 it answered against a
|
|
74
|
+
0.60 baseline. Published 0.133.2 and `origin/main` PROMOTE it at every threshold
|
|
75
|
+
from −0.05 to +0.30 — `productiveRuns: 6`, `unpairedBaselineRuns: 20` sitting in the
|
|
76
|
+
evidence, read by nothing. The control, the same 20 failures scored as the 0 they
|
|
77
|
+
earned, is correctly refused at a mean paired delta of −0.3808. An agent that failed
|
|
78
|
+
77 % of its tasks was promoted, and the paired-delta, overfit-gap and cost gates all
|
|
79
|
+
sat behind that filter.
|
|
80
|
+
|
|
81
|
+
A second shape, same cause: a crashed first attempt plus a scored retry at the same
|
|
82
|
+
`(experimentId, scenarioId, seed)` PROMOTED on 0.133.2, even though the gate's own
|
|
83
|
+
docstring says duplicate identities throw — the crashed row was filtered out before
|
|
84
|
+
the duplicate could be seen. It now throws, exactly as two scored rows already did.
|
|
85
|
+
|
|
86
|
+
**How to re-check a decision you already made.** Read `unpairedBaselineRuns` and
|
|
87
|
+
`unpairedCandidateRuns` on any recorded `GateDecision`: a nonzero value means the
|
|
88
|
+
verdict was computed over a subset. `productiveRuns` below the number of held-out
|
|
89
|
+
items you dealt means the same thing. Those promotions are not valid at the stated
|
|
90
|
+
threshold and should be re-run on this release.
|
|
91
|
+
|
|
92
|
+
### Fixed
|
|
93
|
+
|
|
94
|
+
- `HeldOutGate`'s cost median is taken over the rows that DECIDED the verdict — the
|
|
95
|
+
matched pairs on both splits — instead of over every row the caller passed. The old
|
|
96
|
+
population was a denominator nobody measured and was trivially movable: 48 rows tagged
|
|
97
|
+
`dev` at \$0.0001, which the gate never scores, drag a real \$5.00/task candidate to a
|
|
98
|
+
reported \$0.0001 and clear a \$1.00 `costPerTaskCeiling`. Measured on `origin/main`
|
|
99
|
+
(2789970): 12 fully-covered items at \$5.00/task, ceiling \$1.00 — 0 pad rows rejects
|
|
100
|
+
with `cost_ceiling`, 24 pad rows reports \$2.50005, 48 pad rows reports \$0.0001 and
|
|
101
|
+
PROMOTES. The population is derived from the pairing rather than from a list of split
|
|
102
|
+
tags, so there is no tag that sits outside the rule. On a comparison with no rows
|
|
103
|
+
outside the two decided splits the reported number is unchanged.
|
|
104
|
+
- `HeldOutGate` requires COVERAGE before it decides anything: on both the search and
|
|
105
|
+
the holdout split, `answered / dealt` must be at least the new `minCoverage`
|
|
106
|
+
(default **1** — every item the comparison was dealt carries a real score on both
|
|
107
|
+
arms), else it refuses with the new `incomplete_coverage` rejection code. The
|
|
108
|
+
denominator is measured, not declared: it is what `pairRunRecords` reports when
|
|
109
|
+
given every row of a split rather than only the scored ones, so an item counts as
|
|
110
|
+
dealt because a row for it exists on at least one arm. The gate does not impute a
|
|
111
|
+
value for a missing score — it does not know the failure value of the caller's
|
|
112
|
+
metric, and a caller who does knows to write it onto the record before calling.
|
|
113
|
+
`GateEvidence` gains `holdoutCoverage` and `searchCoverage` (`SplitCoverage`:
|
|
114
|
+
`dealt`, `answered`, `unscoredPairs`, `candidateOnly`, `baselineOnly`, `coverage`),
|
|
115
|
+
so a shrunken denominator can never be read without seeing it.
|
|
116
|
+
Verified monotone against `origin/main` over 6000 randomised comparisons: on
|
|
117
|
+
complete inputs the verdict, rejection code, CI and n are identical in 3000/3000
|
|
118
|
+
cases; across both sweeps there are 0 cases where the new gate promotes something
|
|
119
|
+
the old one refused.
|
|
120
|
+
- `regularizedIncompleteBeta` takes the mandatory symmetry branch `I_x(a,b) = 1 − I_{1−x}(b,a)`.
|
|
121
|
+
`studentTCdf(0.005, 100)` returned `0.89152130` against a true `0.50198972`; a
|
|
122
|
+
perfectly null paired result reported `p < 0.05`. This survived 0.133.1, which
|
|
123
|
+
corrected the normal CDF but not the beta continued fraction beneath it, and
|
|
124
|
+
0.133.2, which changed no statistics math.
|
|
125
|
+
- `mannWhitneyU` and `wilcoxonSignedRank` compute an EXACT conditional p by default
|
|
126
|
+
inside the enumeration thresholds, conditioning on the observed tie pattern.
|
|
127
|
+
- `mannWhitneyU` chooses exact computation from bounded state and work estimates,
|
|
128
|
+
so imbalanced designs such as 1 v 24 remain exact without admitting expensive
|
|
129
|
+
balanced designs. Its automatic permutation seed is invariant to observation
|
|
130
|
+
order and to swapping the two groups.
|
|
131
|
+
- Deleted `wilcoxonSignedRank`'s `n < 6` hard return of `{w: 0, p: 1}`.
|
|
132
|
+
- The asymptotic rank-test path applies the tie correction and the continuity
|
|
133
|
+
correction; both were missing.
|
|
134
|
+
- `mannWhitneyU` and `wilcoxonSignedRank` reject non-finite input. A single `NaN`
|
|
135
|
+
previously spun forever: the tie-grouping loop advanced on `===`, and
|
|
136
|
+
`NaN === NaN` is false.
|
|
137
|
+
- `bonferroni` and `benjaminiHochberg` reject at the inclusive boundary (`p ≤ α`,
|
|
138
|
+
`q ≤ fdr`), matching `holm`, and validate `alpha`/`fdr` and the p-value range.
|
|
139
|
+
`bonferroni([0.0125]×4, 0.05)` returned all-false where `holm` returned all-true.
|
|
140
|
+
BH q-values use R's `(n/rank)·p` form; `(p·n)/rank` lands one ULP above the
|
|
141
|
+
boundary at `p = 0.05, n = 3`.
|
|
142
|
+
- `interRaterReliability` groups by (dimension, item) across judges. It was
|
|
143
|
+
bucketing consecutive scores from the SAME judge, so it measured within-judge
|
|
144
|
+
spread: two identical judges returned `−0.5` where the true α is `+1.0`.
|
|
145
|
+
- `mulberry32(0)` is its own stream. `seed | 0 || 0x9e3779b9` collapsed seed 0 onto
|
|
146
|
+
the golden-ratio constant, so two runs a caller believed were independent
|
|
147
|
+
replicates were the same run. Non-finite seeds now throw.
|
|
148
|
+
- Unseeded bootstraps derive their seed from the data instead of `Math.random`, so
|
|
149
|
+
an interval is reproducible whether or not the caller passes a seed.
|
|
150
|
+
- `studentTCdf` uses the regularized incomplete beta for every finite degree of
|
|
151
|
+
freedom. The deleted normal shortcut changed `df = 102, t = 1.98` from the true
|
|
152
|
+
two-sided `p = 0.050398` to `0.047703`.
|
|
153
|
+
- At the default 95% confidence, campaign promotion decisions use an exact
|
|
154
|
+
one-sided sign test from 6 through 19 paired observations and the bootstrap
|
|
155
|
+
interval from 20 onward. Samples too small to attain the requested confidence
|
|
156
|
+
remain inconclusive.
|
|
157
|
+
- Prior-period reports use the shared Welch implementation. Zero-variance and
|
|
158
|
+
under-sized comparisons carry an explicit status and null inferential fields
|
|
159
|
+
instead of fabricated `p = 1`, `d = 0`, and a zero-width interval.
|
|
160
|
+
|
|
161
|
+
### Changed — BREAKING
|
|
162
|
+
|
|
163
|
+
- `mannWhitneyU(a, b, opts?)` returns `{ u, uA, p, method, pFloor }`. `p` is now the
|
|
164
|
+
exact conditional p inside the threshold: `mannWhitneyU([1,2,3],[4,5,6]).p` moves
|
|
165
|
+
from `0.03769147` to `0.10000000`, which is the smallest p attainable at 3 v 3.
|
|
166
|
+
`uA` carries the direction that `u = min(u₁,u₂)` discards.
|
|
167
|
+
- `wilcoxonSignedRank(before, after, opts?)` returns
|
|
168
|
+
`{ w, p, method, pFloor, nNonZero }`.
|
|
169
|
+
- Both take `method: 'auto' | 'exact' | 'asymptotic'`, default `'auto'`. `'auto'`
|
|
170
|
+
never selects `'asymptotic'`. Requesting `'asymptotic'` where an exact answer
|
|
171
|
+
exists THROWS a `ValidationError` naming the attainable floor, and requesting
|
|
172
|
+
`'exact'` above the threshold throws rather than enumerating an unbounded
|
|
173
|
+
distribution.
|
|
174
|
+
- `pairedTTest` returns `{ t: number | null, df, p: number | null }`. A non-zero
|
|
175
|
+
constant delta returned `{t: Infinity, p: 0}` — absolute certainty from three
|
|
176
|
+
observations — and now returns null, matching `pairedCohensDz`. An all-zero delta
|
|
177
|
+
is still `{t: 0, p: 1}`.
|
|
178
|
+
- `cohensD` returns `number | null`. It returned a silent `0` for a maximal
|
|
179
|
+
zero-variance separation and for under-sized groups.
|
|
180
|
+
- `MetricVerdict.cohensD` and `LiftInsight.pValue` are nullable accordingly.
|
|
181
|
+
- `PairedBootstrapResult` carries `gateEligible`, false below
|
|
182
|
+
`BOOTSTRAP_GATE_MIN_N = 20`.
|
|
183
|
+
- `welchsTTest` returns a status-tagged full result with means, delta, standard
|
|
184
|
+
error, degrees of freedom, Student-t interval, p-value, and Cohen's d.
|
|
185
|
+
- Prior-period `MetricDelta` values carry the same status. Invalid inference has
|
|
186
|
+
null `ci95`, `pValue`, and `cohensD`, and the comparison lists those metric
|
|
187
|
+
names in `inconclusiveMetrics`.
|
|
188
|
+
|
|
189
|
+
### Added
|
|
190
|
+
|
|
191
|
+
- `scripts/generate-statistics-oracle.py` + `tests/fixtures/statistics-oracle.json`:
|
|
192
|
+
154 scipy/statsmodels-generated golden values across 22 statistics, asserted by
|
|
193
|
+
`tests/statistics-oracle.test.ts`. scipy is a CI oracle and is never a runtime
|
|
194
|
+
dependency.
|
|
195
|
+
- `tests/statistics-library-crosscheck.test.ts` cross-checks untied exact null
|
|
196
|
+
distributions against `lib-r-math.js`. Multiple-comparison functions are pinned
|
|
197
|
+
to statsmodels-generated fixture values; `@stdlib/stats-padjust` was removed.
|
|
198
|
+
- First test coverage for `welchsTTest` / `compareToBaseline`, which gated
|
|
199
|
+
improved / regressed / stable verdicts with nothing asserting their numbers.
|
|
200
|
+
- Exported `normalCdf`, `studentTCdf`, `BOOTSTRAP_GATE_MIN_N`,
|
|
201
|
+
`MANN_WHITNEY_EXACT_MAX_STATES`, `MANN_WHITNEY_EXACT_MAX_WORK`,
|
|
202
|
+
`WILCOXON_EXACT_MAX_N`, `DEFAULT_PERMUTATIONS`, and `pairedDeltaTest`.
|
|
203
|
+
|
|
204
|
+
### Trusted-head recovery
|
|
205
|
+
|
|
206
|
+
#### Fixed
|
|
207
|
+
|
|
208
|
+
- A trusted-head pin write that fails after its journal row is already durable no longer leaves that row permanently unpinned.
|
|
209
|
+
Retrying the same `eventId` moves the pin up to the acknowledged entry when the entry is ahead of the pin, so the last row of a ledger — the promotion decision — cannot be truncated away undetected after a full disk or a read-only mount.
|
|
210
|
+
- A pin whose journal was deleted or rebuilt is recoverable instead of refusing every later append and replay forever.
|
|
211
|
+
The refusal still stands, since a missing journal beside a live pin is the deletion the pin exists to catch, but it now names the sidecar file and the operation that resolves it.
|
|
212
|
+
- Trusted-head read and write faults are reported through the journal codec's error taxonomy instead of escaping as raw Node filesystem errors.
|
|
213
|
+
|
|
214
|
+
#### Added
|
|
215
|
+
|
|
216
|
+
- `clearTrustedHeadFile`, exposed as `FileLedgerJournal.clearTrustedHead()` and `SearchLedger.clearTrustedHead()`, discards a pin and returns the guarantee it gave up.
|
|
217
|
+
- `SearchLedger.pinTrustedHead()`, the campaign-level route to adopting a pin for a ledger that has none.
|
|
218
|
+
- `readTrustedHeadFile` is exported from `@tangle-network/agent-eval/ledger-core`, so a consumer of `verifyEntriesAgainstTrustedHead` has a shape-validating way to load a pin.
|
|
219
|
+
|
|
220
|
+
#### Changed
|
|
221
|
+
|
|
222
|
+
- **Breaking:** `verifyEntriesAgainstTrustedHead` takes `{ subject, trustedHeadPath }` as its fourth argument instead of a bare `subject` string, so every refusal can name the sidecar file. Passing the old string throws a `TypeError`.
|
|
223
|
+
- **Breaking:** `SearchLedger` declares `pinTrustedHead` and `clearTrustedHead`; an external implementation of the interface must supply them.
|
|
224
|
+
|
|
7
225
|
## [0.133.2] - 2026-07-27 - protect final evaluation data
|
|
8
226
|
|
|
9
227
|
### Fixed
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
import { a as MultiLayerVerifier, l as VerifyOptions, o as Severity } from "../multi-layer-verifier-BHY1gWAc.js";
|
|
2
|
-
import { C as FindingSubjectKind, D as parseFindingSubject, E as findingSubjectGrammarPromptFor, F as createAnalystAi, H as SemanticConceptJudgeInput, O as renderFindingSubject, P as CreateAnalystAiConfig, S as FindingSubject, T as KIND_EXPECTED_SUBJECTS, U as SemanticConceptJudgeOptions, Y as RunTrace, _ as defaultIsMaterial, a as SkillUsageScanConfig, b as FINDING_SUBJECT_KINDS, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as PersistedFinding, h as FindingsStore, i as SkillUsageReport, k as BehavioralMetrics, l as KNOWLEDGE_POISONING_KIND_SPEC, m as FindingsDiff, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as DiffPolicy, q as RunCritic, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as diffFindings, w as FindingSubjectStringSchema, x as FINDING_SUBJECT_SYNTAX, y as FINDING_SUBJECT_GRAMMAR_PROMPT } from "../skill-usage-
|
|
2
|
+
import { C as FindingSubjectKind, D as parseFindingSubject, E as findingSubjectGrammarPromptFor, F as createAnalystAi, H as SemanticConceptJudgeInput, O as renderFindingSubject, P as CreateAnalystAiConfig, S as FindingSubject, T as KIND_EXPECTED_SUBJECTS, U as SemanticConceptJudgeOptions, Y as RunTrace, _ as defaultIsMaterial, a as SkillUsageScanConfig, b as FINDING_SUBJECT_KINDS, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as PersistedFinding, h as FindingsStore, i as SkillUsageReport, k as BehavioralMetrics, l as KNOWLEDGE_POISONING_KIND_SPEC, m as FindingsDiff, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as DiffPolicy, q as RunCritic, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as diffFindings, w as FindingSubjectStringSchema, x as FINDING_SUBJECT_SYNTAX, y as FINDING_SUBJECT_GRAMMAR_PROMPT } from "../skill-usage-BDQVPIG1.js";
|
|
3
3
|
import { c as CostLedgerHandle } from "../cost-ledger-fGS_u_O1.js";
|
|
4
4
|
import { o as LlmClientOptions } from "../llm-client-BiK4HW0u.js";
|
|
5
5
|
import { A as ChatClient, B as SandboxSdkTransportOpts, F as CreateChatClientOpts, I as CustomTransportOpts, L as DirectProviderTransportOpts, M as ChatResponse, N as ChatTransport, P as CliBridgeTransportOpts, R as MockTransportOpts, V as createChatClient, j as ChatRequest, k as ChatCallOpts, m as JudgeInput, p as JudgeFn, z as RouterTransportOpts } from "../types-Cc3qbqzj.js";
|
|
6
6
|
import { t as TraceAnalysisStore } from "../store-CxJry_cs.js";
|
|
7
|
-
import {
|
|
7
|
+
import { _ as makeFinding, a as AnalystInputKind, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as computeFindingId, h as ProposalFindingOrigin, i as AnalystFinding, l as AnalystRunResult, m as ProposalFinding, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as makeProposalFinding } from "../types-DVjczBM9.js";
|
|
8
|
+
import { _ as RawAnalystEvidenceSchema, a as AnalystRegistryOptions, b as evidenceRefsFromRawFinding, c as CreateTraceAnalystKindOpts, d as createTraceAnalystKind, f as renderPriorFindings, g as RawAnalystEvidence, h as RAW_FINDING_SCHEMA_PROMPT, i as AnalystRegistry, l as TraceAnalystGolden, m as ANALYST_SEVERITIES, n as buildDefaultAnalystRegistry, o as BudgetPolicy, p as renderUpstreamFindings, r as AnalystHooks, s as RegistryRunOpts, t as DefaultAnalystRegistryOptions, u as TraceAnalystKindSpec, v as RawAnalystFinding, x as parseRawFinding, y as RawAnalystFindingSchema } from "../default-registry-Brxr728w.js";
|
|
8
9
|
import { AxFunction } from "@ax-llm/ax";
|
|
9
10
|
//#region src/analyst/adapters.d.ts
|
|
10
11
|
declare function liftSeverity(s: Severity): AnalystSeverity;
|
|
@@ -88,40 +89,15 @@ declare function coerceJson(text: string): unknown;
|
|
|
88
89
|
*/
|
|
89
90
|
declare function coerceToFindingRows(raw: unknown): unknown[];
|
|
90
91
|
//#endregion
|
|
91
|
-
//#region src/analyst/
|
|
92
|
-
/**
|
|
93
|
-
|
|
94
|
-
* rendering — it is NOT the steer gate. Evidence presence is the WRONG
|
|
95
|
-
* discriminator for steering: a legitimate trace-analyst observation may cite
|
|
96
|
-
* nothing (it would be wrongly rejected), and a judge verdict may cite an
|
|
97
|
-
* artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate
|
|
98
|
-
* steering; use this only where "is this grounded in observable evidence" is the
|
|
99
|
-
* literal question. */
|
|
100
|
-
declare function isTraceObservable(finding: AnalystFinding): boolean;
|
|
101
|
-
/** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a
|
|
102
|
-
* finding), identified by provenance set at the lift site — independent of
|
|
103
|
-
* whatever evidence it cites. */
|
|
104
|
-
declare function isJudgeVerdict(finding: AnalystFinding): boolean;
|
|
92
|
+
//#region src/analyst/proposal-findings.d.ts
|
|
93
|
+
/** True when a finding names a source candidate generation may learn from. */
|
|
94
|
+
declare function isProposalFinding(finding: unknown): finding is ProposalFinding;
|
|
105
95
|
/**
|
|
106
|
-
*
|
|
107
|
-
*
|
|
108
|
-
*
|
|
109
|
-
* loop. Returns the findings unchanged for chaining.
|
|
110
|
-
*
|
|
111
|
-
* Call this at the chokepoint where a detector that ALSO scores/gates has its
|
|
112
|
-
* findings turned into a steer (the judge-and-steer dual-role case). It keys on
|
|
113
|
-
* provenance, so it correctly admits evidence-less trace-analyst observations and
|
|
114
|
-
* correctly rejects an artifact-citing judge verdict — the cases an evidence
|
|
115
|
-
* check gets backwards.
|
|
116
|
-
*
|
|
117
|
-
* It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge
|
|
118
|
-
* whose output is laundered through a hand-built finding with no provenance flag
|
|
119
|
-
* is out of its reach — provenance must be honestly set at every judge→finding
|
|
120
|
-
* lift (today: createJudgeAdapter). That is why the integrity rule lives at the
|
|
121
|
-
* lift site, and why ProposeContext.judgeScores?: never is the complementary
|
|
122
|
-
* compile-time tripwire on the obvious direct channel.
|
|
96
|
+
* Reject findings whose source has not been explicitly admitted for candidate
|
|
97
|
+
* generation. Search feedback and observed production behavior are allowed;
|
|
98
|
+
* final evaluation data has no allowed origin.
|
|
123
99
|
*/
|
|
124
|
-
declare function
|
|
100
|
+
declare function assertProposalFindings(findings: unknown, context?: string): ReadonlyArray<ProposalFinding>;
|
|
125
101
|
//#endregion
|
|
126
102
|
//#region src/analyst/structure-findings.d.ts
|
|
127
103
|
interface StructureFindingsOptions {
|
|
@@ -176,5 +152,5 @@ type TraceToolGroupName =
|
|
|
176
152
|
*/
|
|
177
153
|
declare function buildTraceToolsForGroup(group: TraceToolGroupName, store: TraceAnalysisStore): AxFunction[];
|
|
178
154
|
//#endregion
|
|
179
|
-
export { ANALYST_SEVERITIES, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type BudgetPolicy, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateTraceAnalystKindOpts, type CustomTransportOpts, DEFAULT_TRACE_ANALYST_KINDS, type DefaultAnalystRegistryOptions, type DiffPolicy, type DirectProviderTransportOpts, type EvidenceRef, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type MockTransportOpts, type PersistedFinding, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StructureFindingsOptions, type StructureFindingsResult, type TraceAnalystGolden, type TraceAnalystKindSpec, type TraceToolGroupName, type VerifierAdapterOpts,
|
|
155
|
+
export { ANALYST_SEVERITIES, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type BudgetPolicy, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateTraceAnalystKindOpts, type CustomTransportOpts, DEFAULT_TRACE_ANALYST_KINDS, type DefaultAnalystRegistryOptions, type DiffPolicy, type DirectProviderTransportOpts, type EvidenceRef, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type MockTransportOpts, type PersistedFinding, type ProposalFinding, type ProposalFindingOrigin, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StructureFindingsOptions, type StructureFindingsResult, type TraceAnalystGolden, type TraceAnalystKindSpec, type TraceToolGroupName, type VerifierAdapterOpts, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, makeFinding, makeProposalFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
|
|
180
156
|
//# sourceMappingURL=index.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/behavioral-analyst.ts","../../src/analyst/parse-tolerant.ts","../../src/analyst/
|
|
1
|
+
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/behavioral-analyst.ts","../../src/analyst/parse-tolerant.ts","../../src/analyst/proposal-findings.ts","../../src/analyst/structure-findings.ts","../../src/analyst/tool-groups.ts"],"mappings":";;;;;;;;;;iBA4CgB,aAAa,GAAG,WAAgB;UAe/B,oBAAoB;EACnC;EACA;EACA,UAAU,mBAAmB;;;;;EAK7B,UAAU,KAAK,cAAc;;iBAGf,sBAAsB,KAAK,MAAM,oBAAoB,OAAO,QAAQ;UAwEnE;EACf;EACA;EACA,SAAS;;EAET;;iBAGc,uBAAuB,OAAM,uBAA4B,QAAQ;UAiEhE;EACf;EACA;EACA,OAAO;;EAEP,MAAM;;EAEN,OAAO;;EAEP;;iBAGc,mBAAmB,MAAM,mBAAmB,QAAQ;UAiDnD;EACf;EACA;;EAEA,UAAU,KAAK;;EAEf;;iBAGc,kCACd,OAAM,kCACL,QAAQ;;;;;;;iBC5OK,yBACd,SAAS,mBACT;EAAQ;EAAoB;IAC3B;;iBAkCa,qBAAqB,QAAQ;;;;;;;;;;;;;;;iBC3E7B,gBAAgB;;;;;iBAgBhB,WAAW;;;;;;;iBAeX,oBAAoB;;;;iBCXpB,kBAAkB,mBAAmB,WAAW;;;;;;iBAShD,uBACd,mBACA,mBACC,cAAc;;;UCXA;;EAEf;EACA;;EAEA;EACA;EACA;EACA;;EAEA,aAAa;EACb;EACA,WAAW;EACX;EACA,SAAS;;EAET;;EAEA,cAAc,KAAK,sBAAsB;;EAEzC,kBAAkB;;EAElB,YAAY;;UAGG;EACf,UAAU;EACV;;iBAuCoB,kBACpB,MAAM,2BACL,QAAQ;;;;KCrFC;;;;;;;;;;;;;;;;;;iBAuCI,wBACd,OAAO,oBACP,OAAO,qBACN"}
|
package/dist/analyst/index.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
|
-
import { A as parseFindingSubject, C as stripCodeFences, D as FindingSubjectStringSchema, E as FINDING_SUBJECT_SYNTAX, F as
|
|
1
|
+
import { A as parseFindingSubject, C as stripCodeFences, D as FindingSubjectStringSchema, E as FINDING_SUBJECT_SYNTAX, F as createChatClient, I as createAnalystAi, M as behavioralAnalyst, N as deriveEfficiencyFindings, O as KIND_EXPECTED_SUBJECTS, S as coerceToFindingRows, T as FINDING_SUBJECT_KINDS, _ as RawAnalystEvidenceSchema, a as KNOWLEDGE_GAP_KIND_SPEC, b as parseRawFinding, c as buildTraceToolsForGroup, d as renderUpstreamFindings, f as settleUsageReceiptFromCostLedger, g as RAW_FINDING_SCHEMA_PROMPT, h as ANALYST_SEVERITIES, i as KNOWLEDGE_POISONING_KIND_SPEC, j as renderFindingSubject, k as findingSubjectGrammarPromptFor, l as createTraceAnalystKind, m as structureFindings, n as AnalystRegistry, o as IMPROVEMENT_KIND_SPEC, p as validateUsageSettlementTimeout, r as DEFAULT_TRACE_ANALYST_KINDS, s as FAILURE_MODE_KIND_SPEC, t as buildDefaultAnalystRegistry, u as renderPriorFindings, v as RawAnalystFindingSchema, w as FINDING_SUBJECT_GRAMMAR_PROMPT, x as coerceJson, y as evidenceRefsFromRawFinding } from "../default-registry-IjYs7T8l.js";
|
|
2
2
|
import { i as CostLedger } from "../cost-ledger-BrJxbrMy.js";
|
|
3
|
-
import {
|
|
3
|
+
import { c as makeFinding, l as makeProposalFinding, n as isProposalFinding, s as computeFindingId, t as assertProposalFindings } from "../proposal-findings-DCawte-y.js";
|
|
4
|
+
import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, l as emitSkillUsageFindings, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-C0P1VTXD.js";
|
|
4
5
|
//#region src/analyst/adapters.ts
|
|
5
6
|
/**
|
|
6
7
|
* Adapter factories — lift each existing agent-eval primitive into the
|
|
@@ -288,56 +289,6 @@ function createSemanticConceptJudgeAdapter(opts = {}) {
|
|
|
288
289
|
};
|
|
289
290
|
}
|
|
290
291
|
//#endregion
|
|
291
|
-
|
|
292
|
-
/** Evidence grounded in the agent's OWN execution: OTLP trace elements
|
|
293
|
-
* (`span`/`event`) or the artifact it produced (`artifact`). */
|
|
294
|
-
const OBSERVABLE_KINDS = /* @__PURE__ */ new Set([
|
|
295
|
-
"span",
|
|
296
|
-
"event",
|
|
297
|
-
"artifact"
|
|
298
|
-
]);
|
|
299
|
-
/** DESCRIPTIVE predicate: does the finding cite at least one observable
|
|
300
|
-
* (span/event/artifact) evidence ref. Useful for ranking evidence quality or
|
|
301
|
-
* rendering — it is NOT the steer gate. Evidence presence is the WRONG
|
|
302
|
-
* discriminator for steering: a legitimate trace-analyst observation may cite
|
|
303
|
-
* nothing (it would be wrongly rejected), and a judge verdict may cite an
|
|
304
|
-
* artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate
|
|
305
|
-
* steering; use this only where "is this grounded in observable evidence" is the
|
|
306
|
-
* literal question. */
|
|
307
|
-
function isTraceObservable(finding) {
|
|
308
|
-
return finding.evidence_refs.some((ref) => OBSERVABLE_KINDS.has(ref.kind));
|
|
309
|
-
}
|
|
310
|
-
/** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a
|
|
311
|
-
* finding), identified by provenance set at the lift site — independent of
|
|
312
|
-
* whatever evidence it cites. */
|
|
313
|
-
function isJudgeVerdict(finding) {
|
|
314
|
-
return finding.derived_from_judge === true;
|
|
315
|
-
}
|
|
316
|
-
/**
|
|
317
|
-
* THE steer firewall. Fail-loud guard for any path that admits analyst findings
|
|
318
|
-
* as STEERING input (the `f(trace)` role): rejects — naming the offenders — any
|
|
319
|
-
* finding whose provenance is a judge verdict, rather than let `J` leak into the
|
|
320
|
-
* loop. Returns the findings unchanged for chaining.
|
|
321
|
-
*
|
|
322
|
-
* Call this at the chokepoint where a detector that ALSO scores/gates has its
|
|
323
|
-
* findings turned into a steer (the judge-and-steer dual-role case). It keys on
|
|
324
|
-
* provenance, so it correctly admits evidence-less trace-analyst observations and
|
|
325
|
-
* correctly rejects an artifact-citing judge verdict — the cases an evidence
|
|
326
|
-
* check gets backwards.
|
|
327
|
-
*
|
|
328
|
-
* It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge
|
|
329
|
-
* whose output is laundered through a hand-built finding with no provenance flag
|
|
330
|
-
* is out of its reach — provenance must be honestly set at every judge→finding
|
|
331
|
-
* lift (today: createJudgeAdapter). That is why the integrity rule lives at the
|
|
332
|
-
* lift site, and why ProposeContext.judgeScores?: never is the complementary
|
|
333
|
-
* compile-time tripwire on the obvious direct channel.
|
|
334
|
-
*/
|
|
335
|
-
function assertNoJudgeVerdict(findings, context = "steer") {
|
|
336
|
-
const leaks = findings.filter(isJudgeVerdict);
|
|
337
|
-
if (leaks.length > 0) throw new Error(`${context}: a judge verdict cannot be admitted as steering input — that is the held-out judge leaking into the loop. Offending judge-derived findings: [${leaks.map((f) => f.finding_id).join(", ")}]. Steering consumes observations of behavior, never acceptance verdicts.`);
|
|
338
|
-
return findings;
|
|
339
|
-
}
|
|
340
|
-
//#endregion
|
|
341
|
-
export { ANALYST_SEVERITIES, AnalystRegistry, DEFAULT_TRACE_ANALYST_KINDS, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, FindingSubjectStringSchema, FindingsStore, IMPROVEMENT_KIND_SPEC, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, RAW_FINDING_SCHEMA_PROMPT, RawAnalystEvidenceSchema, RawAnalystFindingSchema, SKILL_USAGE_ANALYST, SkillUsageAnalyst, assertNoJudgeVerdict, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isJudgeVerdict, isTraceObservable, liftSeverity, makeFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
|
|
292
|
+
export { ANALYST_SEVERITIES, AnalystRegistry, DEFAULT_TRACE_ANALYST_KINDS, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, FindingSubjectStringSchema, FindingsStore, IMPROVEMENT_KIND_SPEC, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, RAW_FINDING_SCHEMA_PROMPT, RawAnalystEvidenceSchema, RawAnalystFindingSchema, SKILL_USAGE_ANALYST, SkillUsageAnalyst, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, makeFinding, makeProposalFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
|
|
342
293
|
|
|
343
294
|
//# sourceMappingURL=index.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/steer-firewall.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport { CostLedger } from '../cost-ledger'\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n type SemanticConceptJudgeResult,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore } from '../types'\nimport type { ChatClient } from './chat-client'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\nimport { settleUsageReceiptFromCostLedger, validateUsageSettlementTimeout } from './usage-receipt'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** Chat client passed to the JudgeFn. */\n chat: ChatClient\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.chat, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Provenance: this finding IS a judge verdict (an acceptance score), not an\n // observation of behavior. The steer firewall (assertNoJudgeVerdict) rejects\n // it from steering — even when it cites an artifact above — because letting a\n // verdict steer the next attempt is the held-out judge leaking into the loop.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n /** Registry context owns cancellation and the per-analyst cost ledger. */\n options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>\n /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */\n settlementTimeoutMs?: number\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs)\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: {\n kind: 'llm',\n models: opts.options?.model ? [opts.options.model] : undefined,\n settlement_timeout_ms: settlementTimeoutMs,\n },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input, ctx) {\n const costLedger = new CostLedger(ctx.budgetUsd)\n let result: SemanticConceptJudgeResult\n try {\n result = await runSemanticConceptJudge(input, {\n ...opts.options,\n costLedger,\n signal: ctx.signal,\n })\n } finally {\n const usage = await settleUsageReceiptFromCostLedger(costLedger, {\n channel: 'judge',\n timeoutMs: settlementTimeoutMs,\n })\n if (!usage.settled) {\n ctx.log?.('semantic-concept judge provider settlement timed out', {\n pending_calls: usage.pendingCalls,\n timeout_ms: settlementTimeoutMs,\n })\n }\n ctx.recordUsage?.(usage.receipt)\n }\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n },\n }),\n )\n }\n return out\n },\n }\n}\n","// The realness-oracle firewall (docs/learning-flywheel.md, \"The steer is f(trace)\").\n//\n// A realness/authenticity signal has TWO legitimate roles that must stay\n// separated by a firewall:\n// (a) anchor judge J — write-only: scores the chosen output, gates promotion,\n// NEVER seen by the worker/optimizer mid-run (else the loop games it).\n// (b) steer f(trace) — an analyst observes the agent's OWN behavior in the\n// trace (\"imported a stub\", \"used a non-crypto PRNG where encryption was\n// required\") and steers the next attempt. Legitimate, because it is derived\n// from OBSERVABLE BEHAVIOR, not from J's held-out verdict.\n//\n// The correct discriminator is PROVENANCE, not evidence presence. A judge verdict\n// lifted into a finding (createJudgeAdapter → liftJudgeScore) is a verdict even\n// when it cites an artifact; an evidence-less trace-analyst bullet is an\n// observation even though it cites nothing. So the firewall keys on\n// `AnalystFinding.derived_from_judge` (set at the judge lift site), NOT on whether\n// evidence_refs is populated. The instant a verdict steers the next attempt it is\n// a back-channel for J and the loop Goodharts realness exactly as it would\n// Goodhart pass-rate.\n\nimport type { AnalystFinding, EvidenceRef } from './types'\n\n/** Evidence grounded in the agent's OWN execution: OTLP trace elements\n * (`span`/`event`) or the artifact it produced (`artifact`). */\nconst OBSERVABLE_KINDS: ReadonlySet<EvidenceRef['kind']> = new Set<EvidenceRef['kind']>([\n 'span',\n 'event',\n 'artifact',\n])\n\n/** DESCRIPTIVE predicate: does the finding cite at least one observable\n * (span/event/artifact) evidence ref. Useful for ranking evidence quality or\n * rendering — it is NOT the steer gate. Evidence presence is the WRONG\n * discriminator for steering: a legitimate trace-analyst observation may cite\n * nothing (it would be wrongly rejected), and a judge verdict may cite an\n * artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate\n * steering; use this only where \"is this grounded in observable evidence\" is the\n * literal question. */\nexport function isTraceObservable(finding: AnalystFinding): boolean {\n return finding.evidence_refs.some((ref) => OBSERVABLE_KINDS.has(ref.kind))\n}\n\n/** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a\n * finding), identified by provenance set at the lift site — independent of\n * whatever evidence it cites. */\nexport function isJudgeVerdict(finding: AnalystFinding): boolean {\n return finding.derived_from_judge === true\n}\n\n/**\n * THE steer firewall. Fail-loud guard for any path that admits analyst findings\n * as STEERING input (the `f(trace)` role): rejects — naming the offenders — any\n * finding whose provenance is a judge verdict, rather than let `J` leak into the\n * loop. Returns the findings unchanged for chaining.\n *\n * Call this at the chokepoint where a detector that ALSO scores/gates has its\n * findings turned into a steer (the judge-and-steer dual-role case). It keys on\n * provenance, so it correctly admits evidence-less trace-analyst observations and\n * correctly rejects an artifact-citing judge verdict — the cases an evidence\n * check gets backwards.\n *\n * It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge\n * whose output is laundered through a hand-built finding with no provenance flag\n * is out of its reach — provenance must be honestly set at every judge→finding\n * lift (today: createJudgeAdapter). That is why the integrity rule lives at the\n * lift site, and why ProposeContext.judgeScores?: never is the complementary\n * compile-time tripwire on the obvious direct channel.\n */\nexport function assertNoJudgeVerdict(\n findings: ReadonlyArray<AnalystFinding>,\n context = 'steer',\n): ReadonlyArray<AnalystFinding> {\n const leaks = findings.filter(isJudgeVerdict)\n if (leaks.length > 0) {\n throw new Error(\n `${context}: a judge verdict cannot be admitted as steering input — that is the ` +\n `held-out judge leaking into the loop. Offending judge-derived findings: [${leaks\n .map((f) => f.finding_id)\n .join(', ')}]. Steering consumes observations of behavior, never acceptance verdicts.`,\n )\n }\n return findings\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;AAwCA,MAAM,cAAc;AAIpB,SAAgB,aAAa,GAAmC;CAC9D,QAAQ,GAAR;EACE,KAAK,YACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,QACH,OAAO;CACX;AACF;AAeA,SAAgB,sBAA2B,MAA8C;CACvF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,YAAY;EACrB,MAAM,QAAQ,KAAK,KAAK;GACtB,MAAM,SAAS,MAAM,KAAK,SAAS,IAAI;IAAE;IAAK,GAAG,KAAK;GAAQ,CAAC;GAC/D,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,SAAS,OAAO,QAAQ;IACjC,KAAK,MAAM,WAAW,MAAM,UAC1B,IAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;IAI3D,IAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAC1E,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,MAAM;KACf,OAAO,UAAU,MAAM,MAAM,IAAI,MAAM,OAAO,IAAI,MAAM,UAAU;KAClE,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;KAC9E,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MACR,cAAc,MAAM;MACpB,aAAa,MAAM;MACnB,OAAO,MAAM;MACb,aAAa,MAAM;KACrB;IACF,CAAC,CACH;GAEJ;GACA,IAAI,MAAM,qBAAqB;IAC7B,QAAQ,OAAO,OAAO;IACtB,SAAS,OAAO;IAChB,UAAU,OAAO;GACnB,CAAC;GACD,OAAO;EACT;CACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;CAChB,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE,SAAS;EACpB,OAAO,EAAE;EACT,UAAU,aAAa,EAAE,QAAQ;EACjC,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EACL,UAAU,EAAE;CACd,CAAC;AACH;AAYA,SAAgB,uBAAuB,OAA6B,CAAC,GAAsB;CACzF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,SAAS,KAAK,UAAU,IAAI,UAAU;CAC5C,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,cAAc;EACvB,MAAM,QAAQ,OAAO;GACnB,MAAM,QAAQ,OAAO,WAAW,KAAK;GACrC,MAAM,MAAwB,CAAC;GAU/B,KAAK,MAAM,CAAC,KAAK,KAAK,QAAQ;IAR5B;KAAC;KAAW;KAAY;IAAmC;IAC3D;KAAC;KAAgB;KAAQ;IAAsB;IAC/C;KAAC;KAAoB;KAAQ;IAA6C;IAC1E;KAAC;KAAkB;KAAU;IAAyB;IACtD;KAAC;KAAgB;KAAU;IAA6B;IACxD;KAAC;KAAe;KAAQ;IAA6B;IACrD;KAAC;KAAa;KAAY;IAAwB;GAEnB,GAAG;IAClC,MAAM,QAAQ,MAAM;IACpB,IAAI,OAAO,UAAU,YAAY,QAAQ,WACvC,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS;KACT,OAAO;KACP,WAAW,GAAG,IAAI,GAAG,MAAM,QAAQ,CAAC,EAAE,mBAAmB;KACzD,UAAU;KACV,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MAAE,WAAW;MAAK;MAAO;MAAW,QAAQ,MAAM,IAAI;KAAM;IACxE,CAAC,CACH;GAEJ;GAEA,IAAI,MAAM,eAAe,IAAI,WAC3B,IAAI,KACF,YAAY;IACV,YAAY;IACZ;IACA,SAAS;IACT,OAAO;IACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC;IACvD,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU;KAAE,eAAe,MAAM;KAAc,OAAO,MAAM;IAAM;GACpE,CAAC,CACH;GAEF,OAAO;EACT;CACF;AACF;AAgBA,SAAgB,mBAAmB,MAA6C;CAC9E,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;EACjC,SAAS,SAAS;EAClB,MAAM,QAAQ,OAAO;GAEnB,QAAO,MADc,KAAK,MAAM,KAAK,MAAM,KAAK,EAAA,CAE7C,QAAQ,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,CAAC,CAC/C,KAAK,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;EAC3C;CACF;AACF;AAEA,SAAS,YAAY,GAAmB;CAEtC,OAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;CACvF,MAAM,UAAU,YAAY,EAAE,KAAK;CACnC,MAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;CAC7E,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE;EACX,OAAO,GAAG,EAAE,UAAU,GAAG,EAAE,UAAU,UAAU,QAAQ,QAAQ,CAAC,EAAE;EAClE,WAAW,EAAE;EACb;EACA,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EAKL,oBAAoB;EACpB,UAAU;GAAE,YAAY,EAAE;GAAW,WAAW,EAAE;GAAW,UAAU;EAAQ;CACjF,CAAC;AACH;AAaA,SAAgB,kCACd,OAAwC,CAAC,GACL;CACpC,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,sBAAsB,+BAA+B,KAAK,mBAAmB;CACnF,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM;GACJ,MAAM;GACN,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,KAAA;GACrD,uBAAuB;EACzB;EACA,SAAS,GAAG,+BAA+B,WAAW;EACtD,MAAM,QAAQ,OAAO,KAAK;GACxB,MAAM,aAAa,IAAI,WAAW,IAAI,SAAS;GAC/C,IAAI;GACJ,IAAI;IACF,SAAS,MAAM,wBAAwB,OAAO;KAC5C,GAAG,KAAK;KACR;KACA,QAAQ,IAAI;IACd,CAAC;GACH,UAAU;IACR,MAAM,QAAQ,MAAM,iCAAiC,YAAY;KAC/D,SAAS;KACT,WAAW;IACb,CAAC;IACD,IAAI,CAAC,MAAM,SACT,IAAI,MAAM,wDAAwD;KAChE,eAAe,MAAM;KACrB,YAAY;IACd,CAAC;IAEH,IAAI,cAAc,MAAM,OAAO;GACjC;GACA,IAAI,CAAC,OAAO,WACV,OAAO,CACL,YAAY;IACV,YAAY;IACZ;IACA,OAAO;IACP,WAAW,OAAO;IAClB,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;GACnC,CAAC,CACH;GAEF,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,KAAK,OAAO,UAAU;IAG/B,IAAI,EAAE,WAAW,EAAE,SAAS,GAAG;IAC/B,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,EAAE;KACX,OAAO,EAAE,UACL,YAAY,EAAE,QAAQ,aAAa,EAAE,MAAM,QAC3C,YAAY,EAAE,QAAQ;KAC1B,WAAW,EAAE;KACb,UAAU,aAAa,EAAE,QAAQ;KACjC,YAAY;KACZ,eAAe,CAAC;MAAE,MAAM;MAAY,KAAK;MAAmB,SAAS,EAAE;KAAS,CAAC;KACjF,UAAU;MACR,SAAS,EAAE;MACX,SAAS,EAAE;MACX,UAAU,EAAE;KACd;IACF,CAAC,CACH;GACF;GACA,OAAO;EACT;CACF;AACF;;;;;ACxVA,MAAM,mCAAqD,IAAI,IAAyB;CACtF;CACA;CACA;AACF,CAAC;;;;;;;;;AAUD,SAAgB,kBAAkB,SAAkC;CAClE,OAAO,QAAQ,cAAc,MAAM,QAAQ,iBAAiB,IAAI,IAAI,IAAI,CAAC;AAC3E;;;;AAKA,SAAgB,eAAe,SAAkC;CAC/D,OAAO,QAAQ,uBAAuB;AACxC;;;;;;;;;;;;;;;;;;;;AAqBA,SAAgB,qBACd,UACA,UAAU,SACqB;CAC/B,MAAM,QAAQ,SAAS,OAAO,cAAc;CAC5C,IAAI,MAAM,SAAS,GACjB,MAAM,IAAI,MACR,GAAG,QAAQ,gJACmE,MACzE,KAAK,MAAM,EAAE,UAAU,CAAC,CACxB,KAAK,IAAI,EAAE,0EAClB;CAEF,OAAO;AACT"}
|
|
1
|
+
{"version":3,"file":"index.js","names":[],"sources":["../../src/analyst/adapters.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport { CostLedger } from '../cost-ledger'\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n type SemanticConceptJudgeResult,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore } from '../types'\nimport type { ChatClient } from './chat-client'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\nimport { settleUsageReceiptFromCostLedger, validateUsageSettlementTimeout } from './usage-receipt'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** Chat client passed to the JudgeFn. */\n chat: ChatClient\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.chat, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Descriptive origin only. A caller may admit search feedback into candidate\n // generation, but final evaluation findings have no allowed proposal origin.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n /** Registry context owns cancellation and the per-analyst cost ledger. */\n options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>\n /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */\n settlementTimeoutMs?: number\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs)\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: {\n kind: 'llm',\n models: opts.options?.model ? [opts.options.model] : undefined,\n settlement_timeout_ms: settlementTimeoutMs,\n },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input, ctx) {\n const costLedger = new CostLedger(ctx.budgetUsd)\n let result: SemanticConceptJudgeResult\n try {\n result = await runSemanticConceptJudge(input, {\n ...opts.options,\n costLedger,\n signal: ctx.signal,\n })\n } finally {\n const usage = await settleUsageReceiptFromCostLedger(costLedger, {\n channel: 'judge',\n timeoutMs: settlementTimeoutMs,\n })\n if (!usage.settled) {\n ctx.log?.('semantic-concept judge provider settlement timed out', {\n pending_calls: usage.pendingCalls,\n timeout_ms: settlementTimeoutMs,\n })\n }\n ctx.recordUsage?.(usage.receipt)\n }\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n },\n }),\n )\n }\n return out\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;AAwCA,MAAM,cAAc;AAIpB,SAAgB,aAAa,GAAmC;CAC9D,QAAQ,GAAR;EACE,KAAK,YACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,QACH,OAAO;CACX;AACF;AAeA,SAAgB,sBAA2B,MAA8C;CACvF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,YAAY;EACrB,MAAM,QAAQ,KAAK,KAAK;GACtB,MAAM,SAAS,MAAM,KAAK,SAAS,IAAI;IAAE;IAAK,GAAG,KAAK;GAAQ,CAAC;GAC/D,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,SAAS,OAAO,QAAQ;IACjC,KAAK,MAAM,WAAW,MAAM,UAC1B,IAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;IAI3D,IAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAC1E,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,MAAM;KACf,OAAO,UAAU,MAAM,MAAM,IAAI,MAAM,OAAO,IAAI,MAAM,UAAU;KAClE,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;KAC9E,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MACR,cAAc,MAAM;MACpB,aAAa,MAAM;MACnB,OAAO,MAAM;MACb,aAAa,MAAM;KACrB;IACF,CAAC,CACH;GAEJ;GACA,IAAI,MAAM,qBAAqB;IAC7B,QAAQ,OAAO,OAAO;IACtB,SAAS,OAAO;IAChB,UAAU,OAAO;GACnB,CAAC;GACD,OAAO;EACT;CACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;CAChB,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE,SAAS;EACpB,OAAO,EAAE;EACT,UAAU,aAAa,EAAE,QAAQ;EACjC,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EACL,UAAU,EAAE;CACd,CAAC;AACH;AAYA,SAAgB,uBAAuB,OAA6B,CAAC,GAAsB;CACzF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,SAAS,KAAK,UAAU,IAAI,UAAU;CAC5C,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,cAAc;EACvB,MAAM,QAAQ,OAAO;GACnB,MAAM,QAAQ,OAAO,WAAW,KAAK;GACrC,MAAM,MAAwB,CAAC;GAU/B,KAAK,MAAM,CAAC,KAAK,KAAK,QAAQ;IAR5B;KAAC;KAAW;KAAY;IAAmC;IAC3D;KAAC;KAAgB;KAAQ;IAAsB;IAC/C;KAAC;KAAoB;KAAQ;IAA6C;IAC1E;KAAC;KAAkB;KAAU;IAAyB;IACtD;KAAC;KAAgB;KAAU;IAA6B;IACxD;KAAC;KAAe;KAAQ;IAA6B;IACrD;KAAC;KAAa;KAAY;IAAwB;GAEnB,GAAG;IAClC,MAAM,QAAQ,MAAM;IACpB,IAAI,OAAO,UAAU,YAAY,QAAQ,WACvC,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS;KACT,OAAO;KACP,WAAW,GAAG,IAAI,GAAG,MAAM,QAAQ,CAAC,EAAE,mBAAmB;KACzD,UAAU;KACV,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MAAE,WAAW;MAAK;MAAO;MAAW,QAAQ,MAAM,IAAI;KAAM;IACxE,CAAC,CACH;GAEJ;GAEA,IAAI,MAAM,eAAe,IAAI,WAC3B,IAAI,KACF,YAAY;IACV,YAAY;IACZ;IACA,SAAS;IACT,OAAO;IACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC;IACvD,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU;KAAE,eAAe,MAAM;KAAc,OAAO,MAAM;IAAM;GACpE,CAAC,CACH;GAEF,OAAO;EACT;CACF;AACF;AAgBA,SAAgB,mBAAmB,MAA6C;CAC9E,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;EACjC,SAAS,SAAS;EAClB,MAAM,QAAQ,OAAO;GAEnB,QAAO,MADc,KAAK,MAAM,KAAK,MAAM,KAAK,EAAA,CAE7C,QAAQ,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,CAAC,CAC/C,KAAK,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;EAC3C;CACF;AACF;AAEA,SAAS,YAAY,GAAmB;CAEtC,OAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;CACvF,MAAM,UAAU,YAAY,EAAE,KAAK;CACnC,MAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;CAC7E,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE;EACX,OAAO,GAAG,EAAE,UAAU,GAAG,EAAE,UAAU,UAAU,QAAQ,QAAQ,CAAC,EAAE;EAClE,WAAW,EAAE;EACb;EACA,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EAGL,oBAAoB;EACpB,UAAU;GAAE,YAAY,EAAE;GAAW,WAAW,EAAE;GAAW,UAAU;EAAQ;CACjF,CAAC;AACH;AAaA,SAAgB,kCACd,OAAwC,CAAC,GACL;CACpC,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,sBAAsB,+BAA+B,KAAK,mBAAmB;CACnF,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM;GACJ,MAAM;GACN,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,KAAA;GACrD,uBAAuB;EACzB;EACA,SAAS,GAAG,+BAA+B,WAAW;EACtD,MAAM,QAAQ,OAAO,KAAK;GACxB,MAAM,aAAa,IAAI,WAAW,IAAI,SAAS;GAC/C,IAAI;GACJ,IAAI;IACF,SAAS,MAAM,wBAAwB,OAAO;KAC5C,GAAG,KAAK;KACR;KACA,QAAQ,IAAI;IACd,CAAC;GACH,UAAU;IACR,MAAM,QAAQ,MAAM,iCAAiC,YAAY;KAC/D,SAAS;KACT,WAAW;IACb,CAAC;IACD,IAAI,CAAC,MAAM,SACT,IAAI,MAAM,wDAAwD;KAChE,eAAe,MAAM;KACrB,YAAY;IACd,CAAC;IAEH,IAAI,cAAc,MAAM,OAAO;GACjC;GACA,IAAI,CAAC,OAAO,WACV,OAAO,CACL,YAAY;IACV,YAAY;IACZ;IACA,OAAO;IACP,WAAW,OAAO;IAClB,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;GACnC,CAAC,CACH;GAEF,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,KAAK,OAAO,UAAU;IAG/B,IAAI,EAAE,WAAW,EAAE,SAAS,GAAG;IAC/B,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,EAAE;KACX,OAAO,EAAE,UACL,YAAY,EAAE,QAAQ,aAAa,EAAE,MAAM,QAC3C,YAAY,EAAE,QAAQ;KAC1B,WAAW,EAAE;KACb,UAAU,aAAa,EAAE,QAAQ;KACjC,YAAY;KACZ,eAAe,CAAC;MAAE,MAAM;MAAY,KAAK;MAAmB,SAAS,EAAE;KAAS,CAAC;KACjF,UAAU;MACR,SAAS,EAAE;MACX,SAAS,EAAE;MACX,UAAU,EAAE;KACd;IACF,CAAC,CACH;GACF;GACA,OAAO;EACT;CACF;AACF"}
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { a as RunRecord } from "./run-record-DcObtIGh.js";
|
|
2
|
-
import { i as AnalystRegistry } from "./default-registry-
|
|
2
|
+
import { i as AnalystRegistry } from "./default-registry-Brxr728w.js";
|
|
3
3
|
import { a as DatasetScenario } from "./dataset-BvtnC8Dc.js";
|
|
4
|
-
import "./summary-report-
|
|
5
|
-
import { C as InsightReport, b as ExecutionInsight, v as CostProvenanceSummary } from "./client-
|
|
4
|
+
import "./summary-report-DGp0-_XO.js";
|
|
5
|
+
import { C as InsightReport, b as ExecutionInsight, v as CostProvenanceSummary } from "./client-BIyh1RCr.js";
|
|
6
6
|
//#region src/contract/analyze-runs.d.ts
|
|
7
7
|
interface AnalyzeRunsOptions {
|
|
8
8
|
/** The runs to analyze. */
|
|
@@ -69,4 +69,4 @@ declare function summarizeExecution(opts: SummarizeExecutionOptions): ExecutionR
|
|
|
69
69
|
declare function analyzeRuns(opts: AnalyzeRunsOptions): Promise<InsightReport>;
|
|
70
70
|
//#endregion
|
|
71
71
|
export { summarizeExecution as a, analyzeRuns as i, ExecutionReport as n, SummarizeExecutionOptions as r, AnalyzeRunsOptions as t };
|
|
72
|
-
//# sourceMappingURL=analyze-runs-
|
|
72
|
+
//# sourceMappingURL=analyze-runs-DMo3Lb_y.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"analyze-runs-
|
|
1
|
+
{"version":3,"file":"analyze-runs-DMo3Lb_y.d.ts","names":[],"sources":["../src/contract/analyze-runs.ts"],"mappings":";;;;;;UAoEiB;;EAEf,MAAM;;;EAGN;;;;;EAKA;EACA;;;EAGA,kBAAkB;;;EAGlB,UAAU;;;;EAIV;IACE;IACA,cAAc;;;;;EAKhB,cAAc;IAAQ;IAAe;IAAe;;;EAEpD;;;;EAIA;;;;;;;;;EASA,eAAe;;;EAGf;;UAGe;EACf,MAAM;EACN;;UAGe;EACf,WAAW;EACX,gBAAgB;;;iBAIF,mBAAmB,MAAM,4BAA4B;iBAS/C,YAAY,MAAM,qBAAqB,QAAQ"}
|