@tangle-network/agent-eval 0.117.1 → 0.118.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +23 -0
- package/dist/analyst/index.d.ts +2772 -21
- package/dist/analyst/index.js +7 -6
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +706 -10
- package/dist/belief-state/index.js +2 -1
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +958 -14
- package/dist/benchmarks/index.js +12 -10
- package/dist/builder-eval/index.d.ts +449 -4
- package/dist/builder-eval/index.js +4 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +4275 -73
- package/dist/campaign/index.js +12 -10
- package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
- package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
- package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
- package/dist/{chunk-YZPO4UHR.js → chunk-DP3BBMYJ.js} +100 -143
- package/dist/chunk-DP3BBMYJ.js.map +1 -0
- package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
- package/dist/chunk-HKUCJ437.js.map +1 -0
- package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
- package/dist/chunk-KSDQVPLR.js +286 -0
- package/dist/chunk-KSDQVPLR.js.map +1 -0
- package/dist/chunk-MA6HLL3S.js +65 -0
- package/dist/chunk-MA6HLL3S.js.map +1 -0
- package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
- package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
- package/dist/chunk-PXE2VKMX.js +140 -0
- package/dist/chunk-PXE2VKMX.js.map +1 -0
- package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
- package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
- package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
- package/dist/chunk-QKEGNI5B.js.map +1 -0
- package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
- package/dist/chunk-S3UZOQ5Y.js.map +1 -0
- package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
- package/dist/chunk-SVH2ANFD.js.map +1 -0
- package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
- package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
- package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
- package/dist/chunk-WDHBCA3M.js +31 -0
- package/dist/chunk-WDHBCA3M.js.map +1 -0
- package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
- package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
- package/dist/chunk-ZET2UAYW.js.map +1 -0
- package/dist/contract/index.d.ts +4012 -38
- package/dist/contract/index.js +56 -29
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +1013 -9
- package/dist/control.js +4 -3
- package/dist/fuzz.d.ts +194 -4
- package/dist/fuzz.js +3 -3
- package/dist/hosted/index.d.ts +498 -17
- package/dist/index.d.ts +10982 -1286
- package/dist/index.js +97 -80
- package/dist/index.js.map +1 -1
- package/dist/matrix/index.d.ts +139 -4
- package/dist/meta-eval/index.d.ts +862 -15
- package/dist/meta-eval/index.js +2 -1
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/multishot/index.d.ts +214 -14
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +392 -7
- package/dist/pipelines/index.js +5 -3
- package/dist/pipelines/index.js.map +1 -1
- package/dist/reporting.d.ts +1277 -17
- package/dist/rl.d.ts +2359 -28
- package/dist/rl.js +6 -5
- package/dist/rl.js.map +1 -1
- package/dist/storyboard/index.d.ts +86 -1
- package/dist/trace-attributes.d.ts +16 -0
- package/dist/trace-attributes.js +32 -0
- package/dist/trace-attributes.js.map +1 -0
- package/dist/traces.d.ts +1970 -697
- package/dist/traces.js +49 -32
- package/dist/wire/index.d.ts +655 -9
- package/docs/insight-report.md +44 -0
- package/package.json +9 -3
- package/dist/adversarial-B7loGVVX.d.ts +0 -19
- package/dist/analyst-C8HHvfJp.d.ts +0 -88
- package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
- package/dist/baseline-DKq3gJpP.d.ts +0 -141
- package/dist/calibration-C8MTS7cw.d.ts +0 -101
- package/dist/chunk-5UF54T55.js.map +0 -1
- package/dist/chunk-E4BUPP7Z.js.map +0 -1
- package/dist/chunk-FQNLDL4D.js.map +0 -1
- package/dist/chunk-LQUTGLOZ.js.map +0 -1
- package/dist/chunk-S2F4J57L.js.map +0 -1
- package/dist/chunk-YZPO4UHR.js.map +0 -1
- package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
- package/dist/control-6vuGfmDH.d.ts +0 -258
- package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
- package/dist/dataset-NENEzRgk.d.ts +0 -115
- package/dist/default-registry-DaK8b3fv.d.ts +0 -155
- package/dist/emitter-CjD7vUwv.d.ts +0 -122
- package/dist/errors-oeQrLqXC.d.ts +0 -74
- package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
- package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
- package/dist/gepa-eESocoDi.d.ts +0 -642
- package/dist/index-PdX4VnPA.d.ts +0 -423
- package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
- package/dist/integrity-DqlBiLyK.d.ts +0 -81
- package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
- package/dist/kind-factory-ClZmO25A.d.ts +0 -171
- package/dist/llm-client-qoDd18Qz.d.ts +0 -289
- package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
- package/dist/off-policy-DiwuKKg7.d.ts +0 -132
- package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
- package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
- package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
- package/dist/provenance-DpjwyseI.d.ts +0 -541
- package/dist/query-CF7PG61p.d.ts +0 -35
- package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
- package/dist/release-report-C8G2i5Xi.d.ts +0 -236
- package/dist/researcher-C8XyxQsu.d.ts +0 -387
- package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
- package/dist/run-record-BDH49H2E.d.ts +0 -360
- package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
- package/dist/schema-B3Q3l9Z_.d.ts +0 -201
- package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
- package/dist/sequential-5iSVfzl2.d.ts +0 -139
- package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
- package/dist/statistics-KUnG73jH.d.ts +0 -494
- package/dist/storage-DrX3v_5B.d.ts +0 -50
- package/dist/store-C1YxJDEK.d.ts +0 -248
- package/dist/store-DGqD0Pyo.d.ts +0 -116
- package/dist/summary-report-C5bKFfm-.d.ts +0 -445
- package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
- package/dist/types-BSw1rOUB.d.ts +0 -634
- package/dist/types-BUxNaJ8c.d.ts +0 -108
- package/dist/types-BkfcQnxV.d.ts +0 -313
- package/dist/verdict-C9MlYujm.d.ts +0 -35
- /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
- /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
- /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
- /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
- /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
- /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
- /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
- /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
- /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
- /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
|
@@ -1,310 +0,0 @@
|
|
|
1
|
-
import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-C5bKFfm-.js';
|
|
2
|
-
import { C as ContinuousAgreement } from './judge-calibration-7C-IDmKr.js';
|
|
3
|
-
|
|
4
|
-
/**
|
|
5
|
-
* # InsightReport — the rigorous decision packet for any set of agent runs.
|
|
6
|
-
*
|
|
7
|
-
* Returned by `analyzeRuns()` and embedded in `SelfImproveResult.insight` +
|
|
8
|
-
* the hosted-tier `EvalRunEvent.insightReport`. One shape across two surfaces:
|
|
9
|
-
*
|
|
10
|
-
* - **Customer who has a closed loop** (`selfImprove`): the report ships
|
|
11
|
-
* with the loop output. Their dashboard renders ship/hold + lift CI +
|
|
12
|
-
* calibration + cluster + Pareto in one packet.
|
|
13
|
-
* - **Customer who has observed runs but no loop** (`analyzeRuns` directly):
|
|
14
|
-
* same packet from a `RunRecord[]` they already have — production traces,
|
|
15
|
-
* approve/reject corpus, CSV gold set.
|
|
16
|
-
*
|
|
17
|
-
* Every field is optional except the distributional summary — fields are
|
|
18
|
-
* populated when the input data supports them:
|
|
19
|
-
*
|
|
20
|
-
* - `lift` requires both baseline and candidate splits to be present.
|
|
21
|
-
* - `interRater` requires multi-rater feedback (≥2 raters per run).
|
|
22
|
-
* - `judges` populates per-judge stats only when the run records carry
|
|
23
|
-
* `outcome.judgeScores`.
|
|
24
|
-
* - `failureClusters` requires the optional `analystRegistry` to be wired.
|
|
25
|
-
* - `contamination` requires canary scenarios to be passed in.
|
|
26
|
-
* - `outcomeCorrelation` requires a downstream outcome signal.
|
|
27
|
-
* - `sequential` requires the run set to be ordered (treats them as a
|
|
28
|
-
* stream and emits an anytime-valid interim decision).
|
|
29
|
-
*
|
|
30
|
-
* Consumers read the `recommendations` array first — that's the
|
|
31
|
-
* actionable layer, ranked by priority. The numeric sections back it up.
|
|
32
|
-
*/
|
|
33
|
-
|
|
34
|
-
interface InsightReport {
|
|
35
|
-
/** Number of runs analyzed. */
|
|
36
|
-
n: number;
|
|
37
|
-
/** Composite-score distribution across all runs. Always present. */
|
|
38
|
-
composite: ScalarDistribution;
|
|
39
|
-
/** Per-dimension distributions for every dimension that appeared in any
|
|
40
|
-
* run's judge scores. Empty when no judge scores were recorded. */
|
|
41
|
-
perDimension: Record<string, ScalarDistribution>;
|
|
42
|
-
/** Cost/quality distribution and Pareto frontier. */
|
|
43
|
-
costQuality: {
|
|
44
|
-
cost: ScalarDistribution;
|
|
45
|
-
pareto: ParetoFigureSpec;
|
|
46
|
-
/** Cost source coverage. `uncaptured` rows are excluded from the USD
|
|
47
|
-
* distribution and Pareto chart; observed and estimated totals remain
|
|
48
|
-
* separate so reports never present estimates as billed spend. */
|
|
49
|
-
provenance?: CostProvenanceSummary;
|
|
50
|
-
/** Set when the cost/quality view is degraded because the input data
|
|
51
|
-
* doesn't fully support it — e.g. all `costUsd` were zero, or only a
|
|
52
|
-
* single candidate appears (so the Pareto is a single point). The
|
|
53
|
-
* named fields name the degraded sub-view, free-text the reason. */
|
|
54
|
-
degraded?: {
|
|
55
|
-
cost?: string;
|
|
56
|
-
pareto?: string;
|
|
57
|
-
};
|
|
58
|
-
};
|
|
59
|
-
/** Per-judge calibration + bias detection. Populated for every judge name
|
|
60
|
-
* that appears in `outcome.judgeScores`. Bias fields require either a
|
|
61
|
-
* gold reference or multi-rater data. */
|
|
62
|
-
judges: Record<string, JudgeInsight>;
|
|
63
|
-
/** Inter-rater agreement when multiple judges scored the same runs.
|
|
64
|
-
* Includes pairwise kappa and the specific run ids where raters
|
|
65
|
-
* disagree — the cases worth a human meeting. */
|
|
66
|
-
interRater?: InterRaterInsight;
|
|
67
|
-
/** Pairwise lift (baseline → candidate) with bootstrap CI. Present when
|
|
68
|
-
* `RunRecord.splitTag` includes both `holdout` and search/dev splits,
|
|
69
|
-
* or when caller passes an explicit baseline/candidate split. */
|
|
70
|
-
lift?: LiftInsight;
|
|
71
|
-
/** Failure clusters with exemplars. Populated when an AnalystRegistry
|
|
72
|
-
* is wired in `analyzeRuns({ analyst })`. */
|
|
73
|
-
failureClusters?: FailureClusterInsight;
|
|
74
|
-
/** Canary leak count + holdout audit status. Populated when canary
|
|
75
|
-
* scenarios are passed in. */
|
|
76
|
-
contamination?: ContaminationInsight;
|
|
77
|
-
/** Correlation between judge composite and a downstream outcome the
|
|
78
|
-
* caller supplies (engagement, revenue, downstream pass rate, etc.).
|
|
79
|
-
* When present, the optional reward model is the model that maps
|
|
80
|
-
* judge scores → predicted outcome. */
|
|
81
|
-
outcomeCorrelation?: OutcomeCorrelationInsight;
|
|
82
|
-
/** Aggregate release-readiness summary. A consumer needing the full
|
|
83
|
-
* substrate `ReleaseConfidenceScorecard` (SLO-axis evaluation,
|
|
84
|
-
* ActionableSideInfo bag) calls `evaluateReleaseConfidence()` directly;
|
|
85
|
-
* this summary captures the analyzeRuns-derived axes. */
|
|
86
|
-
release: ReleaseSummary;
|
|
87
|
-
/** Delta vs a prior period when `baselineRuns` is passed. Per-metric
|
|
88
|
-
* current vs baseline with Welch CI + Cohen's d + significance flag.
|
|
89
|
-
* Answers "did my last change help?" — the customer-conversion question.
|
|
90
|
-
* Surfaced metrics: composite, cost, duration, tokenUsage, plus any
|
|
91
|
-
* per-dimension judge metric present in both windows. */
|
|
92
|
-
priorPeriodComparison?: PriorPeriodComparison;
|
|
93
|
-
/** Model-free failure-mode breakdown from `RunRecord.failureMode`, ranked
|
|
94
|
-
* by count descending. Present when any run carries a `failureMode`.
|
|
95
|
-
* Complements `failureClusters` (LLM-semantic) with the structured tags
|
|
96
|
-
* the harness already recorded — actionable with no analyst wired. */
|
|
97
|
-
failureModes?: FailureModeTally[];
|
|
98
|
-
/** Top-N actionable recommendations, ranked by priority. The packet's
|
|
99
|
-
* human-readable layer; the numeric sections are the evidence. */
|
|
100
|
-
recommendations: Recommendation[];
|
|
101
|
-
}
|
|
102
|
-
interface CostProvenanceSummary {
|
|
103
|
-
observed: {
|
|
104
|
-
n: number;
|
|
105
|
-
totalUsd: number;
|
|
106
|
-
};
|
|
107
|
-
estimated: {
|
|
108
|
-
n: number;
|
|
109
|
-
totalUsd: number;
|
|
110
|
-
};
|
|
111
|
-
uncaptured: {
|
|
112
|
-
n: number;
|
|
113
|
-
};
|
|
114
|
-
knownFraction: number;
|
|
115
|
-
}
|
|
116
|
-
/** Distributional summary of a scalar-valued metric. */
|
|
117
|
-
interface ScalarDistribution {
|
|
118
|
-
/** Sample count after dropping non-finite values. */
|
|
119
|
-
n: number;
|
|
120
|
-
mean: number;
|
|
121
|
-
p50: number;
|
|
122
|
-
p95: number;
|
|
123
|
-
stddev: number;
|
|
124
|
-
min: number;
|
|
125
|
-
max: number;
|
|
126
|
-
/** Histogram bins using `agent-eval`'s `gainHistogram` primitive. */
|
|
127
|
-
histogram: GainDistributionBin[];
|
|
128
|
-
/** Worst-N runs by score, ascending. Populated for the composite
|
|
129
|
-
* distribution so the report names the runs a customer should
|
|
130
|
-
* inspect first. Undefined when the distribution was computed from a
|
|
131
|
-
* raw value list with no run identity (e.g. cost). */
|
|
132
|
-
tailRuns?: Array<{
|
|
133
|
-
runId: string;
|
|
134
|
-
score: number;
|
|
135
|
-
}>;
|
|
136
|
-
}
|
|
137
|
-
interface JudgeInsight {
|
|
138
|
-
/** Number of times this judge scored a run. */
|
|
139
|
-
n: number;
|
|
140
|
-
/** Mean composite over this judge's runs. */
|
|
141
|
-
meanScore: number;
|
|
142
|
-
/** Calibration against a gold reference, when provided. Cohen's κ for
|
|
143
|
-
* binary thresholding + continuous agreement metrics. */
|
|
144
|
-
calibration?: ContinuousAgreement;
|
|
145
|
-
/** Positional bias — when the judge sees options in different orders,
|
|
146
|
-
* do its preferences track the content or the position? */
|
|
147
|
-
positionalBias?: number;
|
|
148
|
-
/** Self-preference — when the judge sees its own model's output vs a
|
|
149
|
-
* competitor, does it over-pick its own? */
|
|
150
|
-
selfPreference?: number;
|
|
151
|
-
/** Verbosity bias — does the judge reward longer outputs regardless of
|
|
152
|
-
* quality? */
|
|
153
|
-
verbosityBias?: number;
|
|
154
|
-
}
|
|
155
|
-
interface InterRaterInsight {
|
|
156
|
-
/** Number of raters whose scores were aggregated. */
|
|
157
|
-
raters: number;
|
|
158
|
-
/** Number of runs every rater scored. */
|
|
159
|
-
jointlyRated: number;
|
|
160
|
-
/** Cohen's κ averaged across rater pairs. */
|
|
161
|
-
kappa: number;
|
|
162
|
-
/** Pairwise κ per rater pair (key = `"raterA::raterB"`). */
|
|
163
|
-
perPair: Record<string, number>;
|
|
164
|
-
/** Run ids where raters disagree the most — the high-value triage list. */
|
|
165
|
-
disagreementCases: Array<{
|
|
166
|
-
runId: string;
|
|
167
|
-
ratings: Array<{
|
|
168
|
-
rater: string;
|
|
169
|
-
score: number;
|
|
170
|
-
}>;
|
|
171
|
-
range: number;
|
|
172
|
-
}>;
|
|
173
|
-
}
|
|
174
|
-
interface LiftInsight {
|
|
175
|
-
baselineMean: number;
|
|
176
|
-
candidateMean: number;
|
|
177
|
-
/** Candidate − baseline. */
|
|
178
|
-
delta: number;
|
|
179
|
-
/** Lower / upper bound of bootstrap CI on the delta. */
|
|
180
|
-
ci95: [number, number];
|
|
181
|
-
/** Paired-t-test p-value. */
|
|
182
|
-
pValue: number;
|
|
183
|
-
/** Number of paired observations. */
|
|
184
|
-
n: number;
|
|
185
|
-
/** Cohen's d for the delta. */
|
|
186
|
-
cohensD: number;
|
|
187
|
-
/** Minimum detectable effect at current n, 80% power. */
|
|
188
|
-
mde: number;
|
|
189
|
-
/** Sample size needed to detect the observed delta at 80% power. */
|
|
190
|
-
requiredN: number;
|
|
191
|
-
}
|
|
192
|
-
interface FailureClusterInsight {
|
|
193
|
-
/** All clusters identified by the registry, ranked by share descending. */
|
|
194
|
-
clusters: Array<{
|
|
195
|
-
id: string;
|
|
196
|
-
name: string;
|
|
197
|
-
/** Fraction of failed runs in this cluster, 0..1. */
|
|
198
|
-
share: number;
|
|
199
|
-
/** Exemplar `runId`s (≤ 5) the consumer can drill into. */
|
|
200
|
-
exemplars: string[];
|
|
201
|
-
/** Short LLM-generated suggested fix when the registry supports it. */
|
|
202
|
-
suggestedFix?: string;
|
|
203
|
-
}>;
|
|
204
|
-
totalFailures: number;
|
|
205
|
-
}
|
|
206
|
-
/** Model-free failure breakdown over the structured `RunRecord.failureMode`
|
|
207
|
-
* enum. Unlike `failureClusters` (semantic, requires an LLM analyst), this
|
|
208
|
-
* is computed directly from the tags the harness already recorded — so a
|
|
209
|
-
* customer ingesting one batch with no judge/analyst still learns which
|
|
210
|
-
* named failure dominates. */
|
|
211
|
-
interface FailureModeTally {
|
|
212
|
-
/** The `failureMode` tag. */
|
|
213
|
-
mode: string;
|
|
214
|
-
/** Number of runs carrying this tag. */
|
|
215
|
-
count: number;
|
|
216
|
-
/** Share of the whole corpus, 0..1. */
|
|
217
|
-
share: number;
|
|
218
|
-
}
|
|
219
|
-
interface ContaminationInsight {
|
|
220
|
-
/** Canary phrases that leaked into outputs. */
|
|
221
|
-
leaks: number;
|
|
222
|
-
/** Holdout audit verdict — did any holdout-tagged run end up in the
|
|
223
|
-
* search/dev pool, or vice versa? */
|
|
224
|
-
holdoutAuditPassed: boolean;
|
|
225
|
-
details?: Array<{
|
|
226
|
-
runId: string;
|
|
227
|
-
canary: string;
|
|
228
|
-
matched: string;
|
|
229
|
-
}>;
|
|
230
|
-
}
|
|
231
|
-
interface OutcomeCorrelationInsight {
|
|
232
|
-
/** What outcome the consumer is correlating against (e.g.
|
|
233
|
-
* `'engagement_rate'`, `'approval_rate'`, `'downstream_pass'`). */
|
|
234
|
-
metric: string;
|
|
235
|
-
/** Number of (run, outcome) pairs used. */
|
|
236
|
-
n: number;
|
|
237
|
-
/** Pearson correlation between composite score and outcome. */
|
|
238
|
-
pearson: number;
|
|
239
|
-
/** Spearman rank correlation — robust to monotonic non-linearity. */
|
|
240
|
-
spearman: number;
|
|
241
|
-
/** When present, the simple linear reward model fit to the data. */
|
|
242
|
-
rewardModel?: {
|
|
243
|
-
intercept: number;
|
|
244
|
-
slope: number;
|
|
245
|
-
r2: number;
|
|
246
|
-
};
|
|
247
|
-
}
|
|
248
|
-
interface ReleaseSummary {
|
|
249
|
-
/** Overall verdict across axes — fail if any axis fails, else warn if any
|
|
250
|
-
* warns, else pass. */
|
|
251
|
-
status: 'pass' | 'warn' | 'fail';
|
|
252
|
-
axes: Array<{
|
|
253
|
-
name: 'quality-lift' | 'contamination' | 'composite-distribution';
|
|
254
|
-
status: 'pass' | 'warn' | 'fail';
|
|
255
|
-
detail: string;
|
|
256
|
-
}>;
|
|
257
|
-
/** Free-form issues surfaced beyond the standard axes. Empty by default;
|
|
258
|
-
* consumers can post-process to populate. */
|
|
259
|
-
issues: string[];
|
|
260
|
-
}
|
|
261
|
-
interface MetricDelta {
|
|
262
|
-
/** Current-period mean. */
|
|
263
|
-
current: number;
|
|
264
|
-
/** Baseline-period mean. */
|
|
265
|
-
baseline: number;
|
|
266
|
-
/** current - baseline. Positive means improved (or, for cost/duration,
|
|
267
|
-
* the consumer-side interpretation: "higher current" — semantic
|
|
268
|
-
* direction depends on the metric). */
|
|
269
|
-
delta: number;
|
|
270
|
-
/** Welch 95% confidence interval on the delta. Two-sample, unpaired —
|
|
271
|
-
* the baseline and current run sets may have different scenarios. */
|
|
272
|
-
ci95: [number, number];
|
|
273
|
-
/** Welch t-test p-value (two-sided). */
|
|
274
|
-
pValue: number;
|
|
275
|
-
/** Cohen's d (pooled stddev). Effect size, signed. */
|
|
276
|
-
cohensD: number;
|
|
277
|
-
/** Sample sizes. */
|
|
278
|
-
baselineN: number;
|
|
279
|
-
currentN: number;
|
|
280
|
-
/** True when p < 0.05 AND |d| >= 0.2 (small-effect threshold). The
|
|
281
|
-
* conjunction prevents large-effect-but-noisy and significant-but-
|
|
282
|
-
* tiny from triggering recommendations. */
|
|
283
|
-
significant: boolean;
|
|
284
|
-
}
|
|
285
|
-
interface PriorPeriodComparison {
|
|
286
|
-
/** Sample counts. */
|
|
287
|
-
baselineN: number;
|
|
288
|
-
currentN: number;
|
|
289
|
-
/** Optional human-readable label — "vs prior 7 days", "vs v3 release". */
|
|
290
|
-
windowLabel?: string;
|
|
291
|
-
/** Every metric we could compare. Keys: 'composite', 'cost', 'duration',
|
|
292
|
-
* 'tokenUsage' for always-present ones; per-dimension keys when both
|
|
293
|
-
* windows have judge scores on the same dimension. */
|
|
294
|
-
metrics: Record<string, MetricDelta>;
|
|
295
|
-
/** Metric names where current is significantly WORSE than baseline.
|
|
296
|
-
* Direction-aware: for cost/duration, higher current = worse. */
|
|
297
|
-
regressedMetrics: string[];
|
|
298
|
-
/** Metric names where current is significantly BETTER than baseline. */
|
|
299
|
-
improvedMetrics: string[];
|
|
300
|
-
}
|
|
301
|
-
interface Recommendation {
|
|
302
|
-
priority: 'critical' | 'high' | 'medium' | 'low';
|
|
303
|
-
kind: 'ship' | 'hold' | 'investigate' | 'fix' | 'recalibrate' | 'expand-corpus';
|
|
304
|
-
title: string;
|
|
305
|
-
detail: string;
|
|
306
|
-
/** Optional pointer back into the report for the evidence. */
|
|
307
|
-
evidencePath?: string;
|
|
308
|
-
}
|
|
309
|
-
|
|
310
|
-
export type { CostProvenanceSummary as C, FailureClusterInsight as F, InsightReport as I, JudgeInsight as J, LiftInsight as L, OutcomeCorrelationInsight as O, Recommendation as R, ScalarDistribution as S, InterRaterInsight as a, ReleaseSummary as b };
|
|
@@ -1,81 +0,0 @@
|
|
|
1
|
-
import { C as CaptureIntegrityError } from './errors-oeQrLqXC.js';
|
|
2
|
-
import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
|
|
3
|
-
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
4
|
-
|
|
5
|
-
/**
|
|
6
|
-
* Run-completion integrity check — at end of run, verify the expected event
|
|
7
|
-
* types were actually captured. The point is the launch-review failure mode:
|
|
8
|
-
* a run *appears* successful but the raw provider events were never written,
|
|
9
|
-
* so a downstream reviewer can't reconstruct what happened.
|
|
10
|
-
*
|
|
11
|
-
* Pattern:
|
|
12
|
-
*
|
|
13
|
-
* const report = await assertRunCaptured(store, runId, {
|
|
14
|
-
* llmSpansMin: 1,
|
|
15
|
-
* judgeSpansMin: 1,
|
|
16
|
-
* rawSink: providerSink, // must have ≥ 1 event for this run
|
|
17
|
-
* requireRawCoverageOfLlmSpans: true, // every llm span has matching raw events
|
|
18
|
-
* })
|
|
19
|
-
* if (!report.ok) throwIfRunIncomplete(report) // or mark run failed and continue
|
|
20
|
-
*
|
|
21
|
-
* The function is read-only on the store and returns a structured report;
|
|
22
|
-
* the caller chooses the failure mode (throw, mark run failed, log warning).
|
|
23
|
-
* `throwIfRunIncomplete` is the convenient strict mode.
|
|
24
|
-
*/
|
|
25
|
-
|
|
26
|
-
interface RunIntegrityExpectations {
|
|
27
|
-
/** Minimum LLM span count. Default 0 (no requirement). */
|
|
28
|
-
llmSpansMin?: number;
|
|
29
|
-
/** Minimum judge span count. Default 0. */
|
|
30
|
-
judgeSpansMin?: number;
|
|
31
|
-
/** Minimum tool span count. Default 0. */
|
|
32
|
-
toolSpansMin?: number;
|
|
33
|
-
/**
|
|
34
|
-
* Raw provider sink to consult for capture verification. When present,
|
|
35
|
-
* the check requires at least one raw event for the run.
|
|
36
|
-
*/
|
|
37
|
-
rawSink?: RawProviderSink;
|
|
38
|
-
/** Minimum raw provider event count. Default 0; ignored when `rawSink` absent. */
|
|
39
|
-
rawProviderEventsMin?: number;
|
|
40
|
-
/**
|
|
41
|
-
* Every LLM span must have at least one matching raw `request` event
|
|
42
|
-
* (matched by spanId). Catches the common bug where the structured span
|
|
43
|
-
* was emitted but the raw HTTP capture was wired to a different sink.
|
|
44
|
-
*/
|
|
45
|
-
requireRawCoverageOfLlmSpans?: boolean;
|
|
46
|
-
/** Run outcome must be set (not null/undefined). Default false. */
|
|
47
|
-
requireOutcome?: boolean;
|
|
48
|
-
}
|
|
49
|
-
type RunIntegrityIssueCode = 'no_run' | 'missing_llm_spans' | 'missing_judge_spans' | 'missing_tool_spans' | 'missing_raw_events' | 'no_raw_sink' | 'orphan_llm_span' | 'missing_outcome';
|
|
50
|
-
interface RunIntegrityIssue {
|
|
51
|
-
code: RunIntegrityIssueCode;
|
|
52
|
-
message: string;
|
|
53
|
-
detail?: Record<string, unknown>;
|
|
54
|
-
}
|
|
55
|
-
interface RunIntegrityReport {
|
|
56
|
-
ok: boolean;
|
|
57
|
-
runId: string;
|
|
58
|
-
llmSpanCount: number;
|
|
59
|
-
judgeSpanCount: number;
|
|
60
|
-
toolSpanCount: number;
|
|
61
|
-
rawProviderEventCount: number;
|
|
62
|
-
/**
|
|
63
|
-
* Coverage of LLM spans by raw provider events keyed on spanId.
|
|
64
|
-
* `total` is the number of LLM spans; `covered` is the count with at
|
|
65
|
-
* least one matching `request` raw event.
|
|
66
|
-
*/
|
|
67
|
-
rawSpanCoverage: {
|
|
68
|
-
covered: number;
|
|
69
|
-
total: number;
|
|
70
|
-
};
|
|
71
|
-
issues: RunIntegrityIssue[];
|
|
72
|
-
}
|
|
73
|
-
declare class RunIntegrityError extends CaptureIntegrityError {
|
|
74
|
-
readonly report: RunIntegrityReport;
|
|
75
|
-
constructor(report: RunIntegrityReport);
|
|
76
|
-
}
|
|
77
|
-
declare function assertRunCaptured(store: TraceStore, runId: string, expectations?: RunIntegrityExpectations): Promise<RunIntegrityReport>;
|
|
78
|
-
/** Strict mode: throws `RunIntegrityError` when the report isn't ok. */
|
|
79
|
-
declare function throwIfRunIncomplete(report: RunIntegrityReport): void;
|
|
80
|
-
|
|
81
|
-
export { type RunIntegrityExpectations as R, type RunIntegrityReport as a, RunIntegrityError as b, type RunIntegrityIssue as c, type RunIntegrityIssueCode as d, assertRunCaptured as e, throwIfRunIncomplete as t };
|
|
@@ -1,145 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Judge calibration — measure judge quality against human gold + bias.
|
|
3
|
-
*
|
|
4
|
-
* Workflow:
|
|
5
|
-
* 1. Build a golden set: {itemId, humanScore}[].
|
|
6
|
-
* 2. Run candidate judges; each produces {itemId, score}.
|
|
7
|
-
* 3. `calibrateJudge(golden, candidate)` reports κ + Pearson + MAE.
|
|
8
|
-
* 4. `calibrateJudgeContinuous(golden, candidate)` adds quadratic-weighted
|
|
9
|
-
* κ over the un-rounded [0,1] scores plus ICC(2,1), Pearson, Spearman,
|
|
10
|
-
* and bootstrap CIs — use this for fine-grained judges where rounding
|
|
11
|
-
* to int discards information (e.g. 0.78 vs 0.81 both round to 1 and
|
|
12
|
-
* look "perfectly agreed" to integer κ).
|
|
13
|
-
* 5. Run bias probes (positional, verbosity, self-preference) to
|
|
14
|
-
* detect systematic score inflation.
|
|
15
|
-
* 6. For N≥2 judges on the same items, `continuousAgreement(scores)`
|
|
16
|
-
* reports ICC(2,1) + κ_w + Pearson + Spearman with bootstrap CIs.
|
|
17
|
-
*
|
|
18
|
-
* Returns actionable diagnostics, not a single number. Consumers then
|
|
19
|
-
* decide whether to trust the judge, retrain it, or add a tie-breaker.
|
|
20
|
-
*/
|
|
21
|
-
interface GoldenItem {
|
|
22
|
-
itemId: string;
|
|
23
|
-
humanScore: number;
|
|
24
|
-
/** Optional group used for per-group bias audits (e.g. model-of-output family). */
|
|
25
|
-
group?: string;
|
|
26
|
-
}
|
|
27
|
-
interface CandidateScore {
|
|
28
|
-
itemId: string;
|
|
29
|
-
score: number;
|
|
30
|
-
/** Optional — enables positional-bias analysis (did order matter?). */
|
|
31
|
-
positionOfAInput?: 'first' | 'second';
|
|
32
|
-
}
|
|
33
|
-
interface CalibrationResult {
|
|
34
|
-
n: number;
|
|
35
|
-
pearson: number;
|
|
36
|
-
/** Cohen's κ with quadratic weights over integer-rounded scores. */
|
|
37
|
-
kappa: number;
|
|
38
|
-
/** Mean absolute error vs human. */
|
|
39
|
-
mae: number;
|
|
40
|
-
/** Worst-5 miscalibrations (largest |judge - human|). */
|
|
41
|
-
worstItems: Array<{
|
|
42
|
-
itemId: string;
|
|
43
|
-
judge: number;
|
|
44
|
-
human: number;
|
|
45
|
-
delta: number;
|
|
46
|
-
}>;
|
|
47
|
-
}
|
|
48
|
-
/**
|
|
49
|
-
* Measure judge quality against human gold labels: computes Cohen's κ, Pearson correlation, and MAE over matched item ids.
|
|
50
|
-
*/
|
|
51
|
-
declare function calibrateJudge(golden: GoldenItem[], candidate: CandidateScore[]): CalibrationResult;
|
|
52
|
-
interface PositionalBiasResult {
|
|
53
|
-
/**
|
|
54
|
-
* Score delta (first-position - second-position) averaged across items
|
|
55
|
-
* presented in both positions. Non-zero = positional bias.
|
|
56
|
-
*/
|
|
57
|
-
avgDelta: number;
|
|
58
|
-
n: number;
|
|
59
|
-
}
|
|
60
|
-
/**
|
|
61
|
-
* Feed the same items to the judge twice with A/B swapped and pass all
|
|
62
|
-
* results here. Items that don't appear in both positions are ignored.
|
|
63
|
-
*/
|
|
64
|
-
declare function positionalBias(scores: CandidateScore[]): PositionalBiasResult;
|
|
65
|
-
interface VerbosityBiasResult {
|
|
66
|
-
/** Pearson correlation between output length and score. Strong positive = verbosity bias. */
|
|
67
|
-
pearson: number;
|
|
68
|
-
n: number;
|
|
69
|
-
}
|
|
70
|
-
declare function verbosityBias(samples: Array<{
|
|
71
|
-
outputLen: number;
|
|
72
|
-
score: number;
|
|
73
|
-
}>): VerbosityBiasResult;
|
|
74
|
-
interface SelfPreferenceResult {
|
|
75
|
-
/** Mean judge score when judge's family matches output's family. */
|
|
76
|
-
inFamilyMean: number;
|
|
77
|
-
outOfFamilyMean: number;
|
|
78
|
-
deltaMean: number;
|
|
79
|
-
n: number;
|
|
80
|
-
}
|
|
81
|
-
/**
|
|
82
|
-
* Pass the same scenarios scored with judge-model X grading outputs from
|
|
83
|
-
* model X (in-family) and model Y (out-of-family). Non-zero delta
|
|
84
|
-
* indicates self-preference.
|
|
85
|
-
*/
|
|
86
|
-
declare function selfPreference(samples: Array<{
|
|
87
|
-
score: number;
|
|
88
|
-
inFamily: boolean;
|
|
89
|
-
}>): SelfPreferenceResult;
|
|
90
|
-
interface ContinuousAgreement {
|
|
91
|
-
/** Cohen's κ_w with quadratic weights, computed on raw [0,1] scores. */
|
|
92
|
-
weightedKappa: number;
|
|
93
|
-
/** ICC(2,1): two-way random effects, absolute agreement, single rater. */
|
|
94
|
-
icc: number;
|
|
95
|
-
/** Pearson product-moment correlation (averaged over rater pairs if N>2). */
|
|
96
|
-
pearson: number;
|
|
97
|
-
/** Spearman rank correlation (averaged over rater pairs if N>2). */
|
|
98
|
-
spearman: number;
|
|
99
|
-
/** 95% bootstrap percentile CIs over items. */
|
|
100
|
-
ci: {
|
|
101
|
-
icc: [number, number];
|
|
102
|
-
weightedKappa: [number, number];
|
|
103
|
-
};
|
|
104
|
-
/** Number of complete items (no NaN across raters). */
|
|
105
|
-
n: number;
|
|
106
|
-
/** Number of raters. */
|
|
107
|
-
raters: number;
|
|
108
|
-
}
|
|
109
|
-
interface ContinuousAgreementOptions {
|
|
110
|
-
/** Bootstrap iterations. Default 1000. Set to 0 to skip CIs (CI = [NaN, NaN]). */
|
|
111
|
-
bootstrap?: number;
|
|
112
|
-
/** κ weighting scheme. Default 'quadratic'. */
|
|
113
|
-
weights?: 'linear' | 'quadratic';
|
|
114
|
-
/** PRNG seed for reproducible bootstrap. Default 0xC0FFEE. */
|
|
115
|
-
seed?: number;
|
|
116
|
-
/** Confidence level for percentile CI. Default 0.95. */
|
|
117
|
-
ciLevel?: number;
|
|
118
|
-
}
|
|
119
|
-
/**
|
|
120
|
-
* Inter-rater agreement on continuous (typically [0,1]) scores.
|
|
121
|
-
*
|
|
122
|
-
* `scores` has shape [n_items][n_raters]. Rows with any non-finite entry
|
|
123
|
-
* are dropped. Returns NaN metrics if fewer than 2 raters or 2 complete
|
|
124
|
-
* items remain.
|
|
125
|
-
*/
|
|
126
|
-
declare function continuousAgreement(scores: number[][], opts?: ContinuousAgreementOptions): ContinuousAgreement;
|
|
127
|
-
interface ContinuousCalibrationResult extends CalibrationResult {
|
|
128
|
-
/** Cohen's κ_w computed on raw (un-rounded) scores. */
|
|
129
|
-
weightedKappaContinuous: number;
|
|
130
|
-
/** ICC(2,1) treating golden + candidate as two raters. */
|
|
131
|
-
icc: number;
|
|
132
|
-
spearman: number;
|
|
133
|
-
ci: {
|
|
134
|
-
icc: [number, number];
|
|
135
|
-
weightedKappa: [number, number];
|
|
136
|
-
};
|
|
137
|
-
}
|
|
138
|
-
/**
|
|
139
|
-
* Drop-in superset of `calibrateJudge` that adds continuous-value
|
|
140
|
-
* agreement metrics. The old fields (n, pearson, kappa, mae, worstItems)
|
|
141
|
-
* are preserved unchanged so existing callers continue to work.
|
|
142
|
-
*/
|
|
143
|
-
declare function calibrateJudgeContinuous(golden: GoldenItem[], candidate: CandidateScore[], opts?: ContinuousAgreementOptions): ContinuousCalibrationResult;
|
|
144
|
-
|
|
145
|
-
export { type ContinuousAgreement as C, type GoldenItem as G, type PositionalBiasResult as P, type SelfPreferenceResult as S, type VerbosityBiasResult as V, type CalibrationResult as a, type ContinuousCalibrationResult as b, type CandidateScore as c, type ContinuousAgreementOptions as d, calibrateJudge as e, calibrateJudgeContinuous as f, continuousAgreement as g, positionalBias as p, selfPreference as s, verbosityBias as v };
|