@kontourai/survey 1.8.0 → 1.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,195 @@
1
+ /**
2
+ * Extraction-confidence calibration from human review outcomes.
3
+ *
4
+ * Survey owns the review chain, so it owns the one signal no eval vendor can
5
+ * publish: for every reviewed candidate, the stated extraction/candidate
6
+ * confidence (the PREDICTION) against the human review decision (the LABEL).
7
+ * Grouping those labeled samples by extractor (and field) and binning by
8
+ * confidence yields an empirical calibration curve — "confidence in [0.8,0.9)
9
+ * from extractor X was affirmed 17/20 times" — plus a calibration gap and an
10
+ * empirically-grounded auto-accept threshold suggestion.
11
+ *
12
+ * ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration INFORMS policy — the
13
+ * suggested threshold an operator MAY wire into `autoAcceptMinConfidence` — it
14
+ * never decides a claim or mutates a status. Projected claims carry status
15
+ * "proposed", exactly like every other producer proposal.
16
+ *
17
+ * Machine auto-accepts are EXCLUDED by default: an auto-accepted outcome is the
18
+ * threshold accepting its own guess, so counting it as a "correct" label would
19
+ * let the policy validate itself (circular). Only human review outcomes are
20
+ * labeled samples. Set `includeAutoAccepted` to override.
21
+ *
22
+ * All computations are deterministic; `now` + `windowDays` window by `reviewedAt`.
23
+ */
24
+ import type { Claim, Evidence, TrustBundle, VerificationEvent } from "@kontourai/surface";
25
+ import type { CandidateSet, Extraction, ReviewOutcome } from "./types.js";
26
+ /**
27
+ * The three record arrays a calibration derivation reads — a subset of
28
+ * {@link SurveyInput}. Kept narrow so a caller can pass an existing batch's
29
+ * fields directly without constructing a whole SurveyInput.
30
+ */
31
+ export interface CalibrationInput {
32
+ readonly reviewOutcomes: readonly ReviewOutcome[];
33
+ readonly candidateSets: readonly CandidateSet[];
34
+ readonly extractions: readonly Extraction[];
35
+ }
36
+ export interface DeriveCalibrationOptions {
37
+ /**
38
+ * Current time — required only when `windowDays` is set (windowing is by
39
+ * `reviewedAt`). Injected for determinism; never read from the wall clock.
40
+ */
41
+ readonly now?: Date;
42
+ /**
43
+ * Optional rolling window in days. Review outcomes whose `reviewedAt` is older
44
+ * than `now - windowDays` are excluded. Requires `now` — setting `windowDays`
45
+ * without `now` throws (rather than silently disabling windowing). When
46
+ * omitted, all supplied outcomes are considered.
47
+ */
48
+ readonly windowDays?: number;
49
+ /** Number of equal-width confidence bins over [0,1]. Default 10 (deciles). */
50
+ readonly binCount?: number;
51
+ /**
52
+ * The empirical accuracy the suggested threshold must clear. Default 0.95.
53
+ * A `suggestedThreshold` is the lowest bin lower-bound at/above which every
54
+ * populated bin's empirical accuracy meets this target.
55
+ */
56
+ readonly targetAccuracy?: number;
57
+ /**
58
+ * A bin needs at least this many samples to count toward `suggestedThreshold`
59
+ * (both to qualify and to disqualify). Default 1. Raise it to avoid grounding
60
+ * a threshold on a bin with too little evidence.
61
+ */
62
+ readonly minBinSamples?: number;
63
+ /**
64
+ * Include machine auto-accepted outcomes (actor === AUTO_ACCEPT_ACTOR) as
65
+ * labeled samples. Default false — see the module note on circularity.
66
+ */
67
+ readonly includeAutoAccepted?: boolean;
68
+ }
69
+ /** One labeled calibration sample: a prediction paired with the human label. */
70
+ export interface CalibrationSample {
71
+ /** The extractor that produced the proposed value (or "unknown"). */
72
+ readonly extractor: string;
73
+ /** The extraction target / field the value belongs to. */
74
+ readonly field: string;
75
+ /** The proposed candidate's stated confidence, clamped to [0,1]. */
76
+ readonly predictedConfidence: number;
77
+ /** True iff the human review affirmed the proposed value. */
78
+ readonly correct: boolean;
79
+ /** The review outcome this sample came from. */
80
+ readonly reviewOutcomeId: string;
81
+ /** The candidate set the outcome reviewed. */
82
+ readonly candidateSetId: string;
83
+ /** The outcome's `reviewedAt`, when present. */
84
+ readonly reviewedAt: string | undefined;
85
+ }
86
+ /** An equal-width confidence bin with its empirical accuracy. */
87
+ export interface CalibrationBin {
88
+ /** Inclusive lower bound of the bin. */
89
+ readonly lowerBound: number;
90
+ /** Exclusive upper bound (inclusive at 1.0 for the top bin). */
91
+ readonly upperBound: number;
92
+ readonly sampleCount: number;
93
+ readonly correctCount: number;
94
+ /** correctCount / sampleCount; undefined when the bin has no samples. */
95
+ readonly empiricalAccuracy: number | undefined;
96
+ /** Mean predicted confidence of samples in the bin; undefined when empty. */
97
+ readonly meanPredictedConfidence: number | undefined;
98
+ }
99
+ /** A calibration rollup for one extractor (and optionally one field). */
100
+ export interface CalibrationGroup {
101
+ readonly extractor: string;
102
+ /** The field, or undefined for an extractor-level rollup across all fields. */
103
+ readonly field: string | undefined;
104
+ readonly sampleCount: number;
105
+ readonly correctCount: number;
106
+ /** correctCount / sampleCount; undefined when the group has no samples. */
107
+ readonly empiricalAccuracy: number | undefined;
108
+ /** Mean predicted confidence across the group; undefined when no samples. */
109
+ readonly meanPredictedConfidence: number | undefined;
110
+ /**
111
+ * meanPredictedConfidence − empiricalAccuracy. Positive → overconfident
112
+ * (states more confidence than the humans bear out); negative →
113
+ * underconfident. undefined when the group has no samples.
114
+ */
115
+ readonly calibrationGap: number | undefined;
116
+ /** Per-bin empirical accuracy, ascending by lowerBound. */
117
+ readonly bins: readonly CalibrationBin[];
118
+ /**
119
+ * Lowest bin lowerBound at/above which every populated bin (≥ minBinSamples)
120
+ * meets `targetAccuracy`, scanning the top-contiguous run of qualifying bins.
121
+ * undefined when no bin qualifies — the data does not yet support an empirical
122
+ * auto-accept threshold at that target. ADVISORY: an operator wires this into
123
+ * `autoAcceptMinConfidence`; calibration never sets it.
124
+ */
125
+ readonly suggestedThreshold: number | undefined;
126
+ }
127
+ export interface CalibrationMetrics {
128
+ /** One rollup per extractor (field === undefined), sorted by extractor. */
129
+ readonly byExtractor: readonly CalibrationGroup[];
130
+ /** One rollup per (extractor, field) pair, sorted by extractor then field. */
131
+ readonly byExtractorField: readonly CalibrationGroup[];
132
+ /** A single rollup across every labeled sample (extractor "*"). */
133
+ readonly overall: CalibrationGroup;
134
+ /** Number of labeled samples used. */
135
+ readonly sampleCount: number;
136
+ /** Outcomes that could not be turned into a labeled sample (see reasons). */
137
+ readonly skippedCount: number;
138
+ /** ISO 8601 timestamp of the earliest labeled sample's `reviewedAt`. */
139
+ readonly windowStart: string | undefined;
140
+ /** ISO 8601 timestamp of the latest labeled sample's `reviewedAt`. */
141
+ readonly windowEnd: string | undefined;
142
+ /** The `windowDays` option echoed back (undefined when not windowed). */
143
+ readonly windowDays: number | undefined;
144
+ }
145
+ /**
146
+ * Derives extractor/field confidence calibration from review outcomes.
147
+ *
148
+ * Each reviewed candidate set contributes one labeled sample: the confidence of
149
+ * the SYSTEM-proposed candidate (`CandidateSet.selectedCandidateId`) as the
150
+ * prediction, and whether the human review affirmed that proposed value as the
151
+ * label. A sample is skipped when it carries no human label or no prediction:
152
+ *
153
+ * - status "proposed" (not yet reviewed);
154
+ * - a machine auto-accept, unless `includeAutoAccepted` is set;
155
+ * - no `selectedCandidateId`, or the selected candidate / its confidence is
156
+ * missing or non-finite (no prediction to calibrate).
157
+ *
158
+ * A sample is "correct" when the outcome status is verified/assumed AND the
159
+ * reviewer did not switch to a different candidate; "incorrect" when the status
160
+ * is rejected or the reviewer overrode the proposed candidate.
161
+ */
162
+ export declare function deriveCalibration(input: CalibrationInput, options?: DeriveCalibrationOptions): CalibrationMetrics;
163
+ /** A single projected calibration claim triple. */
164
+ export interface CalibrationClaim {
165
+ readonly claim: Claim;
166
+ readonly evidence: Evidence;
167
+ readonly event: VerificationEvent;
168
+ }
169
+ export interface CalibrationClaimsSubject {
170
+ /** Surface subject type (e.g. "extractor"). */
171
+ readonly subjectType: string;
172
+ /** Surface subject id prefix (e.g. a producer or run id). */
173
+ readonly subjectId: string;
174
+ /** Surface facet (e.g. "review.calibration"). */
175
+ readonly facet: string;
176
+ /** Actor id to record on events. */
177
+ readonly actor: string;
178
+ /** ISO 8601 timestamp for claim created/updated times. */
179
+ readonly observedAt: string;
180
+ /** Producer identifier for the evidence `collectedBy` field. */
181
+ readonly collectedBy: string;
182
+ }
183
+ /**
184
+ * Projects per-extractor calibration as Surface-ready claims (claimType
185
+ * "calibration"). One claim per measurable metric per extractor: empirical
186
+ * accuracy, calibration gap, and — when the data supports it — the suggested
187
+ * auto-accept threshold. Every claim is status "proposed": calibration proposes,
188
+ * it never decides (ADR 0003 §4). Groups with no labeled samples are skipped.
189
+ */
190
+ export declare function calibrationToClaims(metrics: CalibrationMetrics, subject: CalibrationClaimsSubject): CalibrationClaim[];
191
+ /**
192
+ * Merges calibration claims into an existing TrustBundle. Convenience wrapper
193
+ * mirroring `mergeTrustBundleWithOversightMetrics`.
194
+ */
195
+ export declare function mergeTrustBundleWithCalibration(bundle: TrustBundle, claims: readonly CalibrationClaim[]): TrustBundle;
@@ -0,0 +1,361 @@
1
+ /**
2
+ * Extraction-confidence calibration from human review outcomes.
3
+ *
4
+ * Survey owns the review chain, so it owns the one signal no eval vendor can
5
+ * publish: for every reviewed candidate, the stated extraction/candidate
6
+ * confidence (the PREDICTION) against the human review decision (the LABEL).
7
+ * Grouping those labeled samples by extractor (and field) and binning by
8
+ * confidence yields an empirical calibration curve — "confidence in [0.8,0.9)
9
+ * from extractor X was affirmed 17/20 times" — plus a calibration gap and an
10
+ * empirically-grounded auto-accept threshold suggestion.
11
+ *
12
+ * ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration INFORMS policy — the
13
+ * suggested threshold an operator MAY wire into `autoAcceptMinConfidence` — it
14
+ * never decides a claim or mutates a status. Projected claims carry status
15
+ * "proposed", exactly like every other producer proposal.
16
+ *
17
+ * Machine auto-accepts are EXCLUDED by default: an auto-accepted outcome is the
18
+ * threshold accepting its own guess, so counting it as a "correct" label would
19
+ * let the policy validate itself (circular). Only human review outcomes are
20
+ * labeled samples. Set `includeAutoAccepted` to override.
21
+ *
22
+ * All computations are deterministic; `now` + `windowDays` window by `reviewedAt`.
23
+ */
24
+ import { AUTO_ACCEPT_ACTOR } from "./producer-profile.js";
25
+ const DEFAULT_BIN_COUNT = 10;
26
+ const DEFAULT_TARGET_ACCURACY = 0.95;
27
+ const DEFAULT_MIN_BIN_SAMPLES = 1;
28
+ // ---------------------------------------------------------------------------
29
+ // deriveCalibration
30
+ // ---------------------------------------------------------------------------
31
+ /**
32
+ * Derives extractor/field confidence calibration from review outcomes.
33
+ *
34
+ * Each reviewed candidate set contributes one labeled sample: the confidence of
35
+ * the SYSTEM-proposed candidate (`CandidateSet.selectedCandidateId`) as the
36
+ * prediction, and whether the human review affirmed that proposed value as the
37
+ * label. A sample is skipped when it carries no human label or no prediction:
38
+ *
39
+ * - status "proposed" (not yet reviewed);
40
+ * - a machine auto-accept, unless `includeAutoAccepted` is set;
41
+ * - no `selectedCandidateId`, or the selected candidate / its confidence is
42
+ * missing or non-finite (no prediction to calibrate).
43
+ *
44
+ * A sample is "correct" when the outcome status is verified/assumed AND the
45
+ * reviewer did not switch to a different candidate; "incorrect" when the status
46
+ * is rejected or the reviewer overrode the proposed candidate.
47
+ */
48
+ export function deriveCalibration(input, options = {}) {
49
+ const binCount = normalizeBinCount(options.binCount);
50
+ const targetAccuracy = options.targetAccuracy ?? DEFAULT_TARGET_ACCURACY;
51
+ const minBinSamples = options.minBinSamples ?? DEFAULT_MIN_BIN_SAMPLES;
52
+ const includeAutoAccepted = options.includeAutoAccepted ?? false;
53
+ const candidateSetById = new Map(input.candidateSets.map((cs) => [cs.id, cs]));
54
+ const candidateById = new Map();
55
+ for (const cs of input.candidateSets) {
56
+ for (const c of cs.candidates)
57
+ candidateById.set(c.id, c);
58
+ }
59
+ const extractionById = new Map(input.extractions.map((e) => [e.id, e]));
60
+ if (options.windowDays !== undefined && options.now === undefined) {
61
+ throw new RangeError("deriveCalibration: `now` is required when `windowDays` is set (windowing is by reviewedAt).");
62
+ }
63
+ const cutoff = options.windowDays !== undefined && options.now !== undefined
64
+ ? options.now.getTime() - options.windowDays * 24 * 60 * 60 * 1000
65
+ : undefined;
66
+ const samples = [];
67
+ let skippedCount = 0;
68
+ for (const outcome of input.reviewOutcomes) {
69
+ if (cutoff !== undefined) {
70
+ const t = outcome.reviewedAt ? Date.parse(outcome.reviewedAt) : NaN;
71
+ if (isNaN(t) || t < cutoff) {
72
+ skippedCount++;
73
+ continue;
74
+ }
75
+ }
76
+ const sample = toSample(outcome, candidateSetById, candidateById, extractionById, includeAutoAccepted);
77
+ if (sample === undefined) {
78
+ skippedCount++;
79
+ continue;
80
+ }
81
+ samples.push(sample);
82
+ }
83
+ const timestamps = samples
84
+ .map((s) => (s.reviewedAt ? Date.parse(s.reviewedAt) : NaN))
85
+ .filter((t) => !isNaN(t))
86
+ .sort((a, b) => a - b);
87
+ const windowStart = timestamps.length > 0 ? new Date(timestamps[0]).toISOString() : undefined;
88
+ const windowEnd = timestamps.length > 0 ? new Date(timestamps[timestamps.length - 1]).toISOString() : undefined;
89
+ const buildGroup = (extractor, field, groupSamples) => computeGroup(extractor, field, groupSamples, binCount, targetAccuracy, minBinSamples);
90
+ // Extractor-level rollups.
91
+ const byExtractorMap = new Map();
92
+ // (extractor, field) rollups. The key is a JSON-encoded [extractor, field]
93
+ // pair so no in-band delimiter can collide with an extractor/field that
94
+ // contains that delimiter; the extractor and field are carried in the value,
95
+ // never parsed back out of the key.
96
+ const byFieldMap = new Map();
97
+ for (const s of samples) {
98
+ pushTo(byExtractorMap, s.extractor, s);
99
+ const fieldKey = JSON.stringify([s.extractor, s.field]);
100
+ const existing = byFieldMap.get(fieldKey);
101
+ if (existing)
102
+ existing.samples.push(s);
103
+ else
104
+ byFieldMap.set(fieldKey, { extractor: s.extractor, field: s.field, samples: [s] });
105
+ }
106
+ const byExtractor = [...byExtractorMap.entries()]
107
+ .sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0))
108
+ .map(([extractor, groupSamples]) => buildGroup(extractor, undefined, groupSamples));
109
+ const byExtractorField = [...byFieldMap.values()]
110
+ .sort((a, b) => (a.extractor < b.extractor ? -1 : a.extractor > b.extractor ? 1 : a.field < b.field ? -1 : a.field > b.field ? 1 : 0))
111
+ .map(({ extractor, field, samples: groupSamples }) => buildGroup(extractor, field, groupSamples));
112
+ const overall = buildGroup("*", undefined, samples);
113
+ return {
114
+ byExtractor,
115
+ byExtractorField,
116
+ overall,
117
+ sampleCount: samples.length,
118
+ skippedCount,
119
+ windowStart,
120
+ windowEnd,
121
+ windowDays: options.windowDays,
122
+ };
123
+ }
124
+ // ---------------------------------------------------------------------------
125
+ // Internal: sample construction
126
+ // ---------------------------------------------------------------------------
127
+ function toSample(outcome, candidateSetById, candidateById, extractionById, includeAutoAccepted) {
128
+ // No human label yet.
129
+ if (outcome.status === "proposed")
130
+ return undefined;
131
+ // Machine auto-accepts are not human labels (circular) unless opted in.
132
+ if (!includeAutoAccepted && outcome.actor === AUTO_ACCEPT_ACTOR)
133
+ return undefined;
134
+ const candidateSet = candidateSetById.get(outcome.candidateSetId);
135
+ if (candidateSet === undefined)
136
+ return undefined;
137
+ // The prediction is the SYSTEM-proposed candidate's confidence.
138
+ const proposedId = candidateSet.selectedCandidateId;
139
+ if (proposedId === undefined)
140
+ return undefined;
141
+ const proposed = candidateById.get(proposedId);
142
+ if (proposed === undefined)
143
+ return undefined;
144
+ const extraction = proposed.extractionId ? extractionById.get(proposed.extractionId) : undefined;
145
+ const rawConfidence = proposed.confidence ?? extraction?.confidence;
146
+ if (rawConfidence === undefined || !Number.isFinite(rawConfidence))
147
+ return undefined;
148
+ const predictedConfidence = clamp01(rawConfidence);
149
+ const extractor = extraction?.extractor ?? "unknown";
150
+ const field = extraction?.target ?? candidateSet.target;
151
+ // The label: did the human affirm the proposed value?
152
+ let correct;
153
+ if (outcome.status === "rejected") {
154
+ correct = false;
155
+ }
156
+ else {
157
+ // verified | assumed — an override to a different candidate means the
158
+ // proposed value did NOT stand.
159
+ const overrode = outcome.candidateId !== undefined && outcome.candidateId !== proposedId;
160
+ correct = !overrode;
161
+ }
162
+ return {
163
+ extractor,
164
+ field,
165
+ predictedConfidence,
166
+ correct,
167
+ reviewOutcomeId: outcome.id,
168
+ candidateSetId: outcome.candidateSetId,
169
+ reviewedAt: outcome.reviewedAt,
170
+ };
171
+ }
172
+ // ---------------------------------------------------------------------------
173
+ // Internal: group + bin computation
174
+ // ---------------------------------------------------------------------------
175
+ function computeGroup(extractor, field, samples, binCount, targetAccuracy, minBinSamples) {
176
+ const sampleCount = samples.length;
177
+ const correctCount = samples.filter((s) => s.correct).length;
178
+ const empiricalAccuracy = sampleCount > 0 ? correctCount / sampleCount : undefined;
179
+ const meanPredictedConfidence = sampleCount > 0
180
+ ? samples.reduce((sum, s) => sum + s.predictedConfidence, 0) / sampleCount
181
+ : undefined;
182
+ const calibrationGap = meanPredictedConfidence !== undefined && empiricalAccuracy !== undefined
183
+ ? meanPredictedConfidence - empiricalAccuracy
184
+ : undefined;
185
+ const bins = computeBins(samples, binCount);
186
+ const suggestedThreshold = computeSuggestedThreshold(bins, targetAccuracy, minBinSamples);
187
+ return {
188
+ extractor,
189
+ field,
190
+ sampleCount,
191
+ correctCount,
192
+ empiricalAccuracy: round(empiricalAccuracy),
193
+ meanPredictedConfidence: round(meanPredictedConfidence),
194
+ calibrationGap: round(calibrationGap),
195
+ bins,
196
+ suggestedThreshold,
197
+ };
198
+ }
199
+ function computeBins(samples, binCount) {
200
+ const width = 1 / binCount;
201
+ const counts = Array.from({ length: binCount }, () => ({ n: 0, correct: 0, sum: 0 }));
202
+ for (const s of samples) {
203
+ const idx = Math.min(binCount - 1, Math.floor(s.predictedConfidence * binCount));
204
+ const bucket = counts[idx];
205
+ bucket.n++;
206
+ bucket.sum += s.predictedConfidence;
207
+ if (s.correct)
208
+ bucket.correct++;
209
+ }
210
+ return counts.map((b, i) => ({
211
+ lowerBound: round(i * width),
212
+ upperBound: round((i + 1) * width),
213
+ sampleCount: b.n,
214
+ correctCount: b.correct,
215
+ empiricalAccuracy: b.n > 0 ? round(b.correct / b.n) : undefined,
216
+ meanPredictedConfidence: b.n > 0 ? round(b.sum / b.n) : undefined,
217
+ }));
218
+ }
219
+ /**
220
+ * The suggested threshold is the lowerBound of the lowest bin in the
221
+ * top-contiguous run of bins that each (a) have ≥ minBinSamples and (b) meet
222
+ * targetAccuracy. Scanning from the highest bin down, a populated bin that
223
+ * fails the target — or an under-sampled bin we cannot vouch for — ends the run.
224
+ * undefined when even the top populated bin does not qualify.
225
+ */
226
+ function computeSuggestedThreshold(bins, targetAccuracy, minBinSamples) {
227
+ let threshold;
228
+ for (let i = bins.length - 1; i >= 0; i--) {
229
+ const bin = bins[i];
230
+ if (bin.sampleCount < minBinSamples)
231
+ break;
232
+ if (bin.empiricalAccuracy === undefined || bin.empiricalAccuracy < targetAccuracy)
233
+ break;
234
+ threshold = bin.lowerBound;
235
+ }
236
+ return threshold;
237
+ }
238
+ /**
239
+ * Projects per-extractor calibration as Surface-ready claims (claimType
240
+ * "calibration"). One claim per measurable metric per extractor: empirical
241
+ * accuracy, calibration gap, and — when the data supports it — the suggested
242
+ * auto-accept threshold. Every claim is status "proposed": calibration proposes,
243
+ * it never decides (ADR 0003 §4). Groups with no labeled samples are skipped.
244
+ */
245
+ export function calibrationToClaims(metrics, subject) {
246
+ const results = [];
247
+ for (const group of metrics.byExtractor) {
248
+ if (group.sampleCount === 0)
249
+ continue;
250
+ const base = `${sanitize(group.extractor)}`;
251
+ const push = (metric, value, detail) => {
252
+ const claimId = `calibration.${subject.subjectId}.${base}.${metric}`;
253
+ const evidenceId = `${claimId}.evidence`;
254
+ const claim = {
255
+ id: claimId,
256
+ subjectType: subject.subjectType,
257
+ subjectId: `${subject.subjectId}.${group.extractor}`,
258
+ facet: subject.facet,
259
+ claimType: "calibration",
260
+ fieldOrBehavior: metric,
261
+ value,
262
+ status: "proposed",
263
+ createdAt: subject.observedAt,
264
+ updatedAt: subject.observedAt,
265
+ impactLevel: "medium",
266
+ confidenceBasis: {
267
+ sourceQuality: "moderate",
268
+ reviewerAuthority: "none",
269
+ evidenceStrength: "weak",
270
+ impactLevel: "medium",
271
+ },
272
+ metadata: {
273
+ calibration: {
274
+ extractor: group.extractor,
275
+ sampleCount: group.sampleCount,
276
+ correctCount: group.correctCount,
277
+ empiricalAccuracy: group.empiricalAccuracy,
278
+ meanPredictedConfidence: group.meanPredictedConfidence,
279
+ calibrationGap: group.calibrationGap,
280
+ suggestedThreshold: group.suggestedThreshold,
281
+ windowStart: metrics.windowStart,
282
+ windowEnd: metrics.windowEnd,
283
+ windowDays: metrics.windowDays,
284
+ },
285
+ },
286
+ };
287
+ const evidence = {
288
+ id: evidenceId,
289
+ claimId,
290
+ evidenceType: "attestation",
291
+ method: "extraction",
292
+ sourceRef: "calibration://computed",
293
+ excerptOrSummary: `calibration.${metric} for extractor ${group.extractor}: ${detail}; ` +
294
+ `over ${group.correctCount}/${group.sampleCount} affirmed human review outcomes`,
295
+ observedAt: subject.observedAt,
296
+ collectedBy: subject.collectedBy,
297
+ };
298
+ const event = {
299
+ id: `${claimId}.event`,
300
+ claimId,
301
+ status: "proposed",
302
+ actor: subject.actor,
303
+ method: "candidate-proposal",
304
+ evidenceIds: [evidenceId],
305
+ createdAt: subject.observedAt,
306
+ };
307
+ results.push({ claim, evidence, event });
308
+ };
309
+ if (group.empiricalAccuracy !== undefined) {
310
+ push("empiricalAccuracy", group.empiricalAccuracy, `empirical accuracy = ${group.empiricalAccuracy}`);
311
+ }
312
+ if (group.calibrationGap !== undefined) {
313
+ push("calibrationGap", group.calibrationGap, `mean confidence − empirical accuracy = ${group.calibrationGap}`);
314
+ }
315
+ if (group.suggestedThreshold !== undefined) {
316
+ push("suggestedThreshold", group.suggestedThreshold, `advisory auto-accept threshold = ${group.suggestedThreshold}`);
317
+ }
318
+ }
319
+ return results;
320
+ }
321
+ /**
322
+ * Merges calibration claims into an existing TrustBundle. Convenience wrapper
323
+ * mirroring `mergeTrustBundleWithOversightMetrics`.
324
+ */
325
+ export function mergeTrustBundleWithCalibration(bundle, claims) {
326
+ return {
327
+ ...bundle,
328
+ claims: [...bundle.claims, ...claims.map((c) => c.claim)],
329
+ evidence: [...bundle.evidence, ...claims.map((c) => c.evidence)],
330
+ events: [...bundle.events, ...claims.map((c) => c.event)],
331
+ };
332
+ }
333
+ // ---------------------------------------------------------------------------
334
+ // Internal helpers
335
+ // ---------------------------------------------------------------------------
336
+ function pushTo(map, key, sample) {
337
+ const existing = map.get(key);
338
+ if (existing)
339
+ existing.push(sample);
340
+ else
341
+ map.set(key, [sample]);
342
+ }
343
+ function clamp01(value) {
344
+ return value < 0 ? 0 : value > 1 ? 1 : value;
345
+ }
346
+ function normalizeBinCount(binCount) {
347
+ if (binCount === undefined)
348
+ return DEFAULT_BIN_COUNT;
349
+ if (!Number.isInteger(binCount) || binCount < 1) {
350
+ throw new RangeError(`binCount must be a positive integer, received ${binCount}`);
351
+ }
352
+ return binCount;
353
+ }
354
+ function round(value) {
355
+ if (value === undefined)
356
+ return undefined;
357
+ return Math.round(value * 10000) / 10000;
358
+ }
359
+ function sanitize(id) {
360
+ return id.replace(/[^A-Za-z0-9._-]/g, "_");
361
+ }
@@ -40,3 +40,5 @@ export { currentProposedReviewItem } from "./current-proposed-review-item.js";
40
40
  export type { CurrentProposedCandidateInput, CurrentProposedReviewItemInput, } from "./current-proposed-review-item.js";
41
41
  export { deriveOversightMetrics, mergeTrustBundleWithOversightMetrics, oversightMetricsToClaims, } from "./oversight-metrics.js";
42
42
  export type { AggregateOversightMetrics, DeriveOversightMetricsOptions, OversightMetrics, OversightMetricsClaimsSubject, OversightQualityClaim, ReviewerOversightMetrics, } from "./oversight-metrics.js";
43
+ export { calibrationToClaims, deriveCalibration, mergeTrustBundleWithCalibration, } from "./calibration.js";
44
+ export type { CalibrationBin, CalibrationClaim, CalibrationClaimsSubject, CalibrationGroup, CalibrationInput, CalibrationMetrics, CalibrationSample, DeriveCalibrationOptions, } from "./calibration.js";
package/dist/src/index.js CHANGED
@@ -19,3 +19,4 @@ export { buildAuthorizedActionAuthorizing, buildPromptRef, isValidAuthorizing, v
19
19
  export { confidenceBasisForReview, defineProductVocabulary, stableId } from "./vocabulary.js";
20
20
  export { currentProposedReviewItem } from "./current-proposed-review-item.js";
21
21
  export { deriveOversightMetrics, mergeTrustBundleWithOversightMetrics, oversightMetricsToClaims, } from "./oversight-metrics.js";
22
+ export { calibrationToClaims, deriveCalibration, mergeTrustBundleWithCalibration, } from "./calibration.js";
@@ -1,6 +1,34 @@
1
1
  import type { TrustBundle } from "@kontourai/surface";
2
+ import { type CalibrationMetrics } from "./calibration.js";
2
3
  import type { SurveyInput } from "./types.js";
4
+ export interface SurveyCalibrationOptions {
5
+ /**
6
+ * Precomputed calibration to source the value from — typically derived over a
7
+ * LONGER history than the current batch (a better-grounded curve, and it avoids
8
+ * the mild self-reference of a claim's own review outcome feeding its value).
9
+ * When omitted, calibration is derived from THIS batch's review outcomes.
10
+ */
11
+ metrics?: CalibrationMetrics;
12
+ /**
13
+ * Minimum labeled samples a group needs before its accuracy is emitted as a
14
+ * value. Groups below the floor leave `value` unset rather than emitting a
15
+ * poorly-grounded number. Default {@link DEFAULT_CALIBRATION_MIN_SAMPLES}.
16
+ */
17
+ minSamples?: number;
18
+ }
3
19
  export interface BuildSurveyTrustBundleOptions {
4
20
  reviewProofs?: boolean;
21
+ /**
22
+ * Populate `conclusionConfidence.value` from empirical review calibration —
23
+ * "how often this extractor's proposals at this confidence were affirmed by a
24
+ * human reviewer" (the produce side of the confidence loop; see #114/#137).
25
+ * `true` derives calibration from this batch; an object supplies precomputed
26
+ * metrics and/or a `minSamples` floor. Absent → `value` stays unset and only
27
+ * the comfort-zone signal is carried (unchanged behavior).
28
+ *
29
+ * ADVISORY (ADR 0003 §4): this only enriches the emitted conclusion confidence;
30
+ * it never changes a claim's `status`.
31
+ */
32
+ calibration?: boolean | SurveyCalibrationOptions;
5
33
  }
6
34
  export declare function buildSurveyTrustBundle(input: SurveyInput, options?: BuildSurveyTrustBundleOptions): TrustBundle;
@@ -1,10 +1,23 @@
1
1
  import { buildReviewProofAnchor } from "./review-proof.js";
2
2
  import { assertReviewOutcomeDiscipline } from "./producer-discipline.js";
3
+ import { deriveCalibration } from "./calibration.js";
4
+ /** Minimum labeled samples a calibration group needs before its empirical
5
+ * accuracy is emitted as a `conclusionConfidence.value`. */
6
+ const DEFAULT_CALIBRATION_MIN_SAMPLES = 20;
3
7
  export function buildSurveyTrustBundle(input, options = {}) {
4
8
  const rawSources = indexById(input.rawSources, "raw source");
5
9
  const extractions = indexById(input.extractions, "extraction");
6
10
  const candidateSets = indexById(input.candidateSets, "candidate set");
7
11
  const reviewsByCandidateSet = groupBy(input.reviewOutcomes, (review) => review.candidateSetId);
12
+ const calibrationOptions = normalizeCalibrationOptions(options.calibration);
13
+ const calibrationMetrics = calibrationOptions
14
+ ? (calibrationOptions.metrics ?? deriveCalibration({
15
+ reviewOutcomes: input.reviewOutcomes,
16
+ candidateSets: input.candidateSets,
17
+ extractions: input.extractions,
18
+ }))
19
+ : undefined;
20
+ const calibrationMinSamples = calibrationOptions?.minSamples ?? DEFAULT_CALIBRATION_MIN_SAMPLES;
8
21
  const claims = [];
9
22
  const evidence = [];
10
23
  const events = [];
@@ -47,6 +60,37 @@ export function buildSurveyTrustBundle(input, options = {}) {
47
60
  survey: buildSurveyMetadata({ projection, rawSource, extraction, candidateSet, candidate, review }),
48
61
  },
49
62
  };
63
+ // Promote the review's comfort-zone signal — and, when calibration is
64
+ // enabled, an empirically-calibrated conclusion probability — into the
65
+ // first-class conclusionConfidence field (Surface 2.9 / Hachure 0.14) so the
66
+ // signal is portable and comparable, not buried in producer metadata.
67
+ //
68
+ // comfortZone is CARRIED from the review. `value` is PRODUCED from empirical
69
+ // review calibration (#114/#137): the affirmation rate of this extractor's
70
+ // proposals at this confidence — a calibrated conclusion probability, distinct
71
+ // from the extraction-confidence ingredient in confidenceBasis.
72
+ //
73
+ // A value is produced only for an AFFIRMED conclusion (status verified/assumed)
74
+ // that clears the sample floor. conclusionConfidence.value is "probability the
75
+ // conclusion is correct"; attaching an affirmation rate to a REJECTED (or
76
+ // not-yet-reviewed) conclusion would assert the opposite of what the human
77
+ // decided, so those claims get no value.
78
+ const comfortZone = review?.withinComfortZone !== undefined
79
+ ? {
80
+ within: review.withinComfortZone,
81
+ ...(review.comfortZoneNote ? { reason: review.comfortZoneNote } : {}),
82
+ }
83
+ : undefined;
84
+ const affirmedConclusion = status === "verified" || status === "assumed";
85
+ const calibrated = calibrationMetrics && review && affirmedConclusion
86
+ ? lookupCalibratedValue(calibrationMetrics, extraction.extractor, extraction.target, calibrationMinSamples)
87
+ : undefined;
88
+ if (comfortZone || calibrated) {
89
+ claim.conclusionConfidence = {
90
+ ...(calibrated ? { value: calibrated.value, method: calibrated.method } : {}),
91
+ ...(comfortZone ? { comfortZone } : {}),
92
+ };
93
+ }
50
94
  if (options.reviewProofs && review) {
51
95
  claim.currentIntegrityAnchor = buildReviewProofAnchor({
52
96
  rawSource,
@@ -356,6 +400,31 @@ function selectCandidate(candidateSet, candidateId) {
356
400
  function selectReview(reviews, candidateId) {
357
401
  return reviews.find((review) => review.candidateId === candidateId) ?? reviews.find((review) => !review.candidateId);
358
402
  }
403
+ function normalizeCalibrationOptions(calibration) {
404
+ if (calibration === undefined || calibration === false)
405
+ return undefined;
406
+ if (calibration === true)
407
+ return {};
408
+ return calibration;
409
+ }
410
+ /**
411
+ * Looks up the empirical affirmation rate for an extractor/field, preferring the
412
+ * finer (extractor, field) group and falling back to the extractor-level group
413
+ * when the field group is below the sample floor. Returns undefined when neither
414
+ * group clears the floor, so an ungrounded claim leaves `value` unset. The
415
+ * `method` records which granularity produced the value.
416
+ */
417
+ function lookupCalibratedValue(metrics, extractor, field, minSamples) {
418
+ const fieldGroup = metrics.byExtractorField.find((g) => g.extractor === extractor && g.field === field);
419
+ if (fieldGroup && fieldGroup.sampleCount >= minSamples && fieldGroup.empiricalAccuracy !== undefined) {
420
+ return { value: fieldGroup.empiricalAccuracy, method: "empirical-review-calibration:extractor-field" };
421
+ }
422
+ const extractorGroup = metrics.byExtractor.find((g) => g.extractor === extractor);
423
+ if (extractorGroup && extractorGroup.sampleCount >= minSamples && extractorGroup.empiricalAccuracy !== undefined) {
424
+ return { value: extractorGroup.empiricalAccuracy, method: "empirical-review-calibration:extractor" };
425
+ }
426
+ return undefined;
427
+ }
359
428
  function evidenceTypeFor(rawSource) {
360
429
  if (rawSource.kind === "policy-standard")
361
430
  return "policy_rule";
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@kontourai/survey",
3
- "version": "1.8.0",
3
+ "version": "1.10.0",
4
4
  "description": "Producer-side source, extraction, candidate, and review contracts for projecting verified claims into Surface.",
5
5
  "license": "Apache-2.0",
6
6
  "type": "module",
@@ -71,7 +71,7 @@
71
71
  "check:generated-css": "node scripts/copy-review-workbench-package-assets.cjs --check"
72
72
  },
73
73
  "dependencies": {
74
- "@kontourai/surface": "^2.0.0"
74
+ "@kontourai/surface": "^2.9.0"
75
75
  },
76
76
  "peerDependencies": {
77
77
  "@anthropic-ai/sdk": ">=0.20.0",