@kontourai/survey 1.7.0 → 1.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,195 @@
1
+ /**
2
+ * Extraction-confidence calibration from human review outcomes.
3
+ *
4
+ * Survey owns the review chain, so it owns the one signal no eval vendor can
5
+ * publish: for every reviewed candidate, the stated extraction/candidate
6
+ * confidence (the PREDICTION) against the human review decision (the LABEL).
7
+ * Grouping those labeled samples by extractor (and field) and binning by
8
+ * confidence yields an empirical calibration curve — "confidence in [0.8,0.9)
9
+ * from extractor X was affirmed 17/20 times" — plus a calibration gap and an
10
+ * empirically-grounded auto-accept threshold suggestion.
11
+ *
12
+ * ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration INFORMS policy — the
13
+ * suggested threshold an operator MAY wire into `autoAcceptMinConfidence` — it
14
+ * never decides a claim or mutates a status. Projected claims carry status
15
+ * "proposed", exactly like every other producer proposal.
16
+ *
17
+ * Machine auto-accepts are EXCLUDED by default: an auto-accepted outcome is the
18
+ * threshold accepting its own guess, so counting it as a "correct" label would
19
+ * let the policy validate itself (circular). Only human review outcomes are
20
+ * labeled samples. Set `includeAutoAccepted` to override.
21
+ *
22
+ * All computations are deterministic; `now` + `windowDays` window by `reviewedAt`.
23
+ */
24
+ import type { Claim, Evidence, TrustBundle, VerificationEvent } from "@kontourai/surface";
25
+ import type { CandidateSet, Extraction, ReviewOutcome } from "./types.js";
26
+ /**
27
+ * The three record arrays a calibration derivation reads — a subset of
28
+ * {@link SurveyInput}. Kept narrow so a caller can pass an existing batch's
29
+ * fields directly without constructing a whole SurveyInput.
30
+ */
31
+ export interface CalibrationInput {
32
+ readonly reviewOutcomes: readonly ReviewOutcome[];
33
+ readonly candidateSets: readonly CandidateSet[];
34
+ readonly extractions: readonly Extraction[];
35
+ }
36
+ export interface DeriveCalibrationOptions {
37
+ /**
38
+ * Current time — required only when `windowDays` is set (windowing is by
39
+ * `reviewedAt`). Injected for determinism; never read from the wall clock.
40
+ */
41
+ readonly now?: Date;
42
+ /**
43
+ * Optional rolling window in days. Review outcomes whose `reviewedAt` is older
44
+ * than `now - windowDays` are excluded. Requires `now` — setting `windowDays`
45
+ * without `now` throws (rather than silently disabling windowing). When
46
+ * omitted, all supplied outcomes are considered.
47
+ */
48
+ readonly windowDays?: number;
49
+ /** Number of equal-width confidence bins over [0,1]. Default 10 (deciles). */
50
+ readonly binCount?: number;
51
+ /**
52
+ * The empirical accuracy the suggested threshold must clear. Default 0.95.
53
+ * A `suggestedThreshold` is the lowest bin lower-bound at/above which every
54
+ * populated bin's empirical accuracy meets this target.
55
+ */
56
+ readonly targetAccuracy?: number;
57
+ /**
58
+ * A bin needs at least this many samples to count toward `suggestedThreshold`
59
+ * (both to qualify and to disqualify). Default 1. Raise it to avoid grounding
60
+ * a threshold on a bin with too little evidence.
61
+ */
62
+ readonly minBinSamples?: number;
63
+ /**
64
+ * Include machine auto-accepted outcomes (actor === AUTO_ACCEPT_ACTOR) as
65
+ * labeled samples. Default false — see the module note on circularity.
66
+ */
67
+ readonly includeAutoAccepted?: boolean;
68
+ }
69
+ /** One labeled calibration sample: a prediction paired with the human label. */
70
+ export interface CalibrationSample {
71
+ /** The extractor that produced the proposed value (or "unknown"). */
72
+ readonly extractor: string;
73
+ /** The extraction target / field the value belongs to. */
74
+ readonly field: string;
75
+ /** The proposed candidate's stated confidence, clamped to [0,1]. */
76
+ readonly predictedConfidence: number;
77
+ /** True iff the human review affirmed the proposed value. */
78
+ readonly correct: boolean;
79
+ /** The review outcome this sample came from. */
80
+ readonly reviewOutcomeId: string;
81
+ /** The candidate set the outcome reviewed. */
82
+ readonly candidateSetId: string;
83
+ /** The outcome's `reviewedAt`, when present. */
84
+ readonly reviewedAt: string | undefined;
85
+ }
86
+ /** An equal-width confidence bin with its empirical accuracy. */
87
+ export interface CalibrationBin {
88
+ /** Inclusive lower bound of the bin. */
89
+ readonly lowerBound: number;
90
+ /** Exclusive upper bound (inclusive at 1.0 for the top bin). */
91
+ readonly upperBound: number;
92
+ readonly sampleCount: number;
93
+ readonly correctCount: number;
94
+ /** correctCount / sampleCount; undefined when the bin has no samples. */
95
+ readonly empiricalAccuracy: number | undefined;
96
+ /** Mean predicted confidence of samples in the bin; undefined when empty. */
97
+ readonly meanPredictedConfidence: number | undefined;
98
+ }
99
+ /** A calibration rollup for one extractor (and optionally one field). */
100
+ export interface CalibrationGroup {
101
+ readonly extractor: string;
102
+ /** The field, or undefined for an extractor-level rollup across all fields. */
103
+ readonly field: string | undefined;
104
+ readonly sampleCount: number;
105
+ readonly correctCount: number;
106
+ /** correctCount / sampleCount; undefined when the group has no samples. */
107
+ readonly empiricalAccuracy: number | undefined;
108
+ /** Mean predicted confidence across the group; undefined when no samples. */
109
+ readonly meanPredictedConfidence: number | undefined;
110
+ /**
111
+ * meanPredictedConfidence − empiricalAccuracy. Positive → overconfident
112
+ * (states more confidence than the humans bear out); negative →
113
+ * underconfident. undefined when the group has no samples.
114
+ */
115
+ readonly calibrationGap: number | undefined;
116
+ /** Per-bin empirical accuracy, ascending by lowerBound. */
117
+ readonly bins: readonly CalibrationBin[];
118
+ /**
119
+ * Lowest bin lowerBound at/above which every populated bin (≥ minBinSamples)
120
+ * meets `targetAccuracy`, scanning the top-contiguous run of qualifying bins.
121
+ * undefined when no bin qualifies — the data does not yet support an empirical
122
+ * auto-accept threshold at that target. ADVISORY: an operator wires this into
123
+ * `autoAcceptMinConfidence`; calibration never sets it.
124
+ */
125
+ readonly suggestedThreshold: number | undefined;
126
+ }
127
+ export interface CalibrationMetrics {
128
+ /** One rollup per extractor (field === undefined), sorted by extractor. */
129
+ readonly byExtractor: readonly CalibrationGroup[];
130
+ /** One rollup per (extractor, field) pair, sorted by extractor then field. */
131
+ readonly byExtractorField: readonly CalibrationGroup[];
132
+ /** A single rollup across every labeled sample (extractor "*"). */
133
+ readonly overall: CalibrationGroup;
134
+ /** Number of labeled samples used. */
135
+ readonly sampleCount: number;
136
+ /** Outcomes that could not be turned into a labeled sample (see reasons). */
137
+ readonly skippedCount: number;
138
+ /** ISO 8601 timestamp of the earliest labeled sample's `reviewedAt`. */
139
+ readonly windowStart: string | undefined;
140
+ /** ISO 8601 timestamp of the latest labeled sample's `reviewedAt`. */
141
+ readonly windowEnd: string | undefined;
142
+ /** The `windowDays` option echoed back (undefined when not windowed). */
143
+ readonly windowDays: number | undefined;
144
+ }
145
+ /**
146
+ * Derives extractor/field confidence calibration from review outcomes.
147
+ *
148
+ * Each reviewed candidate set contributes one labeled sample: the confidence of
149
+ * the SYSTEM-proposed candidate (`CandidateSet.selectedCandidateId`) as the
150
+ * prediction, and whether the human review affirmed that proposed value as the
151
+ * label. A sample is skipped when it carries no human label or no prediction:
152
+ *
153
+ * - status "proposed" (not yet reviewed);
154
+ * - a machine auto-accept, unless `includeAutoAccepted` is set;
155
+ * - no `selectedCandidateId`, or the selected candidate / its confidence is
156
+ * missing or non-finite (no prediction to calibrate).
157
+ *
158
+ * A sample is "correct" when the outcome status is verified/assumed AND the
159
+ * reviewer did not switch to a different candidate; "incorrect" when the status
160
+ * is rejected or the reviewer overrode the proposed candidate.
161
+ */
162
+ export declare function deriveCalibration(input: CalibrationInput, options?: DeriveCalibrationOptions): CalibrationMetrics;
163
+ /** A single projected calibration claim triple. */
164
+ export interface CalibrationClaim {
165
+ readonly claim: Claim;
166
+ readonly evidence: Evidence;
167
+ readonly event: VerificationEvent;
168
+ }
169
+ export interface CalibrationClaimsSubject {
170
+ /** Surface subject type (e.g. "extractor"). */
171
+ readonly subjectType: string;
172
+ /** Surface subject id prefix (e.g. a producer or run id). */
173
+ readonly subjectId: string;
174
+ /** Surface facet (e.g. "review.calibration"). */
175
+ readonly facet: string;
176
+ /** Actor id to record on events. */
177
+ readonly actor: string;
178
+ /** ISO 8601 timestamp for claim created/updated times. */
179
+ readonly observedAt: string;
180
+ /** Producer identifier for the evidence `collectedBy` field. */
181
+ readonly collectedBy: string;
182
+ }
183
+ /**
184
+ * Projects per-extractor calibration as Surface-ready claims (claimType
185
+ * "calibration"). One claim per measurable metric per extractor: empirical
186
+ * accuracy, calibration gap, and — when the data supports it — the suggested
187
+ * auto-accept threshold. Every claim is status "proposed": calibration proposes,
188
+ * it never decides (ADR 0003 §4). Groups with no labeled samples are skipped.
189
+ */
190
+ export declare function calibrationToClaims(metrics: CalibrationMetrics, subject: CalibrationClaimsSubject): CalibrationClaim[];
191
+ /**
192
+ * Merges calibration claims into an existing TrustBundle. Convenience wrapper
193
+ * mirroring `mergeTrustBundleWithOversightMetrics`.
194
+ */
195
+ export declare function mergeTrustBundleWithCalibration(bundle: TrustBundle, claims: readonly CalibrationClaim[]): TrustBundle;
@@ -0,0 +1,361 @@
1
+ /**
2
+ * Extraction-confidence calibration from human review outcomes.
3
+ *
4
+ * Survey owns the review chain, so it owns the one signal no eval vendor can
5
+ * publish: for every reviewed candidate, the stated extraction/candidate
6
+ * confidence (the PREDICTION) against the human review decision (the LABEL).
7
+ * Grouping those labeled samples by extractor (and field) and binning by
8
+ * confidence yields an empirical calibration curve — "confidence in [0.8,0.9)
9
+ * from extractor X was affirmed 17/20 times" — plus a calibration gap and an
10
+ * empirically-grounded auto-accept threshold suggestion.
11
+ *
12
+ * ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration INFORMS policy — the
13
+ * suggested threshold an operator MAY wire into `autoAcceptMinConfidence` — it
14
+ * never decides a claim or mutates a status. Projected claims carry status
15
+ * "proposed", exactly like every other producer proposal.
16
+ *
17
+ * Machine auto-accepts are EXCLUDED by default: an auto-accepted outcome is the
18
+ * threshold accepting its own guess, so counting it as a "correct" label would
19
+ * let the policy validate itself (circular). Only human review outcomes are
20
+ * labeled samples. Set `includeAutoAccepted` to override.
21
+ *
22
+ * All computations are deterministic; `now` + `windowDays` window by `reviewedAt`.
23
+ */
24
+ import { AUTO_ACCEPT_ACTOR } from "./producer-profile.js";
25
+ const DEFAULT_BIN_COUNT = 10;
26
+ const DEFAULT_TARGET_ACCURACY = 0.95;
27
+ const DEFAULT_MIN_BIN_SAMPLES = 1;
28
+ // ---------------------------------------------------------------------------
29
+ // deriveCalibration
30
+ // ---------------------------------------------------------------------------
31
+ /**
32
+ * Derives extractor/field confidence calibration from review outcomes.
33
+ *
34
+ * Each reviewed candidate set contributes one labeled sample: the confidence of
35
+ * the SYSTEM-proposed candidate (`CandidateSet.selectedCandidateId`) as the
36
+ * prediction, and whether the human review affirmed that proposed value as the
37
+ * label. A sample is skipped when it carries no human label or no prediction:
38
+ *
39
+ * - status "proposed" (not yet reviewed);
40
+ * - a machine auto-accept, unless `includeAutoAccepted` is set;
41
+ * - no `selectedCandidateId`, or the selected candidate / its confidence is
42
+ * missing or non-finite (no prediction to calibrate).
43
+ *
44
+ * A sample is "correct" when the outcome status is verified/assumed AND the
45
+ * reviewer did not switch to a different candidate; "incorrect" when the status
46
+ * is rejected or the reviewer overrode the proposed candidate.
47
+ */
48
+ export function deriveCalibration(input, options = {}) {
49
+ const binCount = normalizeBinCount(options.binCount);
50
+ const targetAccuracy = options.targetAccuracy ?? DEFAULT_TARGET_ACCURACY;
51
+ const minBinSamples = options.minBinSamples ?? DEFAULT_MIN_BIN_SAMPLES;
52
+ const includeAutoAccepted = options.includeAutoAccepted ?? false;
53
+ const candidateSetById = new Map(input.candidateSets.map((cs) => [cs.id, cs]));
54
+ const candidateById = new Map();
55
+ for (const cs of input.candidateSets) {
56
+ for (const c of cs.candidates)
57
+ candidateById.set(c.id, c);
58
+ }
59
+ const extractionById = new Map(input.extractions.map((e) => [e.id, e]));
60
+ if (options.windowDays !== undefined && options.now === undefined) {
61
+ throw new RangeError("deriveCalibration: `now` is required when `windowDays` is set (windowing is by reviewedAt).");
62
+ }
63
+ const cutoff = options.windowDays !== undefined && options.now !== undefined
64
+ ? options.now.getTime() - options.windowDays * 24 * 60 * 60 * 1000
65
+ : undefined;
66
+ const samples = [];
67
+ let skippedCount = 0;
68
+ for (const outcome of input.reviewOutcomes) {
69
+ if (cutoff !== undefined) {
70
+ const t = outcome.reviewedAt ? Date.parse(outcome.reviewedAt) : NaN;
71
+ if (isNaN(t) || t < cutoff) {
72
+ skippedCount++;
73
+ continue;
74
+ }
75
+ }
76
+ const sample = toSample(outcome, candidateSetById, candidateById, extractionById, includeAutoAccepted);
77
+ if (sample === undefined) {
78
+ skippedCount++;
79
+ continue;
80
+ }
81
+ samples.push(sample);
82
+ }
83
+ const timestamps = samples
84
+ .map((s) => (s.reviewedAt ? Date.parse(s.reviewedAt) : NaN))
85
+ .filter((t) => !isNaN(t))
86
+ .sort((a, b) => a - b);
87
+ const windowStart = timestamps.length > 0 ? new Date(timestamps[0]).toISOString() : undefined;
88
+ const windowEnd = timestamps.length > 0 ? new Date(timestamps[timestamps.length - 1]).toISOString() : undefined;
89
+ const buildGroup = (extractor, field, groupSamples) => computeGroup(extractor, field, groupSamples, binCount, targetAccuracy, minBinSamples);
90
+ // Extractor-level rollups.
91
+ const byExtractorMap = new Map();
92
+ // (extractor, field) rollups. The key is a JSON-encoded [extractor, field]
93
+ // pair so no in-band delimiter can collide with an extractor/field that
94
+ // contains that delimiter; the extractor and field are carried in the value,
95
+ // never parsed back out of the key.
96
+ const byFieldMap = new Map();
97
+ for (const s of samples) {
98
+ pushTo(byExtractorMap, s.extractor, s);
99
+ const fieldKey = JSON.stringify([s.extractor, s.field]);
100
+ const existing = byFieldMap.get(fieldKey);
101
+ if (existing)
102
+ existing.samples.push(s);
103
+ else
104
+ byFieldMap.set(fieldKey, { extractor: s.extractor, field: s.field, samples: [s] });
105
+ }
106
+ const byExtractor = [...byExtractorMap.entries()]
107
+ .sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0))
108
+ .map(([extractor, groupSamples]) => buildGroup(extractor, undefined, groupSamples));
109
+ const byExtractorField = [...byFieldMap.values()]
110
+ .sort((a, b) => (a.extractor < b.extractor ? -1 : a.extractor > b.extractor ? 1 : a.field < b.field ? -1 : a.field > b.field ? 1 : 0))
111
+ .map(({ extractor, field, samples: groupSamples }) => buildGroup(extractor, field, groupSamples));
112
+ const overall = buildGroup("*", undefined, samples);
113
+ return {
114
+ byExtractor,
115
+ byExtractorField,
116
+ overall,
117
+ sampleCount: samples.length,
118
+ skippedCount,
119
+ windowStart,
120
+ windowEnd,
121
+ windowDays: options.windowDays,
122
+ };
123
+ }
124
+ // ---------------------------------------------------------------------------
125
+ // Internal: sample construction
126
+ // ---------------------------------------------------------------------------
127
+ function toSample(outcome, candidateSetById, candidateById, extractionById, includeAutoAccepted) {
128
+ // No human label yet.
129
+ if (outcome.status === "proposed")
130
+ return undefined;
131
+ // Machine auto-accepts are not human labels (circular) unless opted in.
132
+ if (!includeAutoAccepted && outcome.actor === AUTO_ACCEPT_ACTOR)
133
+ return undefined;
134
+ const candidateSet = candidateSetById.get(outcome.candidateSetId);
135
+ if (candidateSet === undefined)
136
+ return undefined;
137
+ // The prediction is the SYSTEM-proposed candidate's confidence.
138
+ const proposedId = candidateSet.selectedCandidateId;
139
+ if (proposedId === undefined)
140
+ return undefined;
141
+ const proposed = candidateById.get(proposedId);
142
+ if (proposed === undefined)
143
+ return undefined;
144
+ const extraction = proposed.extractionId ? extractionById.get(proposed.extractionId) : undefined;
145
+ const rawConfidence = proposed.confidence ?? extraction?.confidence;
146
+ if (rawConfidence === undefined || !Number.isFinite(rawConfidence))
147
+ return undefined;
148
+ const predictedConfidence = clamp01(rawConfidence);
149
+ const extractor = extraction?.extractor ?? "unknown";
150
+ const field = extraction?.target ?? candidateSet.target;
151
+ // The label: did the human affirm the proposed value?
152
+ let correct;
153
+ if (outcome.status === "rejected") {
154
+ correct = false;
155
+ }
156
+ else {
157
+ // verified | assumed — an override to a different candidate means the
158
+ // proposed value did NOT stand.
159
+ const overrode = outcome.candidateId !== undefined && outcome.candidateId !== proposedId;
160
+ correct = !overrode;
161
+ }
162
+ return {
163
+ extractor,
164
+ field,
165
+ predictedConfidence,
166
+ correct,
167
+ reviewOutcomeId: outcome.id,
168
+ candidateSetId: outcome.candidateSetId,
169
+ reviewedAt: outcome.reviewedAt,
170
+ };
171
+ }
172
+ // ---------------------------------------------------------------------------
173
+ // Internal: group + bin computation
174
+ // ---------------------------------------------------------------------------
175
+ function computeGroup(extractor, field, samples, binCount, targetAccuracy, minBinSamples) {
176
+ const sampleCount = samples.length;
177
+ const correctCount = samples.filter((s) => s.correct).length;
178
+ const empiricalAccuracy = sampleCount > 0 ? correctCount / sampleCount : undefined;
179
+ const meanPredictedConfidence = sampleCount > 0
180
+ ? samples.reduce((sum, s) => sum + s.predictedConfidence, 0) / sampleCount
181
+ : undefined;
182
+ const calibrationGap = meanPredictedConfidence !== undefined && empiricalAccuracy !== undefined
183
+ ? meanPredictedConfidence - empiricalAccuracy
184
+ : undefined;
185
+ const bins = computeBins(samples, binCount);
186
+ const suggestedThreshold = computeSuggestedThreshold(bins, targetAccuracy, minBinSamples);
187
+ return {
188
+ extractor,
189
+ field,
190
+ sampleCount,
191
+ correctCount,
192
+ empiricalAccuracy: round(empiricalAccuracy),
193
+ meanPredictedConfidence: round(meanPredictedConfidence),
194
+ calibrationGap: round(calibrationGap),
195
+ bins,
196
+ suggestedThreshold,
197
+ };
198
+ }
199
+ function computeBins(samples, binCount) {
200
+ const width = 1 / binCount;
201
+ const counts = Array.from({ length: binCount }, () => ({ n: 0, correct: 0, sum: 0 }));
202
+ for (const s of samples) {
203
+ const idx = Math.min(binCount - 1, Math.floor(s.predictedConfidence * binCount));
204
+ const bucket = counts[idx];
205
+ bucket.n++;
206
+ bucket.sum += s.predictedConfidence;
207
+ if (s.correct)
208
+ bucket.correct++;
209
+ }
210
+ return counts.map((b, i) => ({
211
+ lowerBound: round(i * width),
212
+ upperBound: round((i + 1) * width),
213
+ sampleCount: b.n,
214
+ correctCount: b.correct,
215
+ empiricalAccuracy: b.n > 0 ? round(b.correct / b.n) : undefined,
216
+ meanPredictedConfidence: b.n > 0 ? round(b.sum / b.n) : undefined,
217
+ }));
218
+ }
219
+ /**
220
+ * The suggested threshold is the lowerBound of the lowest bin in the
221
+ * top-contiguous run of bins that each (a) have ≥ minBinSamples and (b) meet
222
+ * targetAccuracy. Scanning from the highest bin down, a populated bin that
223
+ * fails the target — or an under-sampled bin we cannot vouch for — ends the run.
224
+ * undefined when even the top populated bin does not qualify.
225
+ */
226
+ function computeSuggestedThreshold(bins, targetAccuracy, minBinSamples) {
227
+ let threshold;
228
+ for (let i = bins.length - 1; i >= 0; i--) {
229
+ const bin = bins[i];
230
+ if (bin.sampleCount < minBinSamples)
231
+ break;
232
+ if (bin.empiricalAccuracy === undefined || bin.empiricalAccuracy < targetAccuracy)
233
+ break;
234
+ threshold = bin.lowerBound;
235
+ }
236
+ return threshold;
237
+ }
238
+ /**
239
+ * Projects per-extractor calibration as Surface-ready claims (claimType
240
+ * "calibration"). One claim per measurable metric per extractor: empirical
241
+ * accuracy, calibration gap, and — when the data supports it — the suggested
242
+ * auto-accept threshold. Every claim is status "proposed": calibration proposes,
243
+ * it never decides (ADR 0003 §4). Groups with no labeled samples are skipped.
244
+ */
245
+ export function calibrationToClaims(metrics, subject) {
246
+ const results = [];
247
+ for (const group of metrics.byExtractor) {
248
+ if (group.sampleCount === 0)
249
+ continue;
250
+ const base = `${sanitize(group.extractor)}`;
251
+ const push = (metric, value, detail) => {
252
+ const claimId = `calibration.${subject.subjectId}.${base}.${metric}`;
253
+ const evidenceId = `${claimId}.evidence`;
254
+ const claim = {
255
+ id: claimId,
256
+ subjectType: subject.subjectType,
257
+ subjectId: `${subject.subjectId}.${group.extractor}`,
258
+ facet: subject.facet,
259
+ claimType: "calibration",
260
+ fieldOrBehavior: metric,
261
+ value,
262
+ status: "proposed",
263
+ createdAt: subject.observedAt,
264
+ updatedAt: subject.observedAt,
265
+ impactLevel: "medium",
266
+ confidenceBasis: {
267
+ sourceQuality: "moderate",
268
+ reviewerAuthority: "none",
269
+ evidenceStrength: "weak",
270
+ impactLevel: "medium",
271
+ },
272
+ metadata: {
273
+ calibration: {
274
+ extractor: group.extractor,
275
+ sampleCount: group.sampleCount,
276
+ correctCount: group.correctCount,
277
+ empiricalAccuracy: group.empiricalAccuracy,
278
+ meanPredictedConfidence: group.meanPredictedConfidence,
279
+ calibrationGap: group.calibrationGap,
280
+ suggestedThreshold: group.suggestedThreshold,
281
+ windowStart: metrics.windowStart,
282
+ windowEnd: metrics.windowEnd,
283
+ windowDays: metrics.windowDays,
284
+ },
285
+ },
286
+ };
287
+ const evidence = {
288
+ id: evidenceId,
289
+ claimId,
290
+ evidenceType: "attestation",
291
+ method: "extraction",
292
+ sourceRef: "calibration://computed",
293
+ excerptOrSummary: `calibration.${metric} for extractor ${group.extractor}: ${detail}; ` +
294
+ `over ${group.correctCount}/${group.sampleCount} affirmed human review outcomes`,
295
+ observedAt: subject.observedAt,
296
+ collectedBy: subject.collectedBy,
297
+ };
298
+ const event = {
299
+ id: `${claimId}.event`,
300
+ claimId,
301
+ status: "proposed",
302
+ actor: subject.actor,
303
+ method: "candidate-proposal",
304
+ evidenceIds: [evidenceId],
305
+ createdAt: subject.observedAt,
306
+ };
307
+ results.push({ claim, evidence, event });
308
+ };
309
+ if (group.empiricalAccuracy !== undefined) {
310
+ push("empiricalAccuracy", group.empiricalAccuracy, `empirical accuracy = ${group.empiricalAccuracy}`);
311
+ }
312
+ if (group.calibrationGap !== undefined) {
313
+ push("calibrationGap", group.calibrationGap, `mean confidence − empirical accuracy = ${group.calibrationGap}`);
314
+ }
315
+ if (group.suggestedThreshold !== undefined) {
316
+ push("suggestedThreshold", group.suggestedThreshold, `advisory auto-accept threshold = ${group.suggestedThreshold}`);
317
+ }
318
+ }
319
+ return results;
320
+ }
321
+ /**
322
+ * Merges calibration claims into an existing TrustBundle. Convenience wrapper
323
+ * mirroring `mergeTrustBundleWithOversightMetrics`.
324
+ */
325
+ export function mergeTrustBundleWithCalibration(bundle, claims) {
326
+ return {
327
+ ...bundle,
328
+ claims: [...bundle.claims, ...claims.map((c) => c.claim)],
329
+ evidence: [...bundle.evidence, ...claims.map((c) => c.evidence)],
330
+ events: [...bundle.events, ...claims.map((c) => c.event)],
331
+ };
332
+ }
333
+ // ---------------------------------------------------------------------------
334
+ // Internal helpers
335
+ // ---------------------------------------------------------------------------
336
+ function pushTo(map, key, sample) {
337
+ const existing = map.get(key);
338
+ if (existing)
339
+ existing.push(sample);
340
+ else
341
+ map.set(key, [sample]);
342
+ }
343
+ function clamp01(value) {
344
+ return value < 0 ? 0 : value > 1 ? 1 : value;
345
+ }
346
+ function normalizeBinCount(binCount) {
347
+ if (binCount === undefined)
348
+ return DEFAULT_BIN_COUNT;
349
+ if (!Number.isInteger(binCount) || binCount < 1) {
350
+ throw new RangeError(`binCount must be a positive integer, received ${binCount}`);
351
+ }
352
+ return binCount;
353
+ }
354
+ function round(value) {
355
+ if (value === undefined)
356
+ return undefined;
357
+ return Math.round(value * 10000) / 10000;
358
+ }
359
+ function sanitize(id) {
360
+ return id.replace(/[^A-Za-z0-9._-]/g, "_");
361
+ }
@@ -12,6 +12,8 @@ export { buildSurveyTrustBundle } from "./to-surface.js";
12
12
  export type { BuildSurveyTrustBundleOptions } from "./to-surface.js";
13
13
  export { buildSurveyLearningProjections } from "./learning-projections.js";
14
14
  export type { LearningProjection, LearningProjectionKind, LearningProjectionSeverity, LearningProjectionSignal, } from "./learning-projections.js";
15
+ export { buildReviewedLearningUpdateProposal } from "./learning-update-proposal.js";
16
+ export type { LearningUpdateEvidenceReference, LearningUpdateProposal, OpaqueEvidenceReference, ProvenanceReference, ReviewedLearningUpdateProposalInput, ReviewProofReference, } from "./learning-update-proposal.js";
15
17
  export { buildCanonicalReviewProofPayload, buildReviewProofAnchor, canonicalReviewProofJson, hashCanonicalReviewProofPayload, verifyCanonicalReviewProofPayload, REVIEW_PROOF_CONTRACT_VERSION, REVIEW_PROOF_PACKAGE_NAME, REVIEW_PROOF_SCHEMA, REVIEW_PROOF_SCHEMA_VERSION, } from "./review-proof.js";
16
18
  export type { CanonicalReviewProofPayload, CanonicalReviewProofPayloadV1, CanonicalReviewProofPayloadV2, ReviewProofInput, } from "./review-proof.js";
17
19
  export { fieldObservation } from "./field-observation.js";
@@ -38,3 +40,5 @@ export { currentProposedReviewItem } from "./current-proposed-review-item.js";
38
40
  export type { CurrentProposedCandidateInput, CurrentProposedReviewItemInput, } from "./current-proposed-review-item.js";
39
41
  export { deriveOversightMetrics, mergeTrustBundleWithOversightMetrics, oversightMetricsToClaims, } from "./oversight-metrics.js";
40
42
  export type { AggregateOversightMetrics, DeriveOversightMetricsOptions, OversightMetrics, OversightMetricsClaimsSubject, OversightQualityClaim, ReviewerOversightMetrics, } from "./oversight-metrics.js";
43
+ export { calibrationToClaims, deriveCalibration, mergeTrustBundleWithCalibration, } from "./calibration.js";
44
+ export type { CalibrationBin, CalibrationClaim, CalibrationClaimsSubject, CalibrationGroup, CalibrationInput, CalibrationMetrics, CalibrationSample, DeriveCalibrationOptions, } from "./calibration.js";
package/dist/src/index.js CHANGED
@@ -5,6 +5,7 @@ export { reviewedCandidateResolution } from "./reviewed-candidate-resolution.js"
5
5
  export { reviewedCurrentProposedResolution } from "./reviewed-current-proposed-resolution.js";
6
6
  export { buildSurveyTrustBundle } from "./to-surface.js";
7
7
  export { buildSurveyLearningProjections } from "./learning-projections.js";
8
+ export { buildReviewedLearningUpdateProposal } from "./learning-update-proposal.js";
8
9
  export { buildCanonicalReviewProofPayload, buildReviewProofAnchor, canonicalReviewProofJson, hashCanonicalReviewProofPayload, verifyCanonicalReviewProofPayload, REVIEW_PROOF_CONTRACT_VERSION, REVIEW_PROOF_PACKAGE_NAME, REVIEW_PROOF_SCHEMA, REVIEW_PROOF_SCHEMA_VERSION, } from "./review-proof.js";
9
10
  export { fieldObservation } from "./field-observation.js";
10
11
  export { repeatedObservation } from "./repeated-observation.js";
@@ -18,3 +19,4 @@ export { buildAuthorizedActionAuthorizing, buildPromptRef, isValidAuthorizing, v
18
19
  export { confidenceBasisForReview, defineProductVocabulary, stableId } from "./vocabulary.js";
19
20
  export { currentProposedReviewItem } from "./current-proposed-review-item.js";
20
21
  export { deriveOversightMetrics, mergeTrustBundleWithOversightMetrics, oversightMetricsToClaims, } from "./oversight-metrics.js";
22
+ export { calibrationToClaims, deriveCalibration, mergeTrustBundleWithCalibration, } from "./calibration.js";
@@ -0,0 +1,54 @@
1
+ import type { ClaimTarget, ProvenanceResolution, RawSourceKind, SurveyInput } from "./types.js";
2
+ export interface ReviewProofReference {
3
+ kind: "review-proof";
4
+ algorithm: "sha256";
5
+ value: string;
6
+ proofSchemaVersion: 2;
7
+ }
8
+ export interface ProvenanceReference {
9
+ kind: "provenance";
10
+ rawSourceId: string;
11
+ origin: RawSourceKind;
12
+ resolution: ProvenanceResolution;
13
+ reviewOutcomeId: string;
14
+ }
15
+ export interface OpaqueEvidenceReference {
16
+ kind: "evidence";
17
+ id: string;
18
+ }
19
+ export type LearningUpdateEvidenceReference = ReviewProofReference | ProvenanceReference | OpaqueEvidenceReference;
20
+ export interface LearningUpdateProposal {
21
+ id: string;
22
+ kind: "learning.update-proposal";
23
+ source: string;
24
+ createdAt: string;
25
+ subject: Pick<ClaimTarget, "subjectType" | "subjectId" | "facet" | "claimType" | "fieldOrBehavior">;
26
+ applicability: {
27
+ target: string;
28
+ };
29
+ proposedDelta: {
30
+ previousValue: unknown;
31
+ proposedValue: unknown;
32
+ };
33
+ evidenceRefs: LearningUpdateEvidenceReference[];
34
+ authorizationRef: {
35
+ reviewOutcomeId: string;
36
+ reviewProofHash: string;
37
+ };
38
+ reviewLineage: {
39
+ candidateSetId: string;
40
+ selectedCandidateId: string;
41
+ unselectedCandidateIds: string[];
42
+ reviewOutcomeId: string;
43
+ selectedClaimId: string;
44
+ };
45
+ }
46
+ export interface ReviewedLearningUpdateProposalInput {
47
+ survey: SurveyInput;
48
+ candidateSetId: string;
49
+ reviewOutcomeId: string;
50
+ selectedClaimId: string;
51
+ proof: ReviewProofReference;
52
+ }
53
+ /** Builds data for a producer-owned application decision. It performs no I/O or domain validation. */
54
+ export declare function buildReviewedLearningUpdateProposal(input: ReviewedLearningUpdateProposalInput): LearningUpdateProposal;
@@ -0,0 +1,180 @@
1
+ import { createHash } from "node:crypto";
2
+ import { canonicalJson } from "./review-workbench/canonical.js";
3
+ /** Builds data for a producer-owned application decision. It performs no I/O or domain validation. */
4
+ export function buildReviewedLearningUpdateProposal(input) {
5
+ const { survey } = input;
6
+ const candidateSet = exactlyOne(survey.candidateSets, ({ id }) => id === input.candidateSetId, "candidate set");
7
+ if (!candidateSet || candidateSet.status !== "resolved")
8
+ throw new Error("learning update requires a resolved candidate set");
9
+ const roleCandidates = candidateSet.candidates.filter((candidate) => {
10
+ const role = candidate.metadata?.candidateRole;
11
+ return role === "current" || role === "proposed";
12
+ });
13
+ const current = roleCandidates.filter(({ metadata }) => metadata?.candidateRole === "current");
14
+ const proposed = roleCandidates.filter(({ metadata }) => metadata?.candidateRole === "proposed");
15
+ assertUniqueIds(candidateSet.candidates, "candidate");
16
+ if (current.length !== 1 || proposed.length !== 1) {
17
+ throw new Error("learning update requires exactly one current and one proposed candidate role");
18
+ }
19
+ const currentCandidate = current[0];
20
+ const proposedCandidate = proposed[0];
21
+ if (candidateSet.selectedCandidateId !== proposedCandidate.id)
22
+ throw new Error("learning update requires the proposed candidate to be selected");
23
+ const review = exactlyOne(survey.reviewOutcomes, ({ id }) => id === input.reviewOutcomeId, "review outcome");
24
+ if (!review || review.candidateSetId !== candidateSet.id || review.candidateId !== candidateSet.selectedCandidateId) {
25
+ throw new Error("review outcome must identify the selected candidate and candidate set");
26
+ }
27
+ if (review.status !== "verified" && review.status !== "assumed")
28
+ throw new Error("learning update requires an accepted review status");
29
+ if (!review.reviewedAt)
30
+ throw new Error("learning update requires reviewedAt");
31
+ if (!review.authorizing)
32
+ throw new Error("learning update requires authorizing review provenance");
33
+ assertCanonicalProofReference(input.proof);
34
+ const selectedClaim = exactlyOne(survey.claims, ({ id }) => id === input.selectedClaimId, "selected claim");
35
+ if (!selectedClaim || selectedClaim.candidateSetId !== candidateSet.id || selectedClaim.candidateId !== proposedCandidate.id) {
36
+ throw new Error("selected claim must identify the selected proposed candidate and candidate set");
37
+ }
38
+ if (selectedClaim.status !== "verified" && selectedClaim.status !== "assumed") {
39
+ throw new Error("learning update requires an accepted selected claim status");
40
+ }
41
+ if (selectedClaim.status !== review.status)
42
+ throw new Error("selected claim status must match the accepted review status");
43
+ const currentClaim = exactlyOne(survey.claims, ({ candidateSetId, candidateId }) => candidateSetId === candidateSet.id && candidateId === currentCandidate.id, "current claim");
44
+ if (!currentClaim || !sameSubject(currentClaim, selectedClaim))
45
+ throw new Error("current and selected claim lineage must share a subject");
46
+ const unselectedCandidates = candidateSet.candidates.filter(({ id }) => id !== proposedCandidate.id);
47
+ for (const candidate of unselectedCandidates) {
48
+ const claim = exactlyOne(survey.claims, ({ candidateSetId, candidateId }) => candidateSetId === candidateSet.id && candidateId === candidate.id, "unselected claim");
49
+ if (claim.status !== "superseded")
50
+ throw new Error("unselected claim status must be superseded");
51
+ }
52
+ assertCanonicalJsonValue(currentCandidate.value);
53
+ assertCanonicalJsonValue(proposedCandidate.value);
54
+ const currentSource = sourceForCandidate(survey, candidateSet.target, currentCandidate.id, currentCandidate.extractionId, currentCandidate.value);
55
+ const proposedSource = sourceForCandidate(survey, candidateSet.target, proposedCandidate.id, proposedCandidate.extractionId, proposedCandidate.value);
56
+ const provenanceRefs = [currentSource, proposedSource].map((source) => {
57
+ if (source.resolution !== "supersession")
58
+ throw new Error(`candidate ${source.id} requires explicit supersession provenance`);
59
+ return { kind: "provenance", rawSourceId: source.id, origin: source.kind, resolution: source.resolution, reviewOutcomeId: review.id };
60
+ });
61
+ const evidenceCandidates = [
62
+ ...(review.evidenceIds ?? []).map((id) => ({ kind: "evidence", id })),
63
+ ...provenanceRefs,
64
+ { ...input.proof },
65
+ ];
66
+ const evidenceRefs = [...new Map(evidenceCandidates.map((reference) => [canonicalJson(reference), reference])).values()]
67
+ .sort((left, right) => evidenceRank(left) - evidenceRank(right) || canonicalJson(left).localeCompare(canonicalJson(right)));
68
+ const subject = pickSubject(selectedClaim);
69
+ const reviewLineage = {
70
+ candidateSetId: candidateSet.id,
71
+ selectedCandidateId: proposedCandidate.id,
72
+ unselectedCandidateIds: unselectedCandidates.map(({ id }) => id).sort(),
73
+ reviewOutcomeId: review.id,
74
+ selectedClaimId: selectedClaim.id,
75
+ };
76
+ const authorizationRef = { reviewOutcomeId: review.id, reviewProofHash: input.proof.value };
77
+ const proposalWithoutId = {
78
+ kind: "learning.update-proposal",
79
+ source: survey.source,
80
+ createdAt: review.reviewedAt,
81
+ subject,
82
+ applicability: { target: candidateSet.target },
83
+ proposedDelta: { previousValue: currentCandidate.value, proposedValue: proposedCandidate.value },
84
+ evidenceRefs,
85
+ authorizationRef,
86
+ reviewLineage,
87
+ };
88
+ const identityPayload = { identitySchemaVersion: 1, ...proposalWithoutId };
89
+ return { id: createHash("sha256").update(canonicalJson(identityPayload)).digest("hex"), ...proposalWithoutId };
90
+ }
91
+ function assertCanonicalProofReference(proof) {
92
+ if (proof?.kind !== "review-proof" || proof.algorithm !== "sha256" || proof.proofSchemaVersion !== 2) {
93
+ throw new Error("learning update requires an identified canonical v2 review proof");
94
+ }
95
+ if (!/^[a-f0-9]{64}$/.test(proof.value))
96
+ throw new Error("canonical review proof must be a lowercase SHA-256 hash");
97
+ }
98
+ function sourceForCandidate(survey, target, candidateId, extractionId, value) {
99
+ const extraction = exactlyOne(survey.extractions, ({ id }) => id === extractionId, "extraction");
100
+ assertCanonicalJsonValue(extraction.value);
101
+ if (!extraction || extraction.target !== target || canonicalJson(extraction.value) !== canonicalJson(value)) {
102
+ throw new Error(`candidate ${candidateId} has inconsistent extraction lineage`);
103
+ }
104
+ const source = exactlyOne(survey.rawSources, ({ id }) => id === extraction.sourceId, "raw source");
105
+ if (!source)
106
+ throw new Error(`candidate ${candidateId} has missing source lineage`);
107
+ return source;
108
+ }
109
+ function pickSubject(claim) {
110
+ return { subjectType: claim.subjectType, subjectId: claim.subjectId, facet: claim.facet, claimType: claim.claimType, fieldOrBehavior: claim.fieldOrBehavior };
111
+ }
112
+ function sameSubject(left, right) {
113
+ return canonicalJson(pickSubject(left)) === canonicalJson(pickSubject(right));
114
+ }
115
+ function evidenceRank(reference) {
116
+ if (reference.kind === "evidence")
117
+ return 0;
118
+ if (reference.kind === "provenance")
119
+ return 1;
120
+ return 2;
121
+ }
122
+ function exactlyOne(records, predicate, label) {
123
+ const matches = records.filter(predicate);
124
+ if (matches.length !== 1)
125
+ throw new Error(`learning update requires exactly one ${label}; found ${matches.length}`);
126
+ return matches[0];
127
+ }
128
+ function assertUniqueIds(records, label) {
129
+ const ids = new Set();
130
+ for (const { id } of records) {
131
+ if (ids.has(id))
132
+ throw new Error(`learning update rejects duplicate ${label} id ${id}`);
133
+ ids.add(id);
134
+ }
135
+ }
136
+ /** Values are limited to JSON primitives, arrays, and plain string-keyed objects. */
137
+ function assertCanonicalJsonValue(value, ancestors = new Set()) {
138
+ if (value === null || typeof value === "string" || typeof value === "boolean")
139
+ return;
140
+ if (typeof value === "number") {
141
+ if (Number.isFinite(value) && !Object.is(value, -0))
142
+ return;
143
+ throw new Error("invalid canonical JSON value: number is not collision-free");
144
+ }
145
+ if (typeof value !== "object")
146
+ throw new Error("invalid canonical JSON value: unsupported primitive");
147
+ if (ancestors.has(value))
148
+ throw new Error("invalid canonical JSON value: cycle");
149
+ const prototype = Object.getPrototypeOf(value);
150
+ if (!Array.isArray(value) && prototype !== Object.prototype && prototype !== null) {
151
+ throw new Error("invalid canonical JSON value: unsupported prototype");
152
+ }
153
+ ancestors.add(value);
154
+ if (Array.isArray(value)) {
155
+ if (Object.getOwnPropertySymbols(value).length > 0)
156
+ throw new Error("invalid canonical JSON value: symbol-keyed array property");
157
+ const descriptors = Object.getOwnPropertyDescriptors(value);
158
+ const expectedKeys = Array.from({ length: value.length }, (_, index) => String(index));
159
+ const actualKeys = Object.keys(descriptors).filter((key) => key !== "length");
160
+ if (actualKeys.length !== expectedKeys.length || actualKeys.some((key, index) => key !== expectedKeys[index])) {
161
+ throw new Error("invalid canonical JSON value: sparse or extra array property");
162
+ }
163
+ for (const key of expectedKeys) {
164
+ const descriptor = descriptors[key];
165
+ if (!descriptor.enumerable || !("value" in descriptor))
166
+ throw new Error("invalid canonical JSON value: array accessor or hidden property");
167
+ assertCanonicalJsonValue(descriptor.value, ancestors);
168
+ }
169
+ }
170
+ else {
171
+ if (Object.getOwnPropertySymbols(value).length > 0)
172
+ throw new Error("invalid canonical JSON value: symbol-keyed property");
173
+ for (const descriptor of Object.values(Object.getOwnPropertyDescriptors(value))) {
174
+ if (!descriptor.enumerable || !("value" in descriptor))
175
+ throw new Error("invalid canonical JSON value: property is not enumerable data");
176
+ assertCanonicalJsonValue(descriptor.value, ancestors);
177
+ }
178
+ }
179
+ ancestors.delete(value);
180
+ }
@@ -47,6 +47,20 @@ export function buildSurveyTrustBundle(input, options = {}) {
47
47
  survey: buildSurveyMetadata({ projection, rawSource, extraction, candidateSet, candidate, review }),
48
48
  },
49
49
  };
50
+ // Promote the review's comfort-zone signal into the first-class
51
+ // conclusionConfidence field (Surface 2.9 / Hachure 0.14) so the calibration
52
+ // signal is portable and comparable, not buried in producer metadata.
53
+ // We carry only comfortZone: `value` is a *calibrated conclusion probability*
54
+ // that Survey does not yet produce (extraction confidence is an ingredient in
55
+ // confidenceBasis, not a calibrated conclusion value) — carry, not produce.
56
+ if (review?.withinComfortZone !== undefined) {
57
+ claim.conclusionConfidence = {
58
+ comfortZone: {
59
+ within: review.withinComfortZone,
60
+ ...(review.comfortZoneNote ? { reason: review.comfortZoneNote } : {}),
61
+ },
62
+ };
63
+ }
50
64
  if (options.reviewProofs && review) {
51
65
  claim.currentIntegrityAnchor = buildReviewProofAnchor({
52
66
  rawSource,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@kontourai/survey",
3
- "version": "1.7.0",
3
+ "version": "1.9.0",
4
4
  "description": "Producer-side source, extraction, candidate, and review contracts for projecting verified claims into Surface.",
5
5
  "license": "Apache-2.0",
6
6
  "type": "module",
@@ -71,7 +71,7 @@
71
71
  "check:generated-css": "node scripts/copy-review-workbench-package-assets.cjs --check"
72
72
  },
73
73
  "dependencies": {
74
- "@kontourai/surface": "^2.0.0"
74
+ "@kontourai/surface": "^2.9.0"
75
75
  },
76
76
  "peerDependencies": {
77
77
  "@anthropic-ai/sdk": ">=0.20.0",