@kontourai/survey 1.9.0 → 1.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,29 @@
1
+ /**
2
+ * Worked example: wiring confidence calibration into a downstream consumer.
3
+ *
4
+ * A consumer that owns human review outcomes can close the confidence loop in two
5
+ * places, both shipped in @kontourai/survey (1.10.0):
6
+ *
7
+ * 1. Ground the auto-accept threshold. `deriveCalibration` over a history of
8
+ * review outcomes yields `suggestedThreshold` — the lowest confidence at
9
+ * which the extractor's proposals were empirically affirmed often enough.
10
+ * Feed that number into a producer profile's `autoAcceptMinConfidence`
11
+ * (see `SchemaMappingOptions.autoAcceptMinConfidence` /
12
+ * `InquiryMappingOptions.minConfidence`) instead of hand-picking it.
13
+ *
14
+ * 2. Produce calibrated conclusion confidence. Pass the same calibration
15
+ * `metrics` into `buildSurveyTrustBundle({ calibration: { metrics } })` and
16
+ * affirmed claims carry `conclusionConfidence.value` = the empirical
17
+ * affirmation rate for their extractor/field.
18
+ *
19
+ * Calibration is advisory (ADR 0003 §4): the threshold is a suggestion the
20
+ * operator wires into policy, and the produced value never changes claim status.
21
+ *
22
+ * Run: `node dist/examples/calibrated-auto-accept.js`
23
+ */
24
+ export interface CalibratedAutoAcceptResult {
25
+ readonly suggestedThreshold: number | undefined;
26
+ readonly groupAccuracy: number | undefined;
27
+ readonly producedValues: ReadonlyArray<number | undefined>;
28
+ }
29
+ export declare function runCalibratedAutoAccept(): CalibratedAutoAcceptResult;
@@ -0,0 +1,135 @@
1
+ /**
2
+ * Worked example: wiring confidence calibration into a downstream consumer.
3
+ *
4
+ * A consumer that owns human review outcomes can close the confidence loop in two
5
+ * places, both shipped in @kontourai/survey (1.10.0):
6
+ *
7
+ * 1. Ground the auto-accept threshold. `deriveCalibration` over a history of
8
+ * review outcomes yields `suggestedThreshold` — the lowest confidence at
9
+ * which the extractor's proposals were empirically affirmed often enough.
10
+ * Feed that number into a producer profile's `autoAcceptMinConfidence`
11
+ * (see `SchemaMappingOptions.autoAcceptMinConfidence` /
12
+ * `InquiryMappingOptions.minConfidence`) instead of hand-picking it.
13
+ *
14
+ * 2. Produce calibrated conclusion confidence. Pass the same calibration
15
+ * `metrics` into `buildSurveyTrustBundle({ calibration: { metrics } })` and
16
+ * affirmed claims carry `conclusionConfidence.value` = the empirical
17
+ * affirmation rate for their extractor/field.
18
+ *
19
+ * Calibration is advisory (ADR 0003 §4): the threshold is a suggestion the
20
+ * operator wires into policy, and the produced value never changes claim status.
21
+ *
22
+ * Run: `node dist/examples/calibrated-auto-accept.js`
23
+ */
24
+ import { buildSurveyTrustBundle, deriveCalibration, SurveyInputBuilder, } from "../src/index.js";
25
+ const EXTRACTOR = "example-extractor";
26
+ const FIELD = "registrationStatus";
27
+ const HISTORY_AT = "2026-06-01T00:00:00.000Z";
28
+ const BATCH_AT = "2026-07-01T00:00:00.000Z";
29
+ /**
30
+ * A synthetic history: for this extractor/field, high-confidence proposals were
31
+ * almost always affirmed by reviewers and low-confidence ones were mostly
32
+ * rejected — the pattern that makes an empirical threshold meaningful.
33
+ */
34
+ function buildReviewHistory() {
35
+ const extractions = [];
36
+ const candidateSets = [];
37
+ const reviewOutcomes = [];
38
+ let n = 0;
39
+ const addSamples = (count, confidence, affirmed) => {
40
+ for (let i = 0; i < count; i += 1) {
41
+ const key = `h-${n}`;
42
+ n += 1;
43
+ extractions.push({
44
+ id: `${key}-ext`,
45
+ sourceId: `${key}-src`,
46
+ target: FIELD,
47
+ value: "ACTIVE",
48
+ confidence,
49
+ extractor: EXTRACTOR,
50
+ extractedAt: HISTORY_AT,
51
+ });
52
+ const candidate = { id: `${key}-cand`, extractionId: `${key}-ext`, value: "ACTIVE", confidence };
53
+ candidateSets.push({
54
+ id: `${key}-cs`,
55
+ target: FIELD,
56
+ status: "resolved",
57
+ selectedCandidateId: candidate.id,
58
+ candidates: [candidate],
59
+ });
60
+ reviewOutcomes.push({
61
+ id: `${key}-ro`,
62
+ candidateSetId: `${key}-cs`,
63
+ candidateId: candidate.id,
64
+ status: i < affirmed ? "verified" : "rejected",
65
+ actor: "example-reviewer",
66
+ reviewedAt: HISTORY_AT,
67
+ });
68
+ }
69
+ };
70
+ addSamples(10, 0.95, 10); // top decile: all affirmed
71
+ addSamples(10, 0.85, 10); // 0.8–0.9: all affirmed
72
+ addSamples(10, 0.75, 5); // 0.7–0.8: half affirmed → below target, ends the run
73
+ addSamples(10, 0.55, 1); // 0.5–0.6: mostly rejected
74
+ return { reviewOutcomes, candidateSets, extractions };
75
+ }
76
+ export function runCalibratedAutoAccept() {
77
+ // (1) Derive the empirical calibration curve over the review history.
78
+ const metrics = deriveCalibration(buildReviewHistory(), {
79
+ targetAccuracy: 0.9, // the accuracy the auto-accept threshold must clear
80
+ minBinSamples: 5, // a decile needs this many samples to ground the threshold
81
+ });
82
+ const suggestedThreshold = metrics.overall.suggestedThreshold;
83
+ const group = metrics.byExtractorField.find((g) => g.extractor === EXTRACTOR && g.field === FIELD);
84
+ // This is the number you feed into your producer profile's auto-accept policy:
85
+ // surveySchemaMapping(context, extractor, { autoAcceptMinConfidence: suggestedThreshold })
86
+ // Proposals at/above it were empirically affirmed often enough to auto-accept.
87
+ // (2) Produce calibrated conclusion confidence on a new batch of affirmed claims.
88
+ const input = new SurveyInputBuilder({ source: "example-consumer:calibrated", generatedAt: BATCH_AT })
89
+ .addObservation(affirmedObservation("entity-1", 0.85))
90
+ .addObservation(affirmedObservation("entity-2", 0.92))
91
+ .build();
92
+ // Prefer metrics computed over history (not just this batch), so a claim's own
93
+ // outcome does not feed its own value.
94
+ const bundle = buildSurveyTrustBundle(input, { calibration: { metrics, minSamples: 20 } });
95
+ const producedValues = bundle.claims.map((c) => c.conclusionConfidence?.value);
96
+ return { suggestedThreshold, groupAccuracy: group?.empiricalAccuracy, producedValues };
97
+ }
98
+ function affirmedObservation(subjectId, confidence) {
99
+ return {
100
+ id: `example.${subjectId}.${FIELD}.current`,
101
+ rawSource: {
102
+ kind: "api-record",
103
+ sourceRef: `records://${subjectId}/registry`,
104
+ observedAt: BATCH_AT,
105
+ locatorScheme: "structured-field",
106
+ },
107
+ extraction: {
108
+ target: FIELD,
109
+ value: "ACTIVE",
110
+ confidence,
111
+ locator: "json:$.registrationStatus",
112
+ extractor: EXTRACTOR,
113
+ extractedAt: BATCH_AT,
114
+ },
115
+ reviewOutcome: { status: "verified", actor: "example-reviewer", reviewedAt: BATCH_AT },
116
+ claim: {
117
+ subjectType: "public-record.entity",
118
+ subjectId,
119
+ facet: "public-record.profile",
120
+ claimType: "public-data.field",
121
+ fieldOrBehavior: FIELD,
122
+ impactLevel: "medium",
123
+ collectedBy: EXTRACTOR,
124
+ },
125
+ };
126
+ }
127
+ // Run standalone (not when imported by a test).
128
+ if (process.argv[1]?.endsWith("calibrated-auto-accept.js")) {
129
+ const result = runCalibratedAutoAccept();
130
+ console.log(JSON.stringify({
131
+ suggestedAutoAcceptThreshold: result.suggestedThreshold,
132
+ empiricalAffirmationRate: result.groupAccuracy,
133
+ producedConclusionConfidenceValues: result.producedValues,
134
+ }, null, 2));
135
+ }
@@ -1,6 +1,34 @@
1
1
  import type { TrustBundle } from "@kontourai/surface";
2
+ import { type CalibrationMetrics } from "./calibration.js";
2
3
  import type { SurveyInput } from "./types.js";
4
+ export interface SurveyCalibrationOptions {
5
+ /**
6
+ * Precomputed calibration to source the value from — typically derived over a
7
+ * LONGER history than the current batch (a better-grounded curve, and it avoids
8
+ * the mild self-reference of a claim's own review outcome feeding its value).
9
+ * When omitted, calibration is derived from THIS batch's review outcomes.
10
+ */
11
+ metrics?: CalibrationMetrics;
12
+ /**
13
+ * Minimum labeled samples a group needs before its accuracy is emitted as a
14
+ * value. Groups below the floor leave `value` unset rather than emitting a
15
+ * poorly-grounded number. Default {@link DEFAULT_CALIBRATION_MIN_SAMPLES}.
16
+ */
17
+ minSamples?: number;
18
+ }
3
19
  export interface BuildSurveyTrustBundleOptions {
4
20
  reviewProofs?: boolean;
21
+ /**
22
+ * Populate `conclusionConfidence.value` from empirical review calibration —
23
+ * "how often this extractor's proposals at this confidence were affirmed by a
24
+ * human reviewer" (the produce side of the confidence loop; see #114/#137).
25
+ * `true` derives calibration from this batch; an object supplies precomputed
26
+ * metrics and/or a `minSamples` floor. Absent → `value` stays unset and only
27
+ * the comfort-zone signal is carried (unchanged behavior).
28
+ *
29
+ * ADVISORY (ADR 0003 §4): this only enriches the emitted conclusion confidence;
30
+ * it never changes a claim's `status`.
31
+ */
32
+ calibration?: boolean | SurveyCalibrationOptions;
5
33
  }
6
34
  export declare function buildSurveyTrustBundle(input: SurveyInput, options?: BuildSurveyTrustBundleOptions): TrustBundle;
@@ -1,10 +1,23 @@
1
1
  import { buildReviewProofAnchor } from "./review-proof.js";
2
2
  import { assertReviewOutcomeDiscipline } from "./producer-discipline.js";
3
+ import { deriveCalibration } from "./calibration.js";
4
+ /** Minimum labeled samples a calibration group needs before its empirical
5
+ * accuracy is emitted as a `conclusionConfidence.value`. */
6
+ const DEFAULT_CALIBRATION_MIN_SAMPLES = 20;
3
7
  export function buildSurveyTrustBundle(input, options = {}) {
4
8
  const rawSources = indexById(input.rawSources, "raw source");
5
9
  const extractions = indexById(input.extractions, "extraction");
6
10
  const candidateSets = indexById(input.candidateSets, "candidate set");
7
11
  const reviewsByCandidateSet = groupBy(input.reviewOutcomes, (review) => review.candidateSetId);
12
+ const calibrationOptions = normalizeCalibrationOptions(options.calibration);
13
+ const calibrationMetrics = calibrationOptions
14
+ ? (calibrationOptions.metrics ?? deriveCalibration({
15
+ reviewOutcomes: input.reviewOutcomes,
16
+ candidateSets: input.candidateSets,
17
+ extractions: input.extractions,
18
+ }))
19
+ : undefined;
20
+ const calibrationMinSamples = calibrationOptions?.minSamples ?? DEFAULT_CALIBRATION_MIN_SAMPLES;
8
21
  const claims = [];
9
22
  const evidence = [];
10
23
  const events = [];
@@ -47,18 +60,35 @@ export function buildSurveyTrustBundle(input, options = {}) {
47
60
  survey: buildSurveyMetadata({ projection, rawSource, extraction, candidateSet, candidate, review }),
48
61
  },
49
62
  };
50
- // Promote the review's comfort-zone signal into the first-class
51
- // conclusionConfidence field (Surface 2.9 / Hachure 0.14) so the calibration
63
+ // Promote the review's comfort-zone signal — and, when calibration is
64
+ // enabled, an empirically-calibrated conclusion probability — into the
65
+ // first-class conclusionConfidence field (Surface 2.9 / Hachure 0.14) so the
52
66
  // signal is portable and comparable, not buried in producer metadata.
53
- // We carry only comfortZone: `value` is a *calibrated conclusion probability*
54
- // that Survey does not yet produce (extraction confidence is an ingredient in
55
- // confidenceBasis, not a calibrated conclusion value) — carry, not produce.
56
- if (review?.withinComfortZone !== undefined) {
67
+ //
68
+ // comfortZone is CARRIED from the review. `value` is PRODUCED from empirical
69
+ // review calibration (#114/#137): the affirmation rate of this extractor's
70
+ // proposals at this confidence — a calibrated conclusion probability, distinct
71
+ // from the extraction-confidence ingredient in confidenceBasis.
72
+ //
73
+ // A value is produced only for an AFFIRMED conclusion (status verified/assumed)
74
+ // that clears the sample floor. conclusionConfidence.value is "probability the
75
+ // conclusion is correct"; attaching an affirmation rate to a REJECTED (or
76
+ // not-yet-reviewed) conclusion would assert the opposite of what the human
77
+ // decided, so those claims get no value.
78
+ const comfortZone = review?.withinComfortZone !== undefined
79
+ ? {
80
+ within: review.withinComfortZone,
81
+ ...(review.comfortZoneNote ? { reason: review.comfortZoneNote } : {}),
82
+ }
83
+ : undefined;
84
+ const affirmedConclusion = status === "verified" || status === "assumed";
85
+ const calibrated = calibrationMetrics && review && affirmedConclusion
86
+ ? lookupCalibratedValue(calibrationMetrics, extraction.extractor, extraction.target, calibrationMinSamples)
87
+ : undefined;
88
+ if (comfortZone || calibrated) {
57
89
  claim.conclusionConfidence = {
58
- comfortZone: {
59
- within: review.withinComfortZone,
60
- ...(review.comfortZoneNote ? { reason: review.comfortZoneNote } : {}),
61
- },
90
+ ...(calibrated ? { value: calibrated.value, method: calibrated.method } : {}),
91
+ ...(comfortZone ? { comfortZone } : {}),
62
92
  };
63
93
  }
64
94
  if (options.reviewProofs && review) {
@@ -370,6 +400,31 @@ function selectCandidate(candidateSet, candidateId) {
370
400
  function selectReview(reviews, candidateId) {
371
401
  return reviews.find((review) => review.candidateId === candidateId) ?? reviews.find((review) => !review.candidateId);
372
402
  }
403
+ function normalizeCalibrationOptions(calibration) {
404
+ if (calibration === undefined || calibration === false)
405
+ return undefined;
406
+ if (calibration === true)
407
+ return {};
408
+ return calibration;
409
+ }
410
+ /**
411
+ * Looks up the empirical affirmation rate for an extractor/field, preferring the
412
+ * finer (extractor, field) group and falling back to the extractor-level group
413
+ * when the field group is below the sample floor. Returns undefined when neither
414
+ * group clears the floor, so an ungrounded claim leaves `value` unset. The
415
+ * `method` records which granularity produced the value.
416
+ */
417
+ function lookupCalibratedValue(metrics, extractor, field, minSamples) {
418
+ const fieldGroup = metrics.byExtractorField.find((g) => g.extractor === extractor && g.field === field);
419
+ if (fieldGroup && fieldGroup.sampleCount >= minSamples && fieldGroup.empiricalAccuracy !== undefined) {
420
+ return { value: fieldGroup.empiricalAccuracy, method: "empirical-review-calibration:extractor-field" };
421
+ }
422
+ const extractorGroup = metrics.byExtractor.find((g) => g.extractor === extractor);
423
+ if (extractorGroup && extractorGroup.sampleCount >= minSamples && extractorGroup.empiricalAccuracy !== undefined) {
424
+ return { value: extractorGroup.empiricalAccuracy, method: "empirical-review-calibration:extractor" };
425
+ }
426
+ return undefined;
427
+ }
373
428
  function evidenceTypeFor(rawSource) {
374
429
  if (rawSource.kind === "policy-standard")
375
430
  return "policy_rule";
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@kontourai/survey",
3
- "version": "1.9.0",
3
+ "version": "1.11.0",
4
4
  "description": "Producer-side source, extraction, candidate, and review contracts for projecting verified claims into Surface.",
5
5
  "license": "Apache-2.0",
6
6
  "type": "module",