@kontourai/survey 1.9.0 → 1.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Worked example: wiring confidence calibration into a downstream consumer.
|
|
3
|
+
*
|
|
4
|
+
* A consumer that owns human review outcomes can close the confidence loop in two
|
|
5
|
+
* places, both shipped in @kontourai/survey (1.10.0):
|
|
6
|
+
*
|
|
7
|
+
* 1. Ground the auto-accept threshold. `deriveCalibration` over a history of
|
|
8
|
+
* review outcomes yields `suggestedThreshold` — the lowest confidence at
|
|
9
|
+
* which the extractor's proposals were empirically affirmed often enough.
|
|
10
|
+
* Feed that number into a producer profile's `autoAcceptMinConfidence`
|
|
11
|
+
* (see `SchemaMappingOptions.autoAcceptMinConfidence` /
|
|
12
|
+
* `InquiryMappingOptions.minConfidence`) instead of hand-picking it.
|
|
13
|
+
*
|
|
14
|
+
* 2. Produce calibrated conclusion confidence. Pass the same calibration
|
|
15
|
+
* `metrics` into `buildSurveyTrustBundle({ calibration: { metrics } })` and
|
|
16
|
+
* affirmed claims carry `conclusionConfidence.value` = the empirical
|
|
17
|
+
* affirmation rate for their extractor/field.
|
|
18
|
+
*
|
|
19
|
+
* Calibration is advisory (ADR 0003 §4): the threshold is a suggestion the
|
|
20
|
+
* operator wires into policy, and the produced value never changes claim status.
|
|
21
|
+
*
|
|
22
|
+
* Run: `node dist/examples/calibrated-auto-accept.js`
|
|
23
|
+
*/
|
|
24
|
+
export interface CalibratedAutoAcceptResult {
|
|
25
|
+
readonly suggestedThreshold: number | undefined;
|
|
26
|
+
readonly groupAccuracy: number | undefined;
|
|
27
|
+
readonly producedValues: ReadonlyArray<number | undefined>;
|
|
28
|
+
}
|
|
29
|
+
export declare function runCalibratedAutoAccept(): CalibratedAutoAcceptResult;
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Worked example: wiring confidence calibration into a downstream consumer.
|
|
3
|
+
*
|
|
4
|
+
* A consumer that owns human review outcomes can close the confidence loop in two
|
|
5
|
+
* places, both shipped in @kontourai/survey (1.10.0):
|
|
6
|
+
*
|
|
7
|
+
* 1. Ground the auto-accept threshold. `deriveCalibration` over a history of
|
|
8
|
+
* review outcomes yields `suggestedThreshold` — the lowest confidence at
|
|
9
|
+
* which the extractor's proposals were empirically affirmed often enough.
|
|
10
|
+
* Feed that number into a producer profile's `autoAcceptMinConfidence`
|
|
11
|
+
* (see `SchemaMappingOptions.autoAcceptMinConfidence` /
|
|
12
|
+
* `InquiryMappingOptions.minConfidence`) instead of hand-picking it.
|
|
13
|
+
*
|
|
14
|
+
* 2. Produce calibrated conclusion confidence. Pass the same calibration
|
|
15
|
+
* `metrics` into `buildSurveyTrustBundle({ calibration: { metrics } })` and
|
|
16
|
+
* affirmed claims carry `conclusionConfidence.value` = the empirical
|
|
17
|
+
* affirmation rate for their extractor/field.
|
|
18
|
+
*
|
|
19
|
+
* Calibration is advisory (ADR 0003 §4): the threshold is a suggestion the
|
|
20
|
+
* operator wires into policy, and the produced value never changes claim status.
|
|
21
|
+
*
|
|
22
|
+
* Run: `node dist/examples/calibrated-auto-accept.js`
|
|
23
|
+
*/
|
|
24
|
+
import { buildSurveyTrustBundle, deriveCalibration, SurveyInputBuilder, } from "../src/index.js";
|
|
25
|
+
const EXTRACTOR = "example-extractor";
|
|
26
|
+
const FIELD = "registrationStatus";
|
|
27
|
+
const HISTORY_AT = "2026-06-01T00:00:00.000Z";
|
|
28
|
+
const BATCH_AT = "2026-07-01T00:00:00.000Z";
|
|
29
|
+
/**
|
|
30
|
+
* A synthetic history: for this extractor/field, high-confidence proposals were
|
|
31
|
+
* almost always affirmed by reviewers and low-confidence ones were mostly
|
|
32
|
+
* rejected — the pattern that makes an empirical threshold meaningful.
|
|
33
|
+
*/
|
|
34
|
+
function buildReviewHistory() {
|
|
35
|
+
const extractions = [];
|
|
36
|
+
const candidateSets = [];
|
|
37
|
+
const reviewOutcomes = [];
|
|
38
|
+
let n = 0;
|
|
39
|
+
const addSamples = (count, confidence, affirmed) => {
|
|
40
|
+
for (let i = 0; i < count; i += 1) {
|
|
41
|
+
const key = `h-${n}`;
|
|
42
|
+
n += 1;
|
|
43
|
+
extractions.push({
|
|
44
|
+
id: `${key}-ext`,
|
|
45
|
+
sourceId: `${key}-src`,
|
|
46
|
+
target: FIELD,
|
|
47
|
+
value: "ACTIVE",
|
|
48
|
+
confidence,
|
|
49
|
+
extractor: EXTRACTOR,
|
|
50
|
+
extractedAt: HISTORY_AT,
|
|
51
|
+
});
|
|
52
|
+
const candidate = { id: `${key}-cand`, extractionId: `${key}-ext`, value: "ACTIVE", confidence };
|
|
53
|
+
candidateSets.push({
|
|
54
|
+
id: `${key}-cs`,
|
|
55
|
+
target: FIELD,
|
|
56
|
+
status: "resolved",
|
|
57
|
+
selectedCandidateId: candidate.id,
|
|
58
|
+
candidates: [candidate],
|
|
59
|
+
});
|
|
60
|
+
reviewOutcomes.push({
|
|
61
|
+
id: `${key}-ro`,
|
|
62
|
+
candidateSetId: `${key}-cs`,
|
|
63
|
+
candidateId: candidate.id,
|
|
64
|
+
status: i < affirmed ? "verified" : "rejected",
|
|
65
|
+
actor: "example-reviewer",
|
|
66
|
+
reviewedAt: HISTORY_AT,
|
|
67
|
+
});
|
|
68
|
+
}
|
|
69
|
+
};
|
|
70
|
+
addSamples(10, 0.95, 10); // top decile: all affirmed
|
|
71
|
+
addSamples(10, 0.85, 10); // 0.8–0.9: all affirmed
|
|
72
|
+
addSamples(10, 0.75, 5); // 0.7–0.8: half affirmed → below target, ends the run
|
|
73
|
+
addSamples(10, 0.55, 1); // 0.5–0.6: mostly rejected
|
|
74
|
+
return { reviewOutcomes, candidateSets, extractions };
|
|
75
|
+
}
|
|
76
|
+
export function runCalibratedAutoAccept() {
|
|
77
|
+
// (1) Derive the empirical calibration curve over the review history.
|
|
78
|
+
const metrics = deriveCalibration(buildReviewHistory(), {
|
|
79
|
+
targetAccuracy: 0.9, // the accuracy the auto-accept threshold must clear
|
|
80
|
+
minBinSamples: 5, // a decile needs this many samples to ground the threshold
|
|
81
|
+
});
|
|
82
|
+
const suggestedThreshold = metrics.overall.suggestedThreshold;
|
|
83
|
+
const group = metrics.byExtractorField.find((g) => g.extractor === EXTRACTOR && g.field === FIELD);
|
|
84
|
+
// This is the number you feed into your producer profile's auto-accept policy:
|
|
85
|
+
// surveySchemaMapping(context, extractor, { autoAcceptMinConfidence: suggestedThreshold })
|
|
86
|
+
// Proposals at/above it were empirically affirmed often enough to auto-accept.
|
|
87
|
+
// (2) Produce calibrated conclusion confidence on a new batch of affirmed claims.
|
|
88
|
+
const input = new SurveyInputBuilder({ source: "example-consumer:calibrated", generatedAt: BATCH_AT })
|
|
89
|
+
.addObservation(affirmedObservation("entity-1", 0.85))
|
|
90
|
+
.addObservation(affirmedObservation("entity-2", 0.92))
|
|
91
|
+
.build();
|
|
92
|
+
// Prefer metrics computed over history (not just this batch), so a claim's own
|
|
93
|
+
// outcome does not feed its own value.
|
|
94
|
+
const bundle = buildSurveyTrustBundle(input, { calibration: { metrics, minSamples: 20 } });
|
|
95
|
+
const producedValues = bundle.claims.map((c) => c.conclusionConfidence?.value);
|
|
96
|
+
return { suggestedThreshold, groupAccuracy: group?.empiricalAccuracy, producedValues };
|
|
97
|
+
}
|
|
98
|
+
function affirmedObservation(subjectId, confidence) {
|
|
99
|
+
return {
|
|
100
|
+
id: `example.${subjectId}.${FIELD}.current`,
|
|
101
|
+
rawSource: {
|
|
102
|
+
kind: "api-record",
|
|
103
|
+
sourceRef: `records://${subjectId}/registry`,
|
|
104
|
+
observedAt: BATCH_AT,
|
|
105
|
+
locatorScheme: "structured-field",
|
|
106
|
+
},
|
|
107
|
+
extraction: {
|
|
108
|
+
target: FIELD,
|
|
109
|
+
value: "ACTIVE",
|
|
110
|
+
confidence,
|
|
111
|
+
locator: "json:$.registrationStatus",
|
|
112
|
+
extractor: EXTRACTOR,
|
|
113
|
+
extractedAt: BATCH_AT,
|
|
114
|
+
},
|
|
115
|
+
reviewOutcome: { status: "verified", actor: "example-reviewer", reviewedAt: BATCH_AT },
|
|
116
|
+
claim: {
|
|
117
|
+
subjectType: "public-record.entity",
|
|
118
|
+
subjectId,
|
|
119
|
+
facet: "public-record.profile",
|
|
120
|
+
claimType: "public-data.field",
|
|
121
|
+
fieldOrBehavior: FIELD,
|
|
122
|
+
impactLevel: "medium",
|
|
123
|
+
collectedBy: EXTRACTOR,
|
|
124
|
+
},
|
|
125
|
+
};
|
|
126
|
+
}
|
|
127
|
+
// Run standalone (not when imported by a test).
|
|
128
|
+
if (process.argv[1]?.endsWith("calibrated-auto-accept.js")) {
|
|
129
|
+
const result = runCalibratedAutoAccept();
|
|
130
|
+
console.log(JSON.stringify({
|
|
131
|
+
suggestedAutoAcceptThreshold: result.suggestedThreshold,
|
|
132
|
+
empiricalAffirmationRate: result.groupAccuracy,
|
|
133
|
+
producedConclusionConfidenceValues: result.producedValues,
|
|
134
|
+
}, null, 2));
|
|
135
|
+
}
|
package/dist/src/to-surface.d.ts
CHANGED
|
@@ -1,6 +1,34 @@
|
|
|
1
1
|
import type { TrustBundle } from "@kontourai/surface";
|
|
2
|
+
import { type CalibrationMetrics } from "./calibration.js";
|
|
2
3
|
import type { SurveyInput } from "./types.js";
|
|
4
|
+
export interface SurveyCalibrationOptions {
|
|
5
|
+
/**
|
|
6
|
+
* Precomputed calibration to source the value from — typically derived over a
|
|
7
|
+
* LONGER history than the current batch (a better-grounded curve, and it avoids
|
|
8
|
+
* the mild self-reference of a claim's own review outcome feeding its value).
|
|
9
|
+
* When omitted, calibration is derived from THIS batch's review outcomes.
|
|
10
|
+
*/
|
|
11
|
+
metrics?: CalibrationMetrics;
|
|
12
|
+
/**
|
|
13
|
+
* Minimum labeled samples a group needs before its accuracy is emitted as a
|
|
14
|
+
* value. Groups below the floor leave `value` unset rather than emitting a
|
|
15
|
+
* poorly-grounded number. Default {@link DEFAULT_CALIBRATION_MIN_SAMPLES}.
|
|
16
|
+
*/
|
|
17
|
+
minSamples?: number;
|
|
18
|
+
}
|
|
3
19
|
export interface BuildSurveyTrustBundleOptions {
|
|
4
20
|
reviewProofs?: boolean;
|
|
21
|
+
/**
|
|
22
|
+
* Populate `conclusionConfidence.value` from empirical review calibration —
|
|
23
|
+
* "how often this extractor's proposals at this confidence were affirmed by a
|
|
24
|
+
* human reviewer" (the produce side of the confidence loop; see #114/#137).
|
|
25
|
+
* `true` derives calibration from this batch; an object supplies precomputed
|
|
26
|
+
* metrics and/or a `minSamples` floor. Absent → `value` stays unset and only
|
|
27
|
+
* the comfort-zone signal is carried (unchanged behavior).
|
|
28
|
+
*
|
|
29
|
+
* ADVISORY (ADR 0003 §4): this only enriches the emitted conclusion confidence;
|
|
30
|
+
* it never changes a claim's `status`.
|
|
31
|
+
*/
|
|
32
|
+
calibration?: boolean | SurveyCalibrationOptions;
|
|
5
33
|
}
|
|
6
34
|
export declare function buildSurveyTrustBundle(input: SurveyInput, options?: BuildSurveyTrustBundleOptions): TrustBundle;
|
package/dist/src/to-surface.js
CHANGED
|
@@ -1,10 +1,23 @@
|
|
|
1
1
|
import { buildReviewProofAnchor } from "./review-proof.js";
|
|
2
2
|
import { assertReviewOutcomeDiscipline } from "./producer-discipline.js";
|
|
3
|
+
import { deriveCalibration } from "./calibration.js";
|
|
4
|
+
/** Minimum labeled samples a calibration group needs before its empirical
|
|
5
|
+
* accuracy is emitted as a `conclusionConfidence.value`. */
|
|
6
|
+
const DEFAULT_CALIBRATION_MIN_SAMPLES = 20;
|
|
3
7
|
export function buildSurveyTrustBundle(input, options = {}) {
|
|
4
8
|
const rawSources = indexById(input.rawSources, "raw source");
|
|
5
9
|
const extractions = indexById(input.extractions, "extraction");
|
|
6
10
|
const candidateSets = indexById(input.candidateSets, "candidate set");
|
|
7
11
|
const reviewsByCandidateSet = groupBy(input.reviewOutcomes, (review) => review.candidateSetId);
|
|
12
|
+
const calibrationOptions = normalizeCalibrationOptions(options.calibration);
|
|
13
|
+
const calibrationMetrics = calibrationOptions
|
|
14
|
+
? (calibrationOptions.metrics ?? deriveCalibration({
|
|
15
|
+
reviewOutcomes: input.reviewOutcomes,
|
|
16
|
+
candidateSets: input.candidateSets,
|
|
17
|
+
extractions: input.extractions,
|
|
18
|
+
}))
|
|
19
|
+
: undefined;
|
|
20
|
+
const calibrationMinSamples = calibrationOptions?.minSamples ?? DEFAULT_CALIBRATION_MIN_SAMPLES;
|
|
8
21
|
const claims = [];
|
|
9
22
|
const evidence = [];
|
|
10
23
|
const events = [];
|
|
@@ -47,18 +60,35 @@ export function buildSurveyTrustBundle(input, options = {}) {
|
|
|
47
60
|
survey: buildSurveyMetadata({ projection, rawSource, extraction, candidateSet, candidate, review }),
|
|
48
61
|
},
|
|
49
62
|
};
|
|
50
|
-
// Promote the review's comfort-zone signal
|
|
51
|
-
//
|
|
63
|
+
// Promote the review's comfort-zone signal — and, when calibration is
|
|
64
|
+
// enabled, an empirically-calibrated conclusion probability — into the
|
|
65
|
+
// first-class conclusionConfidence field (Surface 2.9 / Hachure 0.14) so the
|
|
52
66
|
// signal is portable and comparable, not buried in producer metadata.
|
|
53
|
-
//
|
|
54
|
-
//
|
|
55
|
-
//
|
|
56
|
-
|
|
67
|
+
//
|
|
68
|
+
// comfortZone is CARRIED from the review. `value` is PRODUCED from empirical
|
|
69
|
+
// review calibration (#114/#137): the affirmation rate of this extractor's
|
|
70
|
+
// proposals at this confidence — a calibrated conclusion probability, distinct
|
|
71
|
+
// from the extraction-confidence ingredient in confidenceBasis.
|
|
72
|
+
//
|
|
73
|
+
// A value is produced only for an AFFIRMED conclusion (status verified/assumed)
|
|
74
|
+
// that clears the sample floor. conclusionConfidence.value is "probability the
|
|
75
|
+
// conclusion is correct"; attaching an affirmation rate to a REJECTED (or
|
|
76
|
+
// not-yet-reviewed) conclusion would assert the opposite of what the human
|
|
77
|
+
// decided, so those claims get no value.
|
|
78
|
+
const comfortZone = review?.withinComfortZone !== undefined
|
|
79
|
+
? {
|
|
80
|
+
within: review.withinComfortZone,
|
|
81
|
+
...(review.comfortZoneNote ? { reason: review.comfortZoneNote } : {}),
|
|
82
|
+
}
|
|
83
|
+
: undefined;
|
|
84
|
+
const affirmedConclusion = status === "verified" || status === "assumed";
|
|
85
|
+
const calibrated = calibrationMetrics && review && affirmedConclusion
|
|
86
|
+
? lookupCalibratedValue(calibrationMetrics, extraction.extractor, extraction.target, calibrationMinSamples)
|
|
87
|
+
: undefined;
|
|
88
|
+
if (comfortZone || calibrated) {
|
|
57
89
|
claim.conclusionConfidence = {
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
...(review.comfortZoneNote ? { reason: review.comfortZoneNote } : {}),
|
|
61
|
-
},
|
|
90
|
+
...(calibrated ? { value: calibrated.value, method: calibrated.method } : {}),
|
|
91
|
+
...(comfortZone ? { comfortZone } : {}),
|
|
62
92
|
};
|
|
63
93
|
}
|
|
64
94
|
if (options.reviewProofs && review) {
|
|
@@ -370,6 +400,31 @@ function selectCandidate(candidateSet, candidateId) {
|
|
|
370
400
|
function selectReview(reviews, candidateId) {
|
|
371
401
|
return reviews.find((review) => review.candidateId === candidateId) ?? reviews.find((review) => !review.candidateId);
|
|
372
402
|
}
|
|
403
|
+
function normalizeCalibrationOptions(calibration) {
|
|
404
|
+
if (calibration === undefined || calibration === false)
|
|
405
|
+
return undefined;
|
|
406
|
+
if (calibration === true)
|
|
407
|
+
return {};
|
|
408
|
+
return calibration;
|
|
409
|
+
}
|
|
410
|
+
/**
|
|
411
|
+
* Looks up the empirical affirmation rate for an extractor/field, preferring the
|
|
412
|
+
* finer (extractor, field) group and falling back to the extractor-level group
|
|
413
|
+
* when the field group is below the sample floor. Returns undefined when neither
|
|
414
|
+
* group clears the floor, so an ungrounded claim leaves `value` unset. The
|
|
415
|
+
* `method` records which granularity produced the value.
|
|
416
|
+
*/
|
|
417
|
+
function lookupCalibratedValue(metrics, extractor, field, minSamples) {
|
|
418
|
+
const fieldGroup = metrics.byExtractorField.find((g) => g.extractor === extractor && g.field === field);
|
|
419
|
+
if (fieldGroup && fieldGroup.sampleCount >= minSamples && fieldGroup.empiricalAccuracy !== undefined) {
|
|
420
|
+
return { value: fieldGroup.empiricalAccuracy, method: "empirical-review-calibration:extractor-field" };
|
|
421
|
+
}
|
|
422
|
+
const extractorGroup = metrics.byExtractor.find((g) => g.extractor === extractor);
|
|
423
|
+
if (extractorGroup && extractorGroup.sampleCount >= minSamples && extractorGroup.empiricalAccuracy !== undefined) {
|
|
424
|
+
return { value: extractorGroup.empiricalAccuracy, method: "empirical-review-calibration:extractor" };
|
|
425
|
+
}
|
|
426
|
+
return undefined;
|
|
427
|
+
}
|
|
373
428
|
function evidenceTypeFor(rawSource) {
|
|
374
429
|
if (rawSource.kind === "policy-standard")
|
|
375
430
|
return "policy_rule";
|
package/package.json
CHANGED