@kontourai/survey 1.10.0 → 1.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Worked example: wiring confidence calibration into a downstream consumer.
|
|
3
|
+
*
|
|
4
|
+
* A consumer that owns human review outcomes can close the confidence loop in two
|
|
5
|
+
* places, both shipped in @kontourai/survey (1.10.0):
|
|
6
|
+
*
|
|
7
|
+
* 1. Ground the auto-accept threshold. `deriveCalibration` over a history of
|
|
8
|
+
* review outcomes yields `suggestedThreshold` — the lowest confidence at
|
|
9
|
+
* which the extractor's proposals were empirically affirmed often enough.
|
|
10
|
+
* Feed that number into a producer profile's `autoAcceptMinConfidence`
|
|
11
|
+
* (see `SchemaMappingOptions.autoAcceptMinConfidence` /
|
|
12
|
+
* `InquiryMappingOptions.minConfidence`) instead of hand-picking it.
|
|
13
|
+
*
|
|
14
|
+
* 2. Produce calibrated conclusion confidence. Pass the same calibration
|
|
15
|
+
* `metrics` into `buildSurveyTrustBundle({ calibration: { metrics } })` and
|
|
16
|
+
* affirmed claims carry `conclusionConfidence.value` = the empirical
|
|
17
|
+
* affirmation rate for their extractor/field.
|
|
18
|
+
*
|
|
19
|
+
* Calibration is advisory (ADR 0003 §4): the threshold is a suggestion the
|
|
20
|
+
* operator wires into policy, and the produced value never changes claim status.
|
|
21
|
+
*
|
|
22
|
+
* Run: `node dist/examples/calibrated-auto-accept.js`
|
|
23
|
+
*/
|
|
24
|
+
export interface CalibratedAutoAcceptResult {
|
|
25
|
+
readonly suggestedThreshold: number | undefined;
|
|
26
|
+
readonly groupAccuracy: number | undefined;
|
|
27
|
+
readonly producedValues: ReadonlyArray<number | undefined>;
|
|
28
|
+
}
|
|
29
|
+
export declare function runCalibratedAutoAccept(): CalibratedAutoAcceptResult;
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Worked example: wiring confidence calibration into a downstream consumer.
|
|
3
|
+
*
|
|
4
|
+
* A consumer that owns human review outcomes can close the confidence loop in two
|
|
5
|
+
* places, both shipped in @kontourai/survey (1.10.0):
|
|
6
|
+
*
|
|
7
|
+
* 1. Ground the auto-accept threshold. `deriveCalibration` over a history of
|
|
8
|
+
* review outcomes yields `suggestedThreshold` — the lowest confidence at
|
|
9
|
+
* which the extractor's proposals were empirically affirmed often enough.
|
|
10
|
+
* Feed that number into a producer profile's `autoAcceptMinConfidence`
|
|
11
|
+
* (see `SchemaMappingOptions.autoAcceptMinConfidence` /
|
|
12
|
+
* `InquiryMappingOptions.minConfidence`) instead of hand-picking it.
|
|
13
|
+
*
|
|
14
|
+
* 2. Produce calibrated conclusion confidence. Pass the same calibration
|
|
15
|
+
* `metrics` into `buildSurveyTrustBundle({ calibration: { metrics } })` and
|
|
16
|
+
* affirmed claims carry `conclusionConfidence.value` = the empirical
|
|
17
|
+
* affirmation rate for their extractor/field.
|
|
18
|
+
*
|
|
19
|
+
* Calibration is advisory (ADR 0003 §4): the threshold is a suggestion the
|
|
20
|
+
* operator wires into policy, and the produced value never changes claim status.
|
|
21
|
+
*
|
|
22
|
+
* Run: `node dist/examples/calibrated-auto-accept.js`
|
|
23
|
+
*/
|
|
24
|
+
import { buildSurveyTrustBundle, deriveCalibration, SurveyInputBuilder, } from "../src/index.js";
|
|
25
|
+
const EXTRACTOR = "example-extractor";
|
|
26
|
+
const FIELD = "registrationStatus";
|
|
27
|
+
const HISTORY_AT = "2026-06-01T00:00:00.000Z";
|
|
28
|
+
const BATCH_AT = "2026-07-01T00:00:00.000Z";
|
|
29
|
+
/**
|
|
30
|
+
* A synthetic history: for this extractor/field, high-confidence proposals were
|
|
31
|
+
* almost always affirmed by reviewers and low-confidence ones were mostly
|
|
32
|
+
* rejected — the pattern that makes an empirical threshold meaningful.
|
|
33
|
+
*/
|
|
34
|
+
function buildReviewHistory() {
|
|
35
|
+
const extractions = [];
|
|
36
|
+
const candidateSets = [];
|
|
37
|
+
const reviewOutcomes = [];
|
|
38
|
+
let n = 0;
|
|
39
|
+
const addSamples = (count, confidence, affirmed) => {
|
|
40
|
+
for (let i = 0; i < count; i += 1) {
|
|
41
|
+
const key = `h-${n}`;
|
|
42
|
+
n += 1;
|
|
43
|
+
extractions.push({
|
|
44
|
+
id: `${key}-ext`,
|
|
45
|
+
sourceId: `${key}-src`,
|
|
46
|
+
target: FIELD,
|
|
47
|
+
value: "ACTIVE",
|
|
48
|
+
confidence,
|
|
49
|
+
extractor: EXTRACTOR,
|
|
50
|
+
extractedAt: HISTORY_AT,
|
|
51
|
+
});
|
|
52
|
+
const candidate = { id: `${key}-cand`, extractionId: `${key}-ext`, value: "ACTIVE", confidence };
|
|
53
|
+
candidateSets.push({
|
|
54
|
+
id: `${key}-cs`,
|
|
55
|
+
target: FIELD,
|
|
56
|
+
status: "resolved",
|
|
57
|
+
selectedCandidateId: candidate.id,
|
|
58
|
+
candidates: [candidate],
|
|
59
|
+
});
|
|
60
|
+
reviewOutcomes.push({
|
|
61
|
+
id: `${key}-ro`,
|
|
62
|
+
candidateSetId: `${key}-cs`,
|
|
63
|
+
candidateId: candidate.id,
|
|
64
|
+
status: i < affirmed ? "verified" : "rejected",
|
|
65
|
+
actor: "example-reviewer",
|
|
66
|
+
reviewedAt: HISTORY_AT,
|
|
67
|
+
});
|
|
68
|
+
}
|
|
69
|
+
};
|
|
70
|
+
addSamples(10, 0.95, 10); // top decile: all affirmed
|
|
71
|
+
addSamples(10, 0.85, 10); // 0.8–0.9: all affirmed
|
|
72
|
+
addSamples(10, 0.75, 5); // 0.7–0.8: half affirmed → below target, ends the run
|
|
73
|
+
addSamples(10, 0.55, 1); // 0.5–0.6: mostly rejected
|
|
74
|
+
return { reviewOutcomes, candidateSets, extractions };
|
|
75
|
+
}
|
|
76
|
+
export function runCalibratedAutoAccept() {
|
|
77
|
+
// (1) Derive the empirical calibration curve over the review history.
|
|
78
|
+
const metrics = deriveCalibration(buildReviewHistory(), {
|
|
79
|
+
targetAccuracy: 0.9, // the accuracy the auto-accept threshold must clear
|
|
80
|
+
minBinSamples: 5, // a decile needs this many samples to ground the threshold
|
|
81
|
+
});
|
|
82
|
+
const suggestedThreshold = metrics.overall.suggestedThreshold;
|
|
83
|
+
const group = metrics.byExtractorField.find((g) => g.extractor === EXTRACTOR && g.field === FIELD);
|
|
84
|
+
// This is the number you feed into your producer profile's auto-accept policy:
|
|
85
|
+
// surveySchemaMapping(context, extractor, { autoAcceptMinConfidence: suggestedThreshold })
|
|
86
|
+
// Proposals at/above it were empirically affirmed often enough to auto-accept.
|
|
87
|
+
// (2) Produce calibrated conclusion confidence on a new batch of affirmed claims.
|
|
88
|
+
const input = new SurveyInputBuilder({ source: "example-consumer:calibrated", generatedAt: BATCH_AT })
|
|
89
|
+
.addObservation(affirmedObservation("entity-1", 0.85))
|
|
90
|
+
.addObservation(affirmedObservation("entity-2", 0.92))
|
|
91
|
+
.build();
|
|
92
|
+
// Prefer metrics computed over history (not just this batch), so a claim's own
|
|
93
|
+
// outcome does not feed its own value.
|
|
94
|
+
const bundle = buildSurveyTrustBundle(input, { calibration: { metrics, minSamples: 20 } });
|
|
95
|
+
const producedValues = bundle.claims.map((c) => c.conclusionConfidence?.value);
|
|
96
|
+
return { suggestedThreshold, groupAccuracy: group?.empiricalAccuracy, producedValues };
|
|
97
|
+
}
|
|
98
|
+
function affirmedObservation(subjectId, confidence) {
|
|
99
|
+
return {
|
|
100
|
+
id: `example.${subjectId}.${FIELD}.current`,
|
|
101
|
+
rawSource: {
|
|
102
|
+
kind: "api-record",
|
|
103
|
+
sourceRef: `records://${subjectId}/registry`,
|
|
104
|
+
observedAt: BATCH_AT,
|
|
105
|
+
locatorScheme: "structured-field",
|
|
106
|
+
},
|
|
107
|
+
extraction: {
|
|
108
|
+
target: FIELD,
|
|
109
|
+
value: "ACTIVE",
|
|
110
|
+
confidence,
|
|
111
|
+
locator: "json:$.registrationStatus",
|
|
112
|
+
extractor: EXTRACTOR,
|
|
113
|
+
extractedAt: BATCH_AT,
|
|
114
|
+
},
|
|
115
|
+
reviewOutcome: { status: "verified", actor: "example-reviewer", reviewedAt: BATCH_AT },
|
|
116
|
+
claim: {
|
|
117
|
+
subjectType: "public-record.entity",
|
|
118
|
+
subjectId,
|
|
119
|
+
facet: "public-record.profile",
|
|
120
|
+
claimType: "public-data.field",
|
|
121
|
+
fieldOrBehavior: FIELD,
|
|
122
|
+
impactLevel: "medium",
|
|
123
|
+
collectedBy: EXTRACTOR,
|
|
124
|
+
},
|
|
125
|
+
};
|
|
126
|
+
}
|
|
127
|
+
// Run standalone (not when imported by a test).
|
|
128
|
+
if (process.argv[1]?.endsWith("calibrated-auto-accept.js")) {
|
|
129
|
+
const result = runCalibratedAutoAccept();
|
|
130
|
+
console.log(JSON.stringify({
|
|
131
|
+
suggestedAutoAcceptThreshold: result.suggestedThreshold,
|
|
132
|
+
empiricalAffirmationRate: result.groupAccuracy,
|
|
133
|
+
producedConclusionConfidenceValues: result.producedValues,
|
|
134
|
+
}, null, 2));
|
|
135
|
+
}
|
package/package.json
CHANGED