@kontourai/survey 3.0.0 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -0
- package/dist/example-data/public-directory-review-resource.d.ts +3 -3
- package/dist/examples/calibrated-auto-accept.d.ts +22 -15
- package/dist/examples/calibrated-auto-accept.js +40 -36
- package/dist/src/calibration.d.ts +48 -21
- package/dist/src/calibration.js +72 -33
- package/dist/src/console/review-console-server.d.ts +3 -1
- package/dist/src/console/review-console-server.js +203 -50
- package/dist/src/extraction-envelope.d.ts +22 -0
- package/dist/src/extraction-envelope.js +25 -4
- package/dist/src/index.d.ts +8 -7
- package/dist/src/index.js +2 -2
- package/dist/src/inquiry-mapping.d.ts +15 -1
- package/dist/src/inquiry-mapping.js +10 -2
- package/dist/src/mcp/review-mcp.js +70 -65
- package/dist/src/producer-profile.d.ts +41 -2
- package/dist/src/producer-profile.js +29 -2
- package/dist/src/review-session-file.d.ts +64 -0
- package/dist/src/review-session-file.js +320 -0
- package/dist/src/review-workbench/edited-value.d.ts +70 -0
- package/dist/src/review-workbench/edited-value.js +147 -0
- package/dist/src/review-workbench/review-presentation.d.ts +44 -0
- package/dist/src/review-workbench/review-presentation.js +49 -0
- package/dist/src/review-workbench/review-queue-session.js +8 -1
- package/dist/src/review-workbench/review-session-replay.d.ts +35 -1
- package/dist/src/review-workbench/review-session-replay.js +77 -0
- package/dist/src/review-workbench/review-workbench.d.ts +8 -4
- package/dist/src/review-workbench/review-workbench.js +19 -7
- package/dist/src/review-workbench/server-review-session.d.ts +3 -1
- package/dist/src/review-workbench/server-review-session.js +1 -0
- package/dist/src/reviewed-candidate-resolution.js +13 -7
- package/dist/src/schema-mapping.d.ts +23 -0
- package/dist/src/schema-mapping.js +30 -20
- package/dist/src/to-surface.d.ts +30 -6
- package/dist/src/to-surface.js +196 -18
- package/dist/src/types.d.ts +39 -0
- package/package.json +5 -4
package/README.md
CHANGED
|
@@ -40,6 +40,10 @@ The Review Workbench rendering a real example queue — current vs proposed valu
|
|
|
40
40
|
npm install @kontourai/survey @kontourai/surface
|
|
41
41
|
```
|
|
42
42
|
|
|
43
|
+
Survey supports `@kontourai/surface` 2.13 and later 2.x, and 3.x, and CI tests
|
|
44
|
+
the lowest 2.x and the newest 3.x. Install either major and your project and
|
|
45
|
+
Survey share one Surface copy (`npm ls @kontourai/surface` shows one entry).
|
|
46
|
+
|
|
43
47
|
Requires Node.js >=22. TypeScript >=5.0 is required to compile against
|
|
44
48
|
Survey's published type declarations (`defineProductVocabulary`'s `const`
|
|
45
49
|
type parameters are TS 5.0+ syntax) — JavaScript consumers are unaffected;
|
|
@@ -35,12 +35,12 @@ export declare const publicDirectoryReviewItemExample: {
|
|
|
35
35
|
excerpt: string;
|
|
36
36
|
};
|
|
37
37
|
extraction: {
|
|
38
|
-
model?: undefined;
|
|
39
38
|
extractionId: string;
|
|
40
39
|
target: string;
|
|
41
40
|
confidence: number;
|
|
42
41
|
extractor: string;
|
|
43
42
|
extractedAt: string;
|
|
43
|
+
model?: undefined;
|
|
44
44
|
};
|
|
45
45
|
claimTarget: {
|
|
46
46
|
claimId: string;
|
|
@@ -61,13 +61,13 @@ export declare const publicDirectoryReviewItemExample: {
|
|
|
61
61
|
claimId: string;
|
|
62
62
|
};
|
|
63
63
|
producer: {
|
|
64
|
-
proposalId?: undefined;
|
|
65
|
-
oldValue?: undefined;
|
|
66
64
|
sourceAuthority: {
|
|
67
65
|
authorityClass: string;
|
|
68
66
|
declaredBy: string;
|
|
69
67
|
scope: string;
|
|
70
68
|
};
|
|
69
|
+
proposalId?: undefined;
|
|
70
|
+
oldValue?: undefined;
|
|
71
71
|
};
|
|
72
72
|
} | {
|
|
73
73
|
id: string;
|
|
@@ -1,29 +1,36 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Worked example:
|
|
2
|
+
* Worked example: EXPERIMENTAL confidence calibration from review history.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
4
|
+
* Calibration summarizes how often reviewers affirmed an extractor's proposals.
|
|
5
|
+
* Both outputs below are experimental descriptive statistics (#279), not
|
|
6
|
+
* validated probabilities or policy:
|
|
6
7
|
*
|
|
7
|
-
* 1.
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
* (
|
|
12
|
-
*
|
|
8
|
+
* 1. `deriveCalibration` reports per-decile affirmation rates and, only when a
|
|
9
|
+
* decile has enough samples AND a one-sided 95% Wilson lower bound on its
|
|
10
|
+
* accuracy clears the target, a `suggestedThreshold`. A thin history gets
|
|
11
|
+
* no threshold at all. Do not feed `suggestedThreshold` into an auto-accept
|
|
12
|
+
* policy (`autoAcceptMinConfidence` / `minConfidence`) as-is: it is not an
|
|
13
|
+
* evaluated gate.
|
|
13
14
|
*
|
|
14
|
-
* 2.
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
15
|
+
* 2. `buildSurveyTrustBundle({ calibration: { experimentalConclusionValue:
|
|
16
|
+
* true, metrics } })` sets `conclusionConfidence.value` on affirmed claims
|
|
17
|
+
* to their extractor/field GROUP's affirmation rate. Every affirmed claim in
|
|
18
|
+
* the group gets the same base rate whatever its own confidence, so it is
|
|
19
|
+
* not a per-claim probability. Without the experimental flag no value is set.
|
|
18
20
|
*
|
|
19
|
-
* Calibration is advisory (ADR 0003 §4):
|
|
20
|
-
* operator wires into policy, and the produced value never changes claim status.
|
|
21
|
+
* Calibration is advisory (ADR 0003 §4): it never changes claim status.
|
|
21
22
|
*
|
|
22
23
|
* Run: `node dist/examples/calibrated-auto-accept.js`
|
|
23
24
|
*/
|
|
24
25
|
export interface CalibratedAutoAcceptResult {
|
|
26
|
+
/** Threshold from a 60-samples-per-decile history (enough evidence). */
|
|
25
27
|
readonly suggestedThreshold: number | undefined;
|
|
28
|
+
/** Threshold from a 10-samples-per-decile history (withheld: too thin). */
|
|
29
|
+
readonly sparseHistoryThreshold: number | undefined;
|
|
26
30
|
readonly groupAccuracy: number | undefined;
|
|
31
|
+
/** Values with the experimental opt-in: the group base rate on each claim. */
|
|
27
32
|
readonly producedValues: ReadonlyArray<number | undefined>;
|
|
33
|
+
/** Values with `calibration` enabled but no experimental opt-in: none. */
|
|
34
|
+
readonly defaultValues: ReadonlyArray<number | undefined>;
|
|
28
35
|
}
|
|
29
36
|
export declare function runCalibratedAutoAccept(): CalibratedAutoAcceptResult;
|
|
@@ -1,23 +1,24 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Worked example:
|
|
2
|
+
* Worked example: EXPERIMENTAL confidence calibration from review history.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
4
|
+
* Calibration summarizes how often reviewers affirmed an extractor's proposals.
|
|
5
|
+
* Both outputs below are experimental descriptive statistics (#279), not
|
|
6
|
+
* validated probabilities or policy:
|
|
6
7
|
*
|
|
7
|
-
* 1.
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
* (
|
|
12
|
-
*
|
|
8
|
+
* 1. `deriveCalibration` reports per-decile affirmation rates and, only when a
|
|
9
|
+
* decile has enough samples AND a one-sided 95% Wilson lower bound on its
|
|
10
|
+
* accuracy clears the target, a `suggestedThreshold`. A thin history gets
|
|
11
|
+
* no threshold at all. Do not feed `suggestedThreshold` into an auto-accept
|
|
12
|
+
* policy (`autoAcceptMinConfidence` / `minConfidence`) as-is: it is not an
|
|
13
|
+
* evaluated gate.
|
|
13
14
|
*
|
|
14
|
-
* 2.
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
15
|
+
* 2. `buildSurveyTrustBundle({ calibration: { experimentalConclusionValue:
|
|
16
|
+
* true, metrics } })` sets `conclusionConfidence.value` on affirmed claims
|
|
17
|
+
* to their extractor/field GROUP's affirmation rate. Every affirmed claim in
|
|
18
|
+
* the group gets the same base rate whatever its own confidence, so it is
|
|
19
|
+
* not a per-claim probability. Without the experimental flag no value is set.
|
|
18
20
|
*
|
|
19
|
-
* Calibration is advisory (ADR 0003 §4):
|
|
20
|
-
* operator wires into policy, and the produced value never changes claim status.
|
|
21
|
+
* Calibration is advisory (ADR 0003 §4): it never changes claim status.
|
|
21
22
|
*
|
|
22
23
|
* Run: `node dist/examples/calibrated-auto-accept.js`
|
|
23
24
|
*/
|
|
@@ -31,15 +32,15 @@ const BATCH_AT = "2026-07-01T00:00:00.000Z";
|
|
|
31
32
|
* almost always affirmed by reviewers and low-confidence ones were mostly
|
|
32
33
|
* rejected — the pattern that makes an empirical threshold meaningful.
|
|
33
34
|
*/
|
|
34
|
-
function buildReviewHistory() {
|
|
35
|
+
function buildReviewHistory(samplesPerDecile) {
|
|
35
36
|
const extractions = [];
|
|
36
37
|
const candidateSets = [];
|
|
37
38
|
const reviewOutcomes = [];
|
|
38
|
-
let
|
|
39
|
+
let seq = 0;
|
|
39
40
|
const addSamples = (count, confidence, affirmed) => {
|
|
40
41
|
for (let i = 0; i < count; i += 1) {
|
|
41
|
-
const key = `h-${
|
|
42
|
-
|
|
42
|
+
const key = `h-${seq}`;
|
|
43
|
+
seq += 1;
|
|
43
44
|
extractions.push({
|
|
44
45
|
id: `${key}-ext`,
|
|
45
46
|
sourceId: `${key}-src`,
|
|
@@ -67,33 +68,34 @@ function buildReviewHistory() {
|
|
|
67
68
|
});
|
|
68
69
|
}
|
|
69
70
|
};
|
|
70
|
-
|
|
71
|
-
addSamples(
|
|
72
|
-
addSamples(
|
|
73
|
-
addSamples(
|
|
71
|
+
const n = samplesPerDecile;
|
|
72
|
+
addSamples(n, 0.95, n); // top decile: all affirmed
|
|
73
|
+
addSamples(n, 0.85, n - 1); // 0.8–0.9: all but one affirmed
|
|
74
|
+
addSamples(n, 0.75, Math.round(n / 2)); // 0.7–0.8: half affirmed → ends the run
|
|
75
|
+
addSamples(n, 0.55, Math.round(n / 10)); // 0.5–0.6: mostly rejected
|
|
74
76
|
return { reviewOutcomes, candidateSets, extractions };
|
|
75
77
|
}
|
|
76
78
|
export function runCalibratedAutoAccept() {
|
|
77
79
|
// (1) Derive the empirical calibration curve over the review history.
|
|
78
|
-
const
|
|
79
|
-
|
|
80
|
-
minBinSamples: 5, // a decile needs this many samples to ground the threshold
|
|
81
|
-
});
|
|
80
|
+
const options = { targetAccuracy: 0.9 }; // default 30-sample floor per decile
|
|
81
|
+
const metrics = deriveCalibration(buildReviewHistory(60), options);
|
|
82
82
|
const suggestedThreshold = metrics.overall.suggestedThreshold;
|
|
83
|
+
const sparseHistoryThreshold = deriveCalibration(buildReviewHistory(10), options).overall.suggestedThreshold;
|
|
83
84
|
const group = metrics.byExtractorField.find((g) => g.extractor === EXTRACTOR && g.field === FIELD);
|
|
84
|
-
//
|
|
85
|
-
// surveySchemaMapping(context, extractor, { autoAcceptMinConfidence: suggestedThreshold })
|
|
86
|
-
// Proposals at/above it were empirically affirmed often enough to auto-accept.
|
|
87
|
-
// (2) Produce calibrated conclusion confidence on a new batch of affirmed claims.
|
|
85
|
+
// (2) Attach the group affirmation rate to a new batch of affirmed claims.
|
|
88
86
|
const input = new SurveyInputBuilder({ source: "example-consumer:calibrated", generatedAt: BATCH_AT })
|
|
89
87
|
.addObservation(affirmedObservation("entity-1", 0.85))
|
|
90
88
|
.addObservation(affirmedObservation("entity-2", 0.92))
|
|
91
89
|
.build();
|
|
92
90
|
// Prefer metrics computed over history (not just this batch), so a claim's own
|
|
93
91
|
// outcome does not feed its own value.
|
|
94
|
-
const bundle = buildSurveyTrustBundle(input, {
|
|
92
|
+
const bundle = buildSurveyTrustBundle(input, {
|
|
93
|
+
calibration: { experimentalConclusionValue: true, metrics, minSamples: 20 },
|
|
94
|
+
});
|
|
95
95
|
const producedValues = bundle.claims.map((c) => c.conclusionConfidence?.value);
|
|
96
|
-
|
|
96
|
+
const defaultValues = buildSurveyTrustBundle(input, { calibration: { metrics, minSamples: 20 } })
|
|
97
|
+
.claims.map((c) => c.conclusionConfidence?.value);
|
|
98
|
+
return { suggestedThreshold, sparseHistoryThreshold, groupAccuracy: group?.empiricalAccuracy, producedValues, defaultValues };
|
|
97
99
|
}
|
|
98
100
|
function affirmedObservation(subjectId, confidence) {
|
|
99
101
|
return {
|
|
@@ -128,8 +130,10 @@ function affirmedObservation(subjectId, confidence) {
|
|
|
128
130
|
if (process.argv[1]?.endsWith("calibrated-auto-accept.js")) {
|
|
129
131
|
const result = runCalibratedAutoAccept();
|
|
130
132
|
console.log(JSON.stringify({
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
133
|
+
experimentalSuggestedThreshold: result.suggestedThreshold,
|
|
134
|
+
sparseHistorySuggestedThreshold: result.sparseHistoryThreshold,
|
|
135
|
+
groupAffirmationRate: result.groupAccuracy,
|
|
136
|
+
experimentalConclusionValues: result.producedValues,
|
|
137
|
+
valuesWithoutOptIn: result.defaultValues,
|
|
134
138
|
}, null, 2));
|
|
135
139
|
}
|
|
@@ -9,10 +9,16 @@
|
|
|
9
9
|
* from extractor X was affirmed 17/20 times" — plus a calibration gap and an
|
|
10
10
|
* empirically-grounded auto-accept threshold suggestion.
|
|
11
11
|
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
12
|
+
* EXPERIMENTAL. The curve is a descriptive summary of past review outcomes, not
|
|
13
|
+
* a validated probability model. `suggestedThreshold` is withheld unless every
|
|
14
|
+
* contributing bin clears a sample floor AND a one-sided 95% Wilson lower bound
|
|
15
|
+
* on its accuracy clears the target (#279); even then it is not an auto-accept
|
|
16
|
+
* gate — do not wire it into `autoAcceptMinConfidence` without your own
|
|
17
|
+
* evaluation.
|
|
18
|
+
*
|
|
19
|
+
* ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration never decides a claim
|
|
20
|
+
* or mutates a status. Projected claims carry status "proposed", exactly like
|
|
21
|
+
* every other producer proposal.
|
|
16
22
|
*
|
|
17
23
|
* Machine auto-accepts are EXCLUDED by default: an auto-accepted outcome is the
|
|
18
24
|
* threshold accepting its own guess, so counting it as a "correct" label would
|
|
@@ -49,15 +55,19 @@ export interface DeriveCalibrationOptions {
|
|
|
49
55
|
/** Number of equal-width confidence bins over [0,1]. Default 10 (deciles). */
|
|
50
56
|
readonly binCount?: number;
|
|
51
57
|
/**
|
|
52
|
-
* The
|
|
53
|
-
*
|
|
54
|
-
*
|
|
58
|
+
* The accuracy the suggested threshold must clear. Default 0.95. A
|
|
59
|
+
* `suggestedThreshold` is the lowest bin lower-bound at/above which every bin
|
|
60
|
+
* has ≥ `minBinSamples` samples and a one-sided 95% Wilson lower confidence
|
|
61
|
+
* bound on its accuracy (`CalibrationBin.accuracyLowerBound`) that meets this
|
|
62
|
+
* target — the point estimate alone is not enough.
|
|
55
63
|
*/
|
|
56
64
|
readonly targetAccuracy?: number;
|
|
57
65
|
/**
|
|
58
66
|
* A bin needs at least this many samples to count toward `suggestedThreshold`
|
|
59
|
-
* (
|
|
60
|
-
*
|
|
67
|
+
* (an under-sampled bin ends the qualifying run). Default
|
|
68
|
+
* {@link DEFAULT_MIN_BIN_SAMPLES} (30). The Wilson bound applies on top of this
|
|
69
|
+
* floor, so an all-affirmed bin needs more samples than the floor when the
|
|
70
|
+
* target is high (≈52 all-affirmed samples for the default 0.95 target).
|
|
61
71
|
*/
|
|
62
72
|
readonly minBinSamples?: number;
|
|
63
73
|
/**
|
|
@@ -95,6 +105,13 @@ export interface CalibrationBin {
|
|
|
95
105
|
readonly empiricalAccuracy: number | undefined;
|
|
96
106
|
/** Mean predicted confidence of samples in the bin; undefined when empty. */
|
|
97
107
|
readonly meanPredictedConfidence: number | undefined;
|
|
108
|
+
/**
|
|
109
|
+
* One-sided 95% Wilson lower confidence bound on the bin's accuracy; undefined
|
|
110
|
+
* when the bin is empty. `suggestedThreshold` requires this bound, not the
|
|
111
|
+
* point estimate, to meet `targetAccuracy`. Optional in the type so metrics
|
|
112
|
+
* built by hand before this field existed still type-check.
|
|
113
|
+
*/
|
|
114
|
+
readonly accuracyLowerBound?: number;
|
|
98
115
|
}
|
|
99
116
|
/** A calibration rollup for one extractor (and optionally one field). */
|
|
100
117
|
export interface CalibrationGroup {
|
|
@@ -116,11 +133,12 @@ export interface CalibrationGroup {
|
|
|
116
133
|
/** Per-bin empirical accuracy, ascending by lowerBound. */
|
|
117
134
|
readonly bins: readonly CalibrationBin[];
|
|
118
135
|
/**
|
|
119
|
-
* Lowest bin lowerBound
|
|
120
|
-
*
|
|
121
|
-
* undefined when no bin qualifies — the data does not
|
|
122
|
-
*
|
|
123
|
-
* `
|
|
136
|
+
* EXPERIMENTAL. Lowest bin lowerBound of the top-contiguous run of bins that
|
|
137
|
+
* each have ≥ minBinSamples samples and an `accuracyLowerBound` ≥
|
|
138
|
+
* `targetAccuracy`. undefined when no bin qualifies — the data does not
|
|
139
|
+
* support an empirical threshold at that target. Each contributing bin reports
|
|
140
|
+
* its `sampleCount` and `accuracyLowerBound`. This is a descriptive summary,
|
|
141
|
+
* not a validated auto-accept gate; calibration never sets any policy.
|
|
124
142
|
*/
|
|
125
143
|
readonly suggestedThreshold: number | undefined;
|
|
126
144
|
}
|
|
@@ -145,20 +163,29 @@ export interface CalibrationMetrics {
|
|
|
145
163
|
/**
|
|
146
164
|
* Derives extractor/field confidence calibration from review outcomes.
|
|
147
165
|
*
|
|
166
|
+
* EXPERIMENTAL — see the module note.
|
|
167
|
+
*
|
|
148
168
|
* Each reviewed candidate set contributes one labeled sample: the confidence of
|
|
149
|
-
* the SYSTEM-proposed candidate
|
|
150
|
-
*
|
|
151
|
-
*
|
|
169
|
+
* the SYSTEM-proposed candidate as the prediction, and whether the human review
|
|
170
|
+
* affirmed that proposed value as the label.
|
|
171
|
+
*
|
|
172
|
+
* The proposed candidate is identified by its producer role, never by
|
|
173
|
+
* `CandidateSet.selectedCandidateId` (which records the reviewer's pick on the
|
|
174
|
+
* builder and canonical review paths, #279): the single candidate whose
|
|
175
|
+
* `metadata.candidateRole` or `metadata.role` is `"proposed"`, or the only
|
|
176
|
+
* candidate of a one-candidate set (whatever its role). A sample is skipped (and counted
|
|
177
|
+
* in `skippedCount`) when it carries no human label or no prediction:
|
|
152
178
|
*
|
|
153
179
|
* - status "proposed" (not yet reviewed);
|
|
154
180
|
* - resolution "could_not_confirm" (no human correctness label);
|
|
155
181
|
* - a machine auto-accept, unless `includeAutoAccepted` is set;
|
|
156
|
-
* -
|
|
157
|
-
* missing or non-finite
|
|
182
|
+
* - the proposed candidate cannot be determined (several candidates without
|
|
183
|
+
* exactly one `"proposed"` role), or its confidence is missing or non-finite.
|
|
158
184
|
*
|
|
159
185
|
* A sample is "correct" when the outcome status is verified/assumed AND the
|
|
160
|
-
* reviewer
|
|
161
|
-
*
|
|
186
|
+
* reviewer's pick (`ReviewOutcome.candidateId`, else `selectedCandidateId`) is
|
|
187
|
+
* the proposed candidate; "incorrect" when the status is rejected or the
|
|
188
|
+
* reviewer picked a different candidate (for example, kept the current value).
|
|
162
189
|
*/
|
|
163
190
|
export declare function deriveCalibration(input: CalibrationInput, options?: DeriveCalibrationOptions): CalibrationMetrics;
|
|
164
191
|
/** A single projected calibration claim triple. */
|
package/dist/src/calibration.js
CHANGED
|
@@ -9,10 +9,16 @@
|
|
|
9
9
|
* from extractor X was affirmed 17/20 times" — plus a calibration gap and an
|
|
10
10
|
* empirically-grounded auto-accept threshold suggestion.
|
|
11
11
|
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
12
|
+
* EXPERIMENTAL. The curve is a descriptive summary of past review outcomes, not
|
|
13
|
+
* a validated probability model. `suggestedThreshold` is withheld unless every
|
|
14
|
+
* contributing bin clears a sample floor AND a one-sided 95% Wilson lower bound
|
|
15
|
+
* on its accuracy clears the target (#279); even then it is not an auto-accept
|
|
16
|
+
* gate — do not wire it into `autoAcceptMinConfidence` without your own
|
|
17
|
+
* evaluation.
|
|
18
|
+
*
|
|
19
|
+
* ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration never decides a claim
|
|
20
|
+
* or mutates a status. Projected claims carry status "proposed", exactly like
|
|
21
|
+
* every other producer proposal.
|
|
16
22
|
*
|
|
17
23
|
* Machine auto-accepts are EXCLUDED by default: an auto-accepted outcome is the
|
|
18
24
|
* threshold accepting its own guess, so counting it as a "correct" label would
|
|
@@ -24,27 +30,39 @@
|
|
|
24
30
|
import { AUTO_ACCEPT_ACTOR } from "./producer-profile.js";
|
|
25
31
|
const DEFAULT_BIN_COUNT = 10;
|
|
26
32
|
const DEFAULT_TARGET_ACCURACY = 0.95;
|
|
27
|
-
|
|
33
|
+
/** Default per-bin sample floor for `suggestedThreshold` (#279; was 1). */
|
|
34
|
+
const DEFAULT_MIN_BIN_SAMPLES = 30;
|
|
35
|
+
/** z for a one-sided 95% lower confidence bound. */
|
|
36
|
+
const ONE_SIDED_95_Z = 1.6448536269514722;
|
|
28
37
|
// ---------------------------------------------------------------------------
|
|
29
38
|
// deriveCalibration
|
|
30
39
|
// ---------------------------------------------------------------------------
|
|
31
40
|
/**
|
|
32
41
|
* Derives extractor/field confidence calibration from review outcomes.
|
|
33
42
|
*
|
|
43
|
+
* EXPERIMENTAL — see the module note.
|
|
44
|
+
*
|
|
34
45
|
* Each reviewed candidate set contributes one labeled sample: the confidence of
|
|
35
|
-
* the SYSTEM-proposed candidate
|
|
36
|
-
*
|
|
37
|
-
*
|
|
46
|
+
* the SYSTEM-proposed candidate as the prediction, and whether the human review
|
|
47
|
+
* affirmed that proposed value as the label.
|
|
48
|
+
*
|
|
49
|
+
* The proposed candidate is identified by its producer role, never by
|
|
50
|
+
* `CandidateSet.selectedCandidateId` (which records the reviewer's pick on the
|
|
51
|
+
* builder and canonical review paths, #279): the single candidate whose
|
|
52
|
+
* `metadata.candidateRole` or `metadata.role` is `"proposed"`, or the only
|
|
53
|
+
* candidate of a one-candidate set (whatever its role). A sample is skipped (and counted
|
|
54
|
+
* in `skippedCount`) when it carries no human label or no prediction:
|
|
38
55
|
*
|
|
39
56
|
* - status "proposed" (not yet reviewed);
|
|
40
57
|
* - resolution "could_not_confirm" (no human correctness label);
|
|
41
58
|
* - a machine auto-accept, unless `includeAutoAccepted` is set;
|
|
42
|
-
* -
|
|
43
|
-
* missing or non-finite
|
|
59
|
+
* - the proposed candidate cannot be determined (several candidates without
|
|
60
|
+
* exactly one `"proposed"` role), or its confidence is missing or non-finite.
|
|
44
61
|
*
|
|
45
62
|
* A sample is "correct" when the outcome status is verified/assumed AND the
|
|
46
|
-
* reviewer
|
|
47
|
-
*
|
|
63
|
+
* reviewer's pick (`ReviewOutcome.candidateId`, else `selectedCandidateId`) is
|
|
64
|
+
* the proposed candidate; "incorrect" when the status is rejected or the
|
|
65
|
+
* reviewer picked a different candidate (for example, kept the current value).
|
|
48
66
|
*/
|
|
49
67
|
export function deriveCalibration(input, options = {}) {
|
|
50
68
|
const binCount = normalizeBinCount(options.binCount);
|
|
@@ -52,11 +70,6 @@ export function deriveCalibration(input, options = {}) {
|
|
|
52
70
|
const minBinSamples = options.minBinSamples ?? DEFAULT_MIN_BIN_SAMPLES;
|
|
53
71
|
const includeAutoAccepted = options.includeAutoAccepted ?? false;
|
|
54
72
|
const candidateSetById = new Map(input.candidateSets.map((cs) => [cs.id, cs]));
|
|
55
|
-
const candidateById = new Map();
|
|
56
|
-
for (const cs of input.candidateSets) {
|
|
57
|
-
for (const c of cs.candidates)
|
|
58
|
-
candidateById.set(c.id, c);
|
|
59
|
-
}
|
|
60
73
|
const extractionById = new Map(input.extractions.map((e) => [e.id, e]));
|
|
61
74
|
if (options.windowDays !== undefined && options.now === undefined) {
|
|
62
75
|
throw new RangeError("deriveCalibration: `now` is required when `windowDays` is set (windowing is by reviewedAt).");
|
|
@@ -78,7 +91,7 @@ export function deriveCalibration(input, options = {}) {
|
|
|
78
91
|
continue;
|
|
79
92
|
}
|
|
80
93
|
}
|
|
81
|
-
const sample = toSample(outcome, candidateSetById,
|
|
94
|
+
const sample = toSample(outcome, candidateSetById, extractionById, includeAutoAccepted);
|
|
82
95
|
if (sample === undefined) {
|
|
83
96
|
skippedCount++;
|
|
84
97
|
continue;
|
|
@@ -129,7 +142,7 @@ export function deriveCalibration(input, options = {}) {
|
|
|
129
142
|
// ---------------------------------------------------------------------------
|
|
130
143
|
// Internal: sample construction
|
|
131
144
|
// ---------------------------------------------------------------------------
|
|
132
|
-
function toSample(outcome, candidateSetById,
|
|
145
|
+
function toSample(outcome, candidateSetById, extractionById, includeAutoAccepted) {
|
|
133
146
|
// No human label yet.
|
|
134
147
|
if (outcome.status === "proposed")
|
|
135
148
|
return undefined;
|
|
@@ -139,13 +152,11 @@ function toSample(outcome, candidateSetById, candidateById, extractionById, incl
|
|
|
139
152
|
const candidateSet = candidateSetById.get(outcome.candidateSetId);
|
|
140
153
|
if (candidateSet === undefined)
|
|
141
154
|
return undefined;
|
|
142
|
-
// The prediction is the SYSTEM-proposed candidate's confidence.
|
|
143
|
-
const
|
|
144
|
-
if (proposedId === undefined)
|
|
145
|
-
return undefined;
|
|
146
|
-
const proposed = candidateById.get(proposedId);
|
|
155
|
+
// The prediction is the SYSTEM-proposed candidate's confidence, found by role.
|
|
156
|
+
const proposed = proposedCandidateOf(candidateSet);
|
|
147
157
|
if (proposed === undefined)
|
|
148
158
|
return undefined;
|
|
159
|
+
const proposedId = proposed.id;
|
|
149
160
|
const extraction = proposed.extractionId ? extractionById.get(proposed.extractionId) : undefined;
|
|
150
161
|
const rawConfidence = proposed.confidence ?? extraction?.confidence;
|
|
151
162
|
if (rawConfidence === undefined || !Number.isFinite(rawConfidence))
|
|
@@ -159,10 +170,10 @@ function toSample(outcome, candidateSetById, candidateById, extractionById, incl
|
|
|
159
170
|
correct = false;
|
|
160
171
|
}
|
|
161
172
|
else {
|
|
162
|
-
// verified | assumed —
|
|
163
|
-
//
|
|
164
|
-
const
|
|
165
|
-
correct =
|
|
173
|
+
// verified | assumed — the proposed value stood only if the reviewer's pick
|
|
174
|
+
// is the proposed candidate.
|
|
175
|
+
const pickedId = outcome.candidateId ?? candidateSet.selectedCandidateId;
|
|
176
|
+
correct = pickedId === undefined || pickedId === proposedId;
|
|
166
177
|
}
|
|
167
178
|
return {
|
|
168
179
|
extractor,
|
|
@@ -174,6 +185,24 @@ function toSample(outcome, candidateSetById, candidateById, extractionById, incl
|
|
|
174
185
|
reviewedAt: outcome.reviewedAt,
|
|
175
186
|
};
|
|
176
187
|
}
|
|
188
|
+
/**
|
|
189
|
+
* The candidate whose confidence is the prediction. Returns undefined when it
|
|
190
|
+
* cannot be determined, so the outcome is skipped rather than labeled against
|
|
191
|
+
* the reviewer's own pick.
|
|
192
|
+
*
|
|
193
|
+
* A one-candidate set is sampled whatever the candidate's role ("computed",
|
|
194
|
+
* "source-version", "current", free-form roles, or none): with one candidate
|
|
195
|
+
* there is no pick to confuse with a proposal, and the review either affirmed
|
|
196
|
+
* or rejected that candidate's value. Only a multi-candidate set needs a
|
|
197
|
+
* single "proposed" role marker.
|
|
198
|
+
*/
|
|
199
|
+
function proposedCandidateOf(candidateSet) {
|
|
200
|
+
if (candidateSet.candidates.length === 1)
|
|
201
|
+
return candidateSet.candidates[0];
|
|
202
|
+
const roleOf = (c) => c.metadata?.candidateRole ?? c.metadata?.role;
|
|
203
|
+
const proposed = candidateSet.candidates.filter((c) => roleOf(c) === "proposed");
|
|
204
|
+
return proposed.length === 1 ? proposed[0] : undefined;
|
|
205
|
+
}
|
|
177
206
|
// ---------------------------------------------------------------------------
|
|
178
207
|
// Internal: group + bin computation
|
|
179
208
|
// ---------------------------------------------------------------------------
|
|
@@ -219,14 +248,24 @@ function computeBins(samples, binCount) {
|
|
|
219
248
|
correctCount: b.correct,
|
|
220
249
|
empiricalAccuracy: b.n > 0 ? round(b.correct / b.n) : undefined,
|
|
221
250
|
meanPredictedConfidence: b.n > 0 ? round(b.sum / b.n) : undefined,
|
|
251
|
+
...(b.n > 0 ? { accuracyLowerBound: round(wilsonLowerBound(b.correct, b.n)) } : {}),
|
|
222
252
|
}));
|
|
223
253
|
}
|
|
254
|
+
/** One-sided 95% Wilson score lower bound for `correct` successes out of `n`. */
|
|
255
|
+
function wilsonLowerBound(correct, n) {
|
|
256
|
+
const z = ONE_SIDED_95_Z;
|
|
257
|
+
const p = correct / n;
|
|
258
|
+
const z2 = z * z;
|
|
259
|
+
const centre = p + z2 / (2 * n);
|
|
260
|
+
const margin = z * Math.sqrt((p * (1 - p)) / n + z2 / (4 * n * n));
|
|
261
|
+
return Math.max(0, (centre - margin) / (1 + z2 / n));
|
|
262
|
+
}
|
|
224
263
|
/**
|
|
225
264
|
* The suggested threshold is the lowerBound of the lowest bin in the
|
|
226
|
-
* top-contiguous run of bins that each (a) have ≥ minBinSamples and (b)
|
|
227
|
-
*
|
|
228
|
-
* fails the
|
|
229
|
-
* undefined when
|
|
265
|
+
* top-contiguous run of bins that each (a) have ≥ minBinSamples and (b) have a
|
|
266
|
+
* one-sided 95% Wilson lower bound on accuracy ≥ targetAccuracy. Scanning from
|
|
267
|
+
* the highest bin down, a bin that fails the bound — or an under-sampled bin we
|
|
268
|
+
* cannot vouch for — ends the run. undefined when the top bin does not qualify.
|
|
230
269
|
*/
|
|
231
270
|
function computeSuggestedThreshold(bins, targetAccuracy, minBinSamples) {
|
|
232
271
|
let threshold;
|
|
@@ -234,7 +273,7 @@ function computeSuggestedThreshold(bins, targetAccuracy, minBinSamples) {
|
|
|
234
273
|
const bin = bins[i];
|
|
235
274
|
if (bin.sampleCount < minBinSamples)
|
|
236
275
|
break;
|
|
237
|
-
if (bin.
|
|
276
|
+
if (bin.sampleCount === 0 || wilsonLowerBound(bin.correctCount, bin.sampleCount) < targetAccuracy)
|
|
238
277
|
break;
|
|
239
278
|
threshold = bin.lowerBound;
|
|
240
279
|
}
|
|
@@ -5,7 +5,9 @@
|
|
|
5
5
|
* Routes:
|
|
6
6
|
* GET / HTML shell that mounts the workbench
|
|
7
7
|
* GET /api/session Current session state (snapshot + replayed events)
|
|
8
|
-
* POST /api/events Append review session events
|
|
8
|
+
* POST /api/events Append review session events to the stored log (same
|
|
9
|
+
* validation as MCP server), compare-and-swap on the
|
|
10
|
+
* revision the client last read
|
|
9
11
|
* GET /api/stream SSE stream: emits "update" events when the session file changes
|
|
10
12
|
* GET /api/health Health check
|
|
11
13
|
* GET /dist/* Compiled assets served from the dist tree (traversal-safe)
|