@kontourai/survey 2.5.0 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -1
- package/dist/examples/calibrated-auto-accept.d.ts +22 -15
- package/dist/examples/calibrated-auto-accept.js +40 -36
- package/dist/src/agent-utterance.d.ts +87 -11
- package/dist/src/agent-utterance.js +135 -44
- package/dist/src/calibration.d.ts +48 -21
- package/dist/src/calibration.js +72 -33
- package/dist/src/console/review-console-server.d.ts +3 -1
- package/dist/src/console/review-console-server.js +203 -50
- package/dist/src/extraction-envelope.d.ts +22 -0
- package/dist/src/extraction-envelope.js +25 -4
- package/dist/src/index.d.ts +9 -8
- package/dist/src/index.js +2 -2
- package/dist/src/inquiry-mapping.d.ts +15 -1
- package/dist/src/inquiry-mapping.js +10 -2
- package/dist/src/mcp/review-mcp.js +219 -279
- package/dist/src/producer-profile.d.ts +41 -2
- package/dist/src/producer-profile.js +29 -2
- package/dist/src/review-session-file.d.ts +64 -0
- package/dist/src/review-session-file.js +320 -0
- package/dist/src/review-workbench/edited-value.d.ts +70 -0
- package/dist/src/review-workbench/edited-value.js +147 -0
- package/dist/src/review-workbench/review-presentation.d.ts +44 -0
- package/dist/src/review-workbench/review-presentation.js +49 -0
- package/dist/src/review-workbench/review-queue-session.js +8 -1
- package/dist/src/review-workbench/review-session-replay.d.ts +35 -1
- package/dist/src/review-workbench/review-session-replay.js +77 -0
- package/dist/src/review-workbench/review-workbench.d.ts +8 -4
- package/dist/src/review-workbench/review-workbench.js +19 -7
- package/dist/src/review-workbench/server-review-session.d.ts +3 -1
- package/dist/src/review-workbench/server-review-session.js +1 -0
- package/dist/src/reviewed-candidate-resolution.js +13 -7
- package/dist/src/schema-mapping.d.ts +23 -0
- package/dist/src/schema-mapping.js +30 -20
- package/dist/src/to-surface.d.ts +30 -6
- package/dist/src/to-surface.js +196 -18
- package/dist/src/types.d.ts +39 -0
- package/package.json +8 -4
|
@@ -9,10 +9,16 @@
|
|
|
9
9
|
* from extractor X was affirmed 17/20 times" — plus a calibration gap and an
|
|
10
10
|
* empirically-grounded auto-accept threshold suggestion.
|
|
11
11
|
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
12
|
+
* EXPERIMENTAL. The curve is a descriptive summary of past review outcomes, not
|
|
13
|
+
* a validated probability model. `suggestedThreshold` is withheld unless every
|
|
14
|
+
* contributing bin clears a sample floor AND a one-sided 95% Wilson lower bound
|
|
15
|
+
* on its accuracy clears the target (#279); even then it is not an auto-accept
|
|
16
|
+
* gate — do not wire it into `autoAcceptMinConfidence` without your own
|
|
17
|
+
* evaluation.
|
|
18
|
+
*
|
|
19
|
+
* ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration never decides a claim
|
|
20
|
+
* or mutates a status. Projected claims carry status "proposed", exactly like
|
|
21
|
+
* every other producer proposal.
|
|
16
22
|
*
|
|
17
23
|
* Machine auto-accepts are EXCLUDED by default: an auto-accepted outcome is the
|
|
18
24
|
* threshold accepting its own guess, so counting it as a "correct" label would
|
|
@@ -49,15 +55,19 @@ export interface DeriveCalibrationOptions {
|
|
|
49
55
|
/** Number of equal-width confidence bins over [0,1]. Default 10 (deciles). */
|
|
50
56
|
readonly binCount?: number;
|
|
51
57
|
/**
|
|
52
|
-
* The
|
|
53
|
-
*
|
|
54
|
-
*
|
|
58
|
+
* The accuracy the suggested threshold must clear. Default 0.95. A
|
|
59
|
+
* `suggestedThreshold` is the lowest bin lower-bound at/above which every bin
|
|
60
|
+
* has ≥ `minBinSamples` samples and a one-sided 95% Wilson lower confidence
|
|
61
|
+
* bound on its accuracy (`CalibrationBin.accuracyLowerBound`) that meets this
|
|
62
|
+
* target — the point estimate alone is not enough.
|
|
55
63
|
*/
|
|
56
64
|
readonly targetAccuracy?: number;
|
|
57
65
|
/**
|
|
58
66
|
* A bin needs at least this many samples to count toward `suggestedThreshold`
|
|
59
|
-
* (
|
|
60
|
-
*
|
|
67
|
+
* (an under-sampled bin ends the qualifying run). Default
|
|
68
|
+
* {@link DEFAULT_MIN_BIN_SAMPLES} (30). The Wilson bound applies on top of this
|
|
69
|
+
* floor, so an all-affirmed bin needs more samples than the floor when the
|
|
70
|
+
* target is high (≈52 all-affirmed samples for the default 0.95 target).
|
|
61
71
|
*/
|
|
62
72
|
readonly minBinSamples?: number;
|
|
63
73
|
/**
|
|
@@ -95,6 +105,13 @@ export interface CalibrationBin {
|
|
|
95
105
|
readonly empiricalAccuracy: number | undefined;
|
|
96
106
|
/** Mean predicted confidence of samples in the bin; undefined when empty. */
|
|
97
107
|
readonly meanPredictedConfidence: number | undefined;
|
|
108
|
+
/**
|
|
109
|
+
* One-sided 95% Wilson lower confidence bound on the bin's accuracy; undefined
|
|
110
|
+
* when the bin is empty. `suggestedThreshold` requires this bound, not the
|
|
111
|
+
* point estimate, to meet `targetAccuracy`. Optional in the type so metrics
|
|
112
|
+
* built by hand before this field existed still type-check.
|
|
113
|
+
*/
|
|
114
|
+
readonly accuracyLowerBound?: number;
|
|
98
115
|
}
|
|
99
116
|
/** A calibration rollup for one extractor (and optionally one field). */
|
|
100
117
|
export interface CalibrationGroup {
|
|
@@ -116,11 +133,12 @@ export interface CalibrationGroup {
|
|
|
116
133
|
/** Per-bin empirical accuracy, ascending by lowerBound. */
|
|
117
134
|
readonly bins: readonly CalibrationBin[];
|
|
118
135
|
/**
|
|
119
|
-
* Lowest bin lowerBound
|
|
120
|
-
*
|
|
121
|
-
* undefined when no bin qualifies — the data does not
|
|
122
|
-
*
|
|
123
|
-
* `
|
|
136
|
+
* EXPERIMENTAL. Lowest bin lowerBound of the top-contiguous run of bins that
|
|
137
|
+
* each have ≥ minBinSamples samples and an `accuracyLowerBound` ≥
|
|
138
|
+
* `targetAccuracy`. undefined when no bin qualifies — the data does not
|
|
139
|
+
* support an empirical threshold at that target. Each contributing bin reports
|
|
140
|
+
* its `sampleCount` and `accuracyLowerBound`. This is a descriptive summary,
|
|
141
|
+
* not a validated auto-accept gate; calibration never sets any policy.
|
|
124
142
|
*/
|
|
125
143
|
readonly suggestedThreshold: number | undefined;
|
|
126
144
|
}
|
|
@@ -145,20 +163,29 @@ export interface CalibrationMetrics {
|
|
|
145
163
|
/**
|
|
146
164
|
* Derives extractor/field confidence calibration from review outcomes.
|
|
147
165
|
*
|
|
166
|
+
* EXPERIMENTAL — see the module note.
|
|
167
|
+
*
|
|
148
168
|
* Each reviewed candidate set contributes one labeled sample: the confidence of
|
|
149
|
-
* the SYSTEM-proposed candidate
|
|
150
|
-
*
|
|
151
|
-
*
|
|
169
|
+
* the SYSTEM-proposed candidate as the prediction, and whether the human review
|
|
170
|
+
* affirmed that proposed value as the label.
|
|
171
|
+
*
|
|
172
|
+
* The proposed candidate is identified by its producer role, never by
|
|
173
|
+
* `CandidateSet.selectedCandidateId` (which records the reviewer's pick on the
|
|
174
|
+
* builder and canonical review paths, #279): the single candidate whose
|
|
175
|
+
* `metadata.candidateRole` or `metadata.role` is `"proposed"`, or the only
|
|
176
|
+
* candidate of a one-candidate set (whatever its role). A sample is skipped (and counted
|
|
177
|
+
* in `skippedCount`) when it carries no human label or no prediction:
|
|
152
178
|
*
|
|
153
179
|
* - status "proposed" (not yet reviewed);
|
|
154
180
|
* - resolution "could_not_confirm" (no human correctness label);
|
|
155
181
|
* - a machine auto-accept, unless `includeAutoAccepted` is set;
|
|
156
|
-
* -
|
|
157
|
-
* missing or non-finite
|
|
182
|
+
* - the proposed candidate cannot be determined (several candidates without
|
|
183
|
+
* exactly one `"proposed"` role), or its confidence is missing or non-finite.
|
|
158
184
|
*
|
|
159
185
|
* A sample is "correct" when the outcome status is verified/assumed AND the
|
|
160
|
-
* reviewer
|
|
161
|
-
*
|
|
186
|
+
* reviewer's pick (`ReviewOutcome.candidateId`, else `selectedCandidateId`) is
|
|
187
|
+
* the proposed candidate; "incorrect" when the status is rejected or the
|
|
188
|
+
* reviewer picked a different candidate (for example, kept the current value).
|
|
162
189
|
*/
|
|
163
190
|
export declare function deriveCalibration(input: CalibrationInput, options?: DeriveCalibrationOptions): CalibrationMetrics;
|
|
164
191
|
/** A single projected calibration claim triple. */
|
package/dist/src/calibration.js
CHANGED
|
@@ -9,10 +9,16 @@
|
|
|
9
9
|
* from extractor X was affirmed 17/20 times" — plus a calibration gap and an
|
|
10
10
|
* empirically-grounded auto-accept threshold suggestion.
|
|
11
11
|
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
12
|
+
* EXPERIMENTAL. The curve is a descriptive summary of past review outcomes, not
|
|
13
|
+
* a validated probability model. `suggestedThreshold` is withheld unless every
|
|
14
|
+
* contributing bin clears a sample floor AND a one-sided 95% Wilson lower bound
|
|
15
|
+
* on its accuracy clears the target (#279); even then it is not an auto-accept
|
|
16
|
+
* gate — do not wire it into `autoAcceptMinConfidence` without your own
|
|
17
|
+
* evaluation.
|
|
18
|
+
*
|
|
19
|
+
* ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration never decides a claim
|
|
20
|
+
* or mutates a status. Projected claims carry status "proposed", exactly like
|
|
21
|
+
* every other producer proposal.
|
|
16
22
|
*
|
|
17
23
|
* Machine auto-accepts are EXCLUDED by default: an auto-accepted outcome is the
|
|
18
24
|
* threshold accepting its own guess, so counting it as a "correct" label would
|
|
@@ -24,27 +30,39 @@
|
|
|
24
30
|
import { AUTO_ACCEPT_ACTOR } from "./producer-profile.js";
|
|
25
31
|
const DEFAULT_BIN_COUNT = 10;
|
|
26
32
|
const DEFAULT_TARGET_ACCURACY = 0.95;
|
|
27
|
-
|
|
33
|
+
/** Default per-bin sample floor for `suggestedThreshold` (#279; was 1). */
|
|
34
|
+
const DEFAULT_MIN_BIN_SAMPLES = 30;
|
|
35
|
+
/** z for a one-sided 95% lower confidence bound. */
|
|
36
|
+
const ONE_SIDED_95_Z = 1.6448536269514722;
|
|
28
37
|
// ---------------------------------------------------------------------------
|
|
29
38
|
// deriveCalibration
|
|
30
39
|
// ---------------------------------------------------------------------------
|
|
31
40
|
/**
|
|
32
41
|
* Derives extractor/field confidence calibration from review outcomes.
|
|
33
42
|
*
|
|
43
|
+
* EXPERIMENTAL — see the module note.
|
|
44
|
+
*
|
|
34
45
|
* Each reviewed candidate set contributes one labeled sample: the confidence of
|
|
35
|
-
* the SYSTEM-proposed candidate
|
|
36
|
-
*
|
|
37
|
-
*
|
|
46
|
+
* the SYSTEM-proposed candidate as the prediction, and whether the human review
|
|
47
|
+
* affirmed that proposed value as the label.
|
|
48
|
+
*
|
|
49
|
+
* The proposed candidate is identified by its producer role, never by
|
|
50
|
+
* `CandidateSet.selectedCandidateId` (which records the reviewer's pick on the
|
|
51
|
+
* builder and canonical review paths, #279): the single candidate whose
|
|
52
|
+
* `metadata.candidateRole` or `metadata.role` is `"proposed"`, or the only
|
|
53
|
+
* candidate of a one-candidate set (whatever its role). A sample is skipped (and counted
|
|
54
|
+
* in `skippedCount`) when it carries no human label or no prediction:
|
|
38
55
|
*
|
|
39
56
|
* - status "proposed" (not yet reviewed);
|
|
40
57
|
* - resolution "could_not_confirm" (no human correctness label);
|
|
41
58
|
* - a machine auto-accept, unless `includeAutoAccepted` is set;
|
|
42
|
-
* -
|
|
43
|
-
* missing or non-finite
|
|
59
|
+
* - the proposed candidate cannot be determined (several candidates without
|
|
60
|
+
* exactly one `"proposed"` role), or its confidence is missing or non-finite.
|
|
44
61
|
*
|
|
45
62
|
* A sample is "correct" when the outcome status is verified/assumed AND the
|
|
46
|
-
* reviewer
|
|
47
|
-
*
|
|
63
|
+
* reviewer's pick (`ReviewOutcome.candidateId`, else `selectedCandidateId`) is
|
|
64
|
+
* the proposed candidate; "incorrect" when the status is rejected or the
|
|
65
|
+
* reviewer picked a different candidate (for example, kept the current value).
|
|
48
66
|
*/
|
|
49
67
|
export function deriveCalibration(input, options = {}) {
|
|
50
68
|
const binCount = normalizeBinCount(options.binCount);
|
|
@@ -52,11 +70,6 @@ export function deriveCalibration(input, options = {}) {
|
|
|
52
70
|
const minBinSamples = options.minBinSamples ?? DEFAULT_MIN_BIN_SAMPLES;
|
|
53
71
|
const includeAutoAccepted = options.includeAutoAccepted ?? false;
|
|
54
72
|
const candidateSetById = new Map(input.candidateSets.map((cs) => [cs.id, cs]));
|
|
55
|
-
const candidateById = new Map();
|
|
56
|
-
for (const cs of input.candidateSets) {
|
|
57
|
-
for (const c of cs.candidates)
|
|
58
|
-
candidateById.set(c.id, c);
|
|
59
|
-
}
|
|
60
73
|
const extractionById = new Map(input.extractions.map((e) => [e.id, e]));
|
|
61
74
|
if (options.windowDays !== undefined && options.now === undefined) {
|
|
62
75
|
throw new RangeError("deriveCalibration: `now` is required when `windowDays` is set (windowing is by reviewedAt).");
|
|
@@ -78,7 +91,7 @@ export function deriveCalibration(input, options = {}) {
|
|
|
78
91
|
continue;
|
|
79
92
|
}
|
|
80
93
|
}
|
|
81
|
-
const sample = toSample(outcome, candidateSetById,
|
|
94
|
+
const sample = toSample(outcome, candidateSetById, extractionById, includeAutoAccepted);
|
|
82
95
|
if (sample === undefined) {
|
|
83
96
|
skippedCount++;
|
|
84
97
|
continue;
|
|
@@ -129,7 +142,7 @@ export function deriveCalibration(input, options = {}) {
|
|
|
129
142
|
// ---------------------------------------------------------------------------
|
|
130
143
|
// Internal: sample construction
|
|
131
144
|
// ---------------------------------------------------------------------------
|
|
132
|
-
function toSample(outcome, candidateSetById,
|
|
145
|
+
function toSample(outcome, candidateSetById, extractionById, includeAutoAccepted) {
|
|
133
146
|
// No human label yet.
|
|
134
147
|
if (outcome.status === "proposed")
|
|
135
148
|
return undefined;
|
|
@@ -139,13 +152,11 @@ function toSample(outcome, candidateSetById, candidateById, extractionById, incl
|
|
|
139
152
|
const candidateSet = candidateSetById.get(outcome.candidateSetId);
|
|
140
153
|
if (candidateSet === undefined)
|
|
141
154
|
return undefined;
|
|
142
|
-
// The prediction is the SYSTEM-proposed candidate's confidence.
|
|
143
|
-
const
|
|
144
|
-
if (proposedId === undefined)
|
|
145
|
-
return undefined;
|
|
146
|
-
const proposed = candidateById.get(proposedId);
|
|
155
|
+
// The prediction is the SYSTEM-proposed candidate's confidence, found by role.
|
|
156
|
+
const proposed = proposedCandidateOf(candidateSet);
|
|
147
157
|
if (proposed === undefined)
|
|
148
158
|
return undefined;
|
|
159
|
+
const proposedId = proposed.id;
|
|
149
160
|
const extraction = proposed.extractionId ? extractionById.get(proposed.extractionId) : undefined;
|
|
150
161
|
const rawConfidence = proposed.confidence ?? extraction?.confidence;
|
|
151
162
|
if (rawConfidence === undefined || !Number.isFinite(rawConfidence))
|
|
@@ -159,10 +170,10 @@ function toSample(outcome, candidateSetById, candidateById, extractionById, incl
|
|
|
159
170
|
correct = false;
|
|
160
171
|
}
|
|
161
172
|
else {
|
|
162
|
-
// verified | assumed —
|
|
163
|
-
//
|
|
164
|
-
const
|
|
165
|
-
correct =
|
|
173
|
+
// verified | assumed — the proposed value stood only if the reviewer's pick
|
|
174
|
+
// is the proposed candidate.
|
|
175
|
+
const pickedId = outcome.candidateId ?? candidateSet.selectedCandidateId;
|
|
176
|
+
correct = pickedId === undefined || pickedId === proposedId;
|
|
166
177
|
}
|
|
167
178
|
return {
|
|
168
179
|
extractor,
|
|
@@ -174,6 +185,24 @@ function toSample(outcome, candidateSetById, candidateById, extractionById, incl
|
|
|
174
185
|
reviewedAt: outcome.reviewedAt,
|
|
175
186
|
};
|
|
176
187
|
}
|
|
188
|
+
/**
|
|
189
|
+
* The candidate whose confidence is the prediction. Returns undefined when it
|
|
190
|
+
* cannot be determined, so the outcome is skipped rather than labeled against
|
|
191
|
+
* the reviewer's own pick.
|
|
192
|
+
*
|
|
193
|
+
* A one-candidate set is sampled whatever the candidate's role ("computed",
|
|
194
|
+
* "source-version", "current", free-form roles, or none): with one candidate
|
|
195
|
+
* there is no pick to confuse with a proposal, and the review either affirmed
|
|
196
|
+
* or rejected that candidate's value. Only a multi-candidate set needs a
|
|
197
|
+
* single "proposed" role marker.
|
|
198
|
+
*/
|
|
199
|
+
function proposedCandidateOf(candidateSet) {
|
|
200
|
+
if (candidateSet.candidates.length === 1)
|
|
201
|
+
return candidateSet.candidates[0];
|
|
202
|
+
const roleOf = (c) => c.metadata?.candidateRole ?? c.metadata?.role;
|
|
203
|
+
const proposed = candidateSet.candidates.filter((c) => roleOf(c) === "proposed");
|
|
204
|
+
return proposed.length === 1 ? proposed[0] : undefined;
|
|
205
|
+
}
|
|
177
206
|
// ---------------------------------------------------------------------------
|
|
178
207
|
// Internal: group + bin computation
|
|
179
208
|
// ---------------------------------------------------------------------------
|
|
@@ -219,14 +248,24 @@ function computeBins(samples, binCount) {
|
|
|
219
248
|
correctCount: b.correct,
|
|
220
249
|
empiricalAccuracy: b.n > 0 ? round(b.correct / b.n) : undefined,
|
|
221
250
|
meanPredictedConfidence: b.n > 0 ? round(b.sum / b.n) : undefined,
|
|
251
|
+
...(b.n > 0 ? { accuracyLowerBound: round(wilsonLowerBound(b.correct, b.n)) } : {}),
|
|
222
252
|
}));
|
|
223
253
|
}
|
|
254
|
+
/** One-sided 95% Wilson score lower bound for `correct` successes out of `n`. */
|
|
255
|
+
function wilsonLowerBound(correct, n) {
|
|
256
|
+
const z = ONE_SIDED_95_Z;
|
|
257
|
+
const p = correct / n;
|
|
258
|
+
const z2 = z * z;
|
|
259
|
+
const centre = p + z2 / (2 * n);
|
|
260
|
+
const margin = z * Math.sqrt((p * (1 - p)) / n + z2 / (4 * n * n));
|
|
261
|
+
return Math.max(0, (centre - margin) / (1 + z2 / n));
|
|
262
|
+
}
|
|
224
263
|
/**
|
|
225
264
|
* The suggested threshold is the lowerBound of the lowest bin in the
|
|
226
|
-
* top-contiguous run of bins that each (a) have ≥ minBinSamples and (b)
|
|
227
|
-
*
|
|
228
|
-
* fails the
|
|
229
|
-
* undefined when
|
|
265
|
+
* top-contiguous run of bins that each (a) have ≥ minBinSamples and (b) have a
|
|
266
|
+
* one-sided 95% Wilson lower bound on accuracy ≥ targetAccuracy. Scanning from
|
|
267
|
+
* the highest bin down, a bin that fails the bound — or an under-sampled bin we
|
|
268
|
+
* cannot vouch for — ends the run. undefined when the top bin does not qualify.
|
|
230
269
|
*/
|
|
231
270
|
function computeSuggestedThreshold(bins, targetAccuracy, minBinSamples) {
|
|
232
271
|
let threshold;
|
|
@@ -234,7 +273,7 @@ function computeSuggestedThreshold(bins, targetAccuracy, minBinSamples) {
|
|
|
234
273
|
const bin = bins[i];
|
|
235
274
|
if (bin.sampleCount < minBinSamples)
|
|
236
275
|
break;
|
|
237
|
-
if (bin.
|
|
276
|
+
if (bin.sampleCount === 0 || wilsonLowerBound(bin.correctCount, bin.sampleCount) < targetAccuracy)
|
|
238
277
|
break;
|
|
239
278
|
threshold = bin.lowerBound;
|
|
240
279
|
}
|
|
@@ -5,7 +5,9 @@
|
|
|
5
5
|
* Routes:
|
|
6
6
|
* GET / HTML shell that mounts the workbench
|
|
7
7
|
* GET /api/session Current session state (snapshot + replayed events)
|
|
8
|
-
* POST /api/events Append review session events
|
|
8
|
+
* POST /api/events Append review session events to the stored log (same
|
|
9
|
+
* validation as MCP server), compare-and-swap on the
|
|
10
|
+
* revision the client last read
|
|
9
11
|
* GET /api/stream SSE stream: emits "update" events when the session file changes
|
|
10
12
|
* GET /api/health Health check
|
|
11
13
|
* GET /dist/* Compiled assets served from the dist tree (traversal-safe)
|