@kontourai/survey 2.5.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/README.md +6 -1
  2. package/dist/examples/calibrated-auto-accept.d.ts +22 -15
  3. package/dist/examples/calibrated-auto-accept.js +40 -36
  4. package/dist/src/agent-utterance.d.ts +87 -11
  5. package/dist/src/agent-utterance.js +135 -44
  6. package/dist/src/calibration.d.ts +48 -21
  7. package/dist/src/calibration.js +72 -33
  8. package/dist/src/console/review-console-server.d.ts +3 -1
  9. package/dist/src/console/review-console-server.js +203 -50
  10. package/dist/src/extraction-envelope.d.ts +22 -0
  11. package/dist/src/extraction-envelope.js +25 -4
  12. package/dist/src/index.d.ts +9 -8
  13. package/dist/src/index.js +2 -2
  14. package/dist/src/inquiry-mapping.d.ts +15 -1
  15. package/dist/src/inquiry-mapping.js +10 -2
  16. package/dist/src/mcp/review-mcp.js +219 -279
  17. package/dist/src/producer-profile.d.ts +41 -2
  18. package/dist/src/producer-profile.js +29 -2
  19. package/dist/src/review-session-file.d.ts +64 -0
  20. package/dist/src/review-session-file.js +320 -0
  21. package/dist/src/review-workbench/edited-value.d.ts +70 -0
  22. package/dist/src/review-workbench/edited-value.js +147 -0
  23. package/dist/src/review-workbench/review-presentation.d.ts +44 -0
  24. package/dist/src/review-workbench/review-presentation.js +49 -0
  25. package/dist/src/review-workbench/review-queue-session.js +8 -1
  26. package/dist/src/review-workbench/review-session-replay.d.ts +35 -1
  27. package/dist/src/review-workbench/review-session-replay.js +77 -0
  28. package/dist/src/review-workbench/review-workbench.d.ts +8 -4
  29. package/dist/src/review-workbench/review-workbench.js +19 -7
  30. package/dist/src/review-workbench/server-review-session.d.ts +3 -1
  31. package/dist/src/review-workbench/server-review-session.js +1 -0
  32. package/dist/src/reviewed-candidate-resolution.js +13 -7
  33. package/dist/src/schema-mapping.d.ts +23 -0
  34. package/dist/src/schema-mapping.js +30 -20
  35. package/dist/src/to-surface.d.ts +30 -6
  36. package/dist/src/to-surface.js +196 -18
  37. package/dist/src/types.d.ts +39 -0
  38. package/package.json +8 -4
@@ -9,10 +9,16 @@
9
9
  * from extractor X was affirmed 17/20 times" — plus a calibration gap and an
10
10
  * empirically-grounded auto-accept threshold suggestion.
11
11
  *
12
- * ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration INFORMS policy — the
13
- * suggested threshold an operator MAY wire into `autoAcceptMinConfidence` — it
14
- * never decides a claim or mutates a status. Projected claims carry status
15
- * "proposed", exactly like every other producer proposal.
12
+ * EXPERIMENTAL. The curve is a descriptive summary of past review outcomes, not
13
+ * a validated probability model. `suggestedThreshold` is withheld unless every
14
+ * contributing bin clears a sample floor AND a one-sided 95% Wilson lower bound
15
+ * on its accuracy clears the target (#279); even then it is not an auto-accept
16
+ * gate — do not wire it into `autoAcceptMinConfidence` without your own
17
+ * evaluation.
18
+ *
19
+ * ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration never decides a claim
20
+ * or mutates a status. Projected claims carry status "proposed", exactly like
21
+ * every other producer proposal.
16
22
  *
17
23
  * Machine auto-accepts are EXCLUDED by default: an auto-accepted outcome is the
18
24
  * threshold accepting its own guess, so counting it as a "correct" label would
@@ -49,15 +55,19 @@ export interface DeriveCalibrationOptions {
49
55
  /** Number of equal-width confidence bins over [0,1]. Default 10 (deciles). */
50
56
  readonly binCount?: number;
51
57
  /**
52
- * The empirical accuracy the suggested threshold must clear. Default 0.95.
53
- * A `suggestedThreshold` is the lowest bin lower-bound at/above which every
54
- * populated bin's empirical accuracy meets this target.
58
+ * The accuracy the suggested threshold must clear. Default 0.95. A
59
+ * `suggestedThreshold` is the lowest bin lower-bound at/above which every bin
60
+ * has ≥ `minBinSamples` samples and a one-sided 95% Wilson lower confidence
61
+ * bound on its accuracy (`CalibrationBin.accuracyLowerBound`) that meets this
62
+ * target — the point estimate alone is not enough.
55
63
  */
56
64
  readonly targetAccuracy?: number;
57
65
  /**
58
66
  * A bin needs at least this many samples to count toward `suggestedThreshold`
59
- * (both to qualify and to disqualify). Default 1. Raise it to avoid grounding
60
- * a threshold on a bin with too little evidence.
67
+ * (an under-sampled bin ends the qualifying run). Default
68
+ * {@link DEFAULT_MIN_BIN_SAMPLES} (30). The Wilson bound applies on top of this
69
+ * floor, so an all-affirmed bin needs more samples than the floor when the
70
+ * target is high (≈52 all-affirmed samples for the default 0.95 target).
61
71
  */
62
72
  readonly minBinSamples?: number;
63
73
  /**
@@ -95,6 +105,13 @@ export interface CalibrationBin {
95
105
  readonly empiricalAccuracy: number | undefined;
96
106
  /** Mean predicted confidence of samples in the bin; undefined when empty. */
97
107
  readonly meanPredictedConfidence: number | undefined;
108
+ /**
109
+ * One-sided 95% Wilson lower confidence bound on the bin's accuracy; undefined
110
+ * when the bin is empty. `suggestedThreshold` requires this bound, not the
111
+ * point estimate, to meet `targetAccuracy`. Optional in the type so metrics
112
+ * built by hand before this field existed still type-check.
113
+ */
114
+ readonly accuracyLowerBound?: number;
98
115
  }
99
116
  /** A calibration rollup for one extractor (and optionally one field). */
100
117
  export interface CalibrationGroup {
@@ -116,11 +133,12 @@ export interface CalibrationGroup {
116
133
  /** Per-bin empirical accuracy, ascending by lowerBound. */
117
134
  readonly bins: readonly CalibrationBin[];
118
135
  /**
119
- * Lowest bin lowerBound at/above which every populated bin (≥ minBinSamples)
120
- * meets `targetAccuracy`, scanning the top-contiguous run of qualifying bins.
121
- * undefined when no bin qualifies — the data does not yet support an empirical
122
- * auto-accept threshold at that target. ADVISORY: an operator wires this into
123
- * `autoAcceptMinConfidence`; calibration never sets it.
136
+ * EXPERIMENTAL. Lowest bin lowerBound of the top-contiguous run of bins that
137
+ * each have ≥ minBinSamples samples and an `accuracyLowerBound` ≥
138
+ * `targetAccuracy`. undefined when no bin qualifies — the data does not
139
+ * support an empirical threshold at that target. Each contributing bin reports
140
+ * its `sampleCount` and `accuracyLowerBound`. This is a descriptive summary,
141
+ * not a validated auto-accept gate; calibration never sets any policy.
124
142
  */
125
143
  readonly suggestedThreshold: number | undefined;
126
144
  }
@@ -145,20 +163,29 @@ export interface CalibrationMetrics {
145
163
  /**
146
164
  * Derives extractor/field confidence calibration from review outcomes.
147
165
  *
166
+ * EXPERIMENTAL — see the module note.
167
+ *
148
168
  * Each reviewed candidate set contributes one labeled sample: the confidence of
149
- * the SYSTEM-proposed candidate (`CandidateSet.selectedCandidateId`) as the
150
- * prediction, and whether the human review affirmed that proposed value as the
151
- * label. A sample is skipped when it carries no human label or no prediction:
169
+ * the SYSTEM-proposed candidate as the prediction, and whether the human review
170
+ * affirmed that proposed value as the label.
171
+ *
172
+ * The proposed candidate is identified by its producer role, never by
173
+ * `CandidateSet.selectedCandidateId` (which records the reviewer's pick on the
174
+ * builder and canonical review paths, #279): the single candidate whose
175
+ * `metadata.candidateRole` or `metadata.role` is `"proposed"`, or the only
176
+ * candidate of a one-candidate set (whatever its role). A sample is skipped (and counted
177
+ * in `skippedCount`) when it carries no human label or no prediction:
152
178
  *
153
179
  * - status "proposed" (not yet reviewed);
154
180
  * - resolution "could_not_confirm" (no human correctness label);
155
181
  * - a machine auto-accept, unless `includeAutoAccepted` is set;
156
- * - no `selectedCandidateId`, or the selected candidate / its confidence is
157
- * missing or non-finite (no prediction to calibrate).
182
+ * - the proposed candidate cannot be determined (several candidates without
183
+ * exactly one `"proposed"` role), or its confidence is missing or non-finite.
158
184
  *
159
185
  * A sample is "correct" when the outcome status is verified/assumed AND the
160
- * reviewer did not switch to a different candidate; "incorrect" when the status
161
- * is rejected or the reviewer overrode the proposed candidate.
186
+ * reviewer's pick (`ReviewOutcome.candidateId`, else `selectedCandidateId`) is
187
+ * the proposed candidate; "incorrect" when the status is rejected or the
188
+ * reviewer picked a different candidate (for example, kept the current value).
162
189
  */
163
190
  export declare function deriveCalibration(input: CalibrationInput, options?: DeriveCalibrationOptions): CalibrationMetrics;
164
191
  /** A single projected calibration claim triple. */
@@ -9,10 +9,16 @@
9
9
  * from extractor X was affirmed 17/20 times" — plus a calibration gap and an
10
10
  * empirically-grounded auto-accept threshold suggestion.
11
11
  *
12
- * ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration INFORMS policy — the
13
- * suggested threshold an operator MAY wire into `autoAcceptMinConfidence` — it
14
- * never decides a claim or mutates a status. Projected claims carry status
15
- * "proposed", exactly like every other producer proposal.
12
+ * EXPERIMENTAL. The curve is a descriptive summary of past review outcomes, not
13
+ * a validated probability model. `suggestedThreshold` is withheld unless every
14
+ * contributing bin clears a sample floor AND a one-sided 95% Wilson lower bound
15
+ * on its accuracy clears the target (#279); even then it is not an auto-accept
16
+ * gate — do not wire it into `autoAcceptMinConfidence` without your own
17
+ * evaluation.
18
+ *
19
+ * ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration never decides a claim
20
+ * or mutates a status. Projected claims carry status "proposed", exactly like
21
+ * every other producer proposal.
16
22
  *
17
23
  * Machine auto-accepts are EXCLUDED by default: an auto-accepted outcome is the
18
24
  * threshold accepting its own guess, so counting it as a "correct" label would
@@ -24,27 +30,39 @@
24
30
  import { AUTO_ACCEPT_ACTOR } from "./producer-profile.js";
25
31
  const DEFAULT_BIN_COUNT = 10;
26
32
  const DEFAULT_TARGET_ACCURACY = 0.95;
27
- const DEFAULT_MIN_BIN_SAMPLES = 1;
33
+ /** Default per-bin sample floor for `suggestedThreshold` (#279; was 1). */
34
+ const DEFAULT_MIN_BIN_SAMPLES = 30;
35
+ /** z for a one-sided 95% lower confidence bound. */
36
+ const ONE_SIDED_95_Z = 1.6448536269514722;
28
37
  // ---------------------------------------------------------------------------
29
38
  // deriveCalibration
30
39
  // ---------------------------------------------------------------------------
31
40
  /**
32
41
  * Derives extractor/field confidence calibration from review outcomes.
33
42
  *
43
+ * EXPERIMENTAL — see the module note.
44
+ *
34
45
  * Each reviewed candidate set contributes one labeled sample: the confidence of
35
- * the SYSTEM-proposed candidate (`CandidateSet.selectedCandidateId`) as the
36
- * prediction, and whether the human review affirmed that proposed value as the
37
- * label. A sample is skipped when it carries no human label or no prediction:
46
+ * the SYSTEM-proposed candidate as the prediction, and whether the human review
47
+ * affirmed that proposed value as the label.
48
+ *
49
+ * The proposed candidate is identified by its producer role, never by
50
+ * `CandidateSet.selectedCandidateId` (which records the reviewer's pick on the
51
+ * builder and canonical review paths, #279): the single candidate whose
52
+ * `metadata.candidateRole` or `metadata.role` is `"proposed"`, or the only
53
+ * candidate of a one-candidate set (whatever its role). A sample is skipped (and counted
54
+ * in `skippedCount`) when it carries no human label or no prediction:
38
55
  *
39
56
  * - status "proposed" (not yet reviewed);
40
57
  * - resolution "could_not_confirm" (no human correctness label);
41
58
  * - a machine auto-accept, unless `includeAutoAccepted` is set;
42
- * - no `selectedCandidateId`, or the selected candidate / its confidence is
43
- * missing or non-finite (no prediction to calibrate).
59
+ * - the proposed candidate cannot be determined (several candidates without
60
+ * exactly one `"proposed"` role), or its confidence is missing or non-finite.
44
61
  *
45
62
  * A sample is "correct" when the outcome status is verified/assumed AND the
46
- * reviewer did not switch to a different candidate; "incorrect" when the status
47
- * is rejected or the reviewer overrode the proposed candidate.
63
+ * reviewer's pick (`ReviewOutcome.candidateId`, else `selectedCandidateId`) is
64
+ * the proposed candidate; "incorrect" when the status is rejected or the
65
+ * reviewer picked a different candidate (for example, kept the current value).
48
66
  */
49
67
  export function deriveCalibration(input, options = {}) {
50
68
  const binCount = normalizeBinCount(options.binCount);
@@ -52,11 +70,6 @@ export function deriveCalibration(input, options = {}) {
52
70
  const minBinSamples = options.minBinSamples ?? DEFAULT_MIN_BIN_SAMPLES;
53
71
  const includeAutoAccepted = options.includeAutoAccepted ?? false;
54
72
  const candidateSetById = new Map(input.candidateSets.map((cs) => [cs.id, cs]));
55
- const candidateById = new Map();
56
- for (const cs of input.candidateSets) {
57
- for (const c of cs.candidates)
58
- candidateById.set(c.id, c);
59
- }
60
73
  const extractionById = new Map(input.extractions.map((e) => [e.id, e]));
61
74
  if (options.windowDays !== undefined && options.now === undefined) {
62
75
  throw new RangeError("deriveCalibration: `now` is required when `windowDays` is set (windowing is by reviewedAt).");
@@ -78,7 +91,7 @@ export function deriveCalibration(input, options = {}) {
78
91
  continue;
79
92
  }
80
93
  }
81
- const sample = toSample(outcome, candidateSetById, candidateById, extractionById, includeAutoAccepted);
94
+ const sample = toSample(outcome, candidateSetById, extractionById, includeAutoAccepted);
82
95
  if (sample === undefined) {
83
96
  skippedCount++;
84
97
  continue;
@@ -129,7 +142,7 @@ export function deriveCalibration(input, options = {}) {
129
142
  // ---------------------------------------------------------------------------
130
143
  // Internal: sample construction
131
144
  // ---------------------------------------------------------------------------
132
- function toSample(outcome, candidateSetById, candidateById, extractionById, includeAutoAccepted) {
145
+ function toSample(outcome, candidateSetById, extractionById, includeAutoAccepted) {
133
146
  // No human label yet.
134
147
  if (outcome.status === "proposed")
135
148
  return undefined;
@@ -139,13 +152,11 @@ function toSample(outcome, candidateSetById, candidateById, extractionById, incl
139
152
  const candidateSet = candidateSetById.get(outcome.candidateSetId);
140
153
  if (candidateSet === undefined)
141
154
  return undefined;
142
- // The prediction is the SYSTEM-proposed candidate's confidence.
143
- const proposedId = candidateSet.selectedCandidateId;
144
- if (proposedId === undefined)
145
- return undefined;
146
- const proposed = candidateById.get(proposedId);
155
+ // The prediction is the SYSTEM-proposed candidate's confidence, found by role.
156
+ const proposed = proposedCandidateOf(candidateSet);
147
157
  if (proposed === undefined)
148
158
  return undefined;
159
+ const proposedId = proposed.id;
149
160
  const extraction = proposed.extractionId ? extractionById.get(proposed.extractionId) : undefined;
150
161
  const rawConfidence = proposed.confidence ?? extraction?.confidence;
151
162
  if (rawConfidence === undefined || !Number.isFinite(rawConfidence))
@@ -159,10 +170,10 @@ function toSample(outcome, candidateSetById, candidateById, extractionById, incl
159
170
  correct = false;
160
171
  }
161
172
  else {
162
- // verified | assumed — an override to a different candidate means the
163
- // proposed value did NOT stand.
164
- const overrode = outcome.candidateId !== undefined && outcome.candidateId !== proposedId;
165
- correct = !overrode;
173
+ // verified | assumed — the proposed value stood only if the reviewer's pick
174
+ // is the proposed candidate.
175
+ const pickedId = outcome.candidateId ?? candidateSet.selectedCandidateId;
176
+ correct = pickedId === undefined || pickedId === proposedId;
166
177
  }
167
178
  return {
168
179
  extractor,
@@ -174,6 +185,24 @@ function toSample(outcome, candidateSetById, candidateById, extractionById, incl
174
185
  reviewedAt: outcome.reviewedAt,
175
186
  };
176
187
  }
188
+ /**
189
+ * The candidate whose confidence is the prediction. Returns undefined when it
190
+ * cannot be determined, so the outcome is skipped rather than labeled against
191
+ * the reviewer's own pick.
192
+ *
193
+ * A one-candidate set is sampled whatever the candidate's role ("computed",
194
+ * "source-version", "current", free-form roles, or none): with one candidate
195
+ * there is no pick to confuse with a proposal, and the review either affirmed
196
+ * or rejected that candidate's value. Only a multi-candidate set needs a
197
+ * single "proposed" role marker.
198
+ */
199
+ function proposedCandidateOf(candidateSet) {
200
+ if (candidateSet.candidates.length === 1)
201
+ return candidateSet.candidates[0];
202
+ const roleOf = (c) => c.metadata?.candidateRole ?? c.metadata?.role;
203
+ const proposed = candidateSet.candidates.filter((c) => roleOf(c) === "proposed");
204
+ return proposed.length === 1 ? proposed[0] : undefined;
205
+ }
177
206
  // ---------------------------------------------------------------------------
178
207
  // Internal: group + bin computation
179
208
  // ---------------------------------------------------------------------------
@@ -219,14 +248,24 @@ function computeBins(samples, binCount) {
219
248
  correctCount: b.correct,
220
249
  empiricalAccuracy: b.n > 0 ? round(b.correct / b.n) : undefined,
221
250
  meanPredictedConfidence: b.n > 0 ? round(b.sum / b.n) : undefined,
251
+ ...(b.n > 0 ? { accuracyLowerBound: round(wilsonLowerBound(b.correct, b.n)) } : {}),
222
252
  }));
223
253
  }
254
+ /** One-sided 95% Wilson score lower bound for `correct` successes out of `n`. */
255
+ function wilsonLowerBound(correct, n) {
256
+ const z = ONE_SIDED_95_Z;
257
+ const p = correct / n;
258
+ const z2 = z * z;
259
+ const centre = p + z2 / (2 * n);
260
+ const margin = z * Math.sqrt((p * (1 - p)) / n + z2 / (4 * n * n));
261
+ return Math.max(0, (centre - margin) / (1 + z2 / n));
262
+ }
224
263
  /**
225
264
  * The suggested threshold is the lowerBound of the lowest bin in the
226
- * top-contiguous run of bins that each (a) have ≥ minBinSamples and (b) meet
227
- * targetAccuracy. Scanning from the highest bin down, a populated bin that
228
- * fails the target — or an under-sampled bin we cannot vouch for — ends the run.
229
- * undefined when even the top populated bin does not qualify.
265
+ * top-contiguous run of bins that each (a) have ≥ minBinSamples and (b) have a
266
+ * one-sided 95% Wilson lower bound on accuracy ≥ targetAccuracy. Scanning from
267
+ * the highest bin down, a bin that fails the bound — or an under-sampled bin we
268
+ * cannot vouch for — ends the run. undefined when the top bin does not qualify.
230
269
  */
231
270
  function computeSuggestedThreshold(bins, targetAccuracy, minBinSamples) {
232
271
  let threshold;
@@ -234,7 +273,7 @@ function computeSuggestedThreshold(bins, targetAccuracy, minBinSamples) {
234
273
  const bin = bins[i];
235
274
  if (bin.sampleCount < minBinSamples)
236
275
  break;
237
- if (bin.empiricalAccuracy === undefined || bin.empiricalAccuracy < targetAccuracy)
276
+ if (bin.sampleCount === 0 || wilsonLowerBound(bin.correctCount, bin.sampleCount) < targetAccuracy)
238
277
  break;
239
278
  threshold = bin.lowerBound;
240
279
  }
@@ -5,7 +5,9 @@
5
5
  * Routes:
6
6
  * GET / HTML shell that mounts the workbench
7
7
  * GET /api/session Current session state (snapshot + replayed events)
8
- * POST /api/events Append review session events (same contract as MCP server)
8
+ * POST /api/events Append review session events to the stored log (same
9
+ * validation as MCP server), compare-and-swap on the
10
+ * revision the client last read
9
11
  * GET /api/stream SSE stream: emits "update" events when the session file changes
10
12
  * GET /api/health Health check
11
13
  * GET /dist/* Compiled assets served from the dist tree (traversal-safe)