@kontourai/survey 3.0.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/README.md +4 -0
  2. package/dist/example-data/public-directory-review-resource.d.ts +3 -3
  3. package/dist/examples/calibrated-auto-accept.d.ts +22 -15
  4. package/dist/examples/calibrated-auto-accept.js +40 -36
  5. package/dist/src/calibration.d.ts +48 -21
  6. package/dist/src/calibration.js +72 -33
  7. package/dist/src/console/review-console-server.d.ts +3 -1
  8. package/dist/src/console/review-console-server.js +203 -50
  9. package/dist/src/extraction-envelope.d.ts +22 -0
  10. package/dist/src/extraction-envelope.js +25 -4
  11. package/dist/src/index.d.ts +8 -7
  12. package/dist/src/index.js +2 -2
  13. package/dist/src/inquiry-mapping.d.ts +15 -1
  14. package/dist/src/inquiry-mapping.js +10 -2
  15. package/dist/src/mcp/review-mcp.js +70 -65
  16. package/dist/src/producer-profile.d.ts +41 -2
  17. package/dist/src/producer-profile.js +29 -2
  18. package/dist/src/review-session-file.d.ts +64 -0
  19. package/dist/src/review-session-file.js +320 -0
  20. package/dist/src/review-workbench/edited-value.d.ts +70 -0
  21. package/dist/src/review-workbench/edited-value.js +147 -0
  22. package/dist/src/review-workbench/review-presentation.d.ts +44 -0
  23. package/dist/src/review-workbench/review-presentation.js +49 -0
  24. package/dist/src/review-workbench/review-queue-session.js +8 -1
  25. package/dist/src/review-workbench/review-session-replay.d.ts +35 -1
  26. package/dist/src/review-workbench/review-session-replay.js +77 -0
  27. package/dist/src/review-workbench/review-workbench.d.ts +8 -4
  28. package/dist/src/review-workbench/review-workbench.js +19 -7
  29. package/dist/src/review-workbench/server-review-session.d.ts +3 -1
  30. package/dist/src/review-workbench/server-review-session.js +1 -0
  31. package/dist/src/reviewed-candidate-resolution.js +13 -7
  32. package/dist/src/schema-mapping.d.ts +23 -0
  33. package/dist/src/schema-mapping.js +30 -20
  34. package/dist/src/to-surface.d.ts +30 -6
  35. package/dist/src/to-surface.js +196 -18
  36. package/dist/src/types.d.ts +39 -0
  37. package/package.json +5 -4
package/README.md CHANGED
@@ -40,6 +40,10 @@ The Review Workbench rendering a real example queue — current vs proposed valu
40
40
  npm install @kontourai/survey @kontourai/surface
41
41
  ```
42
42
 
43
+ Survey supports `@kontourai/surface` 2.13 and later 2.x, and 3.x, and CI tests
44
+ the lowest 2.x and the newest 3.x. Install either major and your project and
45
+ Survey share one Surface copy (`npm ls @kontourai/surface` shows one entry).
46
+
43
47
  Requires Node.js >=22. TypeScript >=5.0 is required to compile against
44
48
  Survey's published type declarations (`defineProductVocabulary`'s `const`
45
49
  type parameters are TS 5.0+ syntax) — JavaScript consumers are unaffected;
@@ -35,12 +35,12 @@ export declare const publicDirectoryReviewItemExample: {
35
35
  excerpt: string;
36
36
  };
37
37
  extraction: {
38
- model?: undefined;
39
38
  extractionId: string;
40
39
  target: string;
41
40
  confidence: number;
42
41
  extractor: string;
43
42
  extractedAt: string;
43
+ model?: undefined;
44
44
  };
45
45
  claimTarget: {
46
46
  claimId: string;
@@ -61,13 +61,13 @@ export declare const publicDirectoryReviewItemExample: {
61
61
  claimId: string;
62
62
  };
63
63
  producer: {
64
- proposalId?: undefined;
65
- oldValue?: undefined;
66
64
  sourceAuthority: {
67
65
  authorityClass: string;
68
66
  declaredBy: string;
69
67
  scope: string;
70
68
  };
69
+ proposalId?: undefined;
70
+ oldValue?: undefined;
71
71
  };
72
72
  } | {
73
73
  id: string;
@@ -1,29 +1,36 @@
1
1
  /**
2
- * Worked example: wiring confidence calibration into a downstream consumer.
2
+ * Worked example: EXPERIMENTAL confidence calibration from review history.
3
3
  *
4
- * A consumer that owns human review outcomes can close the confidence loop in two
5
- * places, both shipped in @kontourai/survey (1.10.0):
4
+ * Calibration summarizes how often reviewers affirmed an extractor's proposals.
5
+ * Both outputs below are experimental descriptive statistics (#279), not
6
+ * validated probabilities or policy:
6
7
  *
7
- * 1. Ground the auto-accept threshold. `deriveCalibration` over a history of
8
- * review outcomes yields `suggestedThreshold` — the lowest confidence at
9
- * which the extractor's proposals were empirically affirmed often enough.
10
- * Feed that number into a producer profile's `autoAcceptMinConfidence`
11
- * (see `SchemaMappingOptions.autoAcceptMinConfidence` /
12
- * `InquiryMappingOptions.minConfidence`) instead of hand-picking it.
8
+ * 1. `deriveCalibration` reports per-decile affirmation rates and, only when a
9
+ * decile has enough samples AND a one-sided 95% Wilson lower bound on its
10
+ * accuracy clears the target, a `suggestedThreshold`. A thin history gets
11
+ * no threshold at all. Do not feed `suggestedThreshold` into an auto-accept
12
+ * policy (`autoAcceptMinConfidence` / `minConfidence`) as-is: it is not an
13
+ * evaluated gate.
13
14
  *
14
- * 2. Produce calibrated conclusion confidence. Pass the same calibration
15
- * `metrics` into `buildSurveyTrustBundle({ calibration: { metrics } })` and
16
- * affirmed claims carry `conclusionConfidence.value` = the empirical
17
- * affirmation rate for their extractor/field.
15
+ * 2. `buildSurveyTrustBundle({ calibration: { experimentalConclusionValue:
16
+ * true, metrics } })` sets `conclusionConfidence.value` on affirmed claims
17
+ * to their extractor/field GROUP's affirmation rate. Every affirmed claim in
18
+ * the group gets the same base rate whatever its own confidence, so it is
19
+ * not a per-claim probability. Without the experimental flag no value is set.
18
20
  *
19
- * Calibration is advisory (ADR 0003 §4): the threshold is a suggestion the
20
- * operator wires into policy, and the produced value never changes claim status.
21
+ * Calibration is advisory (ADR 0003 §4): it never changes claim status.
21
22
  *
22
23
  * Run: `node dist/examples/calibrated-auto-accept.js`
23
24
  */
24
25
  export interface CalibratedAutoAcceptResult {
26
+ /** Threshold from a 60-samples-per-decile history (enough evidence). */
25
27
  readonly suggestedThreshold: number | undefined;
28
+ /** Threshold from a 10-samples-per-decile history (withheld: too thin). */
29
+ readonly sparseHistoryThreshold: number | undefined;
26
30
  readonly groupAccuracy: number | undefined;
31
+ /** Values with the experimental opt-in: the group base rate on each claim. */
27
32
  readonly producedValues: ReadonlyArray<number | undefined>;
33
+ /** Values with `calibration` enabled but no experimental opt-in: none. */
34
+ readonly defaultValues: ReadonlyArray<number | undefined>;
28
35
  }
29
36
  export declare function runCalibratedAutoAccept(): CalibratedAutoAcceptResult;
@@ -1,23 +1,24 @@
1
1
  /**
2
- * Worked example: wiring confidence calibration into a downstream consumer.
2
+ * Worked example: EXPERIMENTAL confidence calibration from review history.
3
3
  *
4
- * A consumer that owns human review outcomes can close the confidence loop in two
5
- * places, both shipped in @kontourai/survey (1.10.0):
4
+ * Calibration summarizes how often reviewers affirmed an extractor's proposals.
5
+ * Both outputs below are experimental descriptive statistics (#279), not
6
+ * validated probabilities or policy:
6
7
  *
7
- * 1. Ground the auto-accept threshold. `deriveCalibration` over a history of
8
- * review outcomes yields `suggestedThreshold` — the lowest confidence at
9
- * which the extractor's proposals were empirically affirmed often enough.
10
- * Feed that number into a producer profile's `autoAcceptMinConfidence`
11
- * (see `SchemaMappingOptions.autoAcceptMinConfidence` /
12
- * `InquiryMappingOptions.minConfidence`) instead of hand-picking it.
8
+ * 1. `deriveCalibration` reports per-decile affirmation rates and, only when a
9
+ * decile has enough samples AND a one-sided 95% Wilson lower bound on its
10
+ * accuracy clears the target, a `suggestedThreshold`. A thin history gets
11
+ * no threshold at all. Do not feed `suggestedThreshold` into an auto-accept
12
+ * policy (`autoAcceptMinConfidence` / `minConfidence`) as-is: it is not an
13
+ * evaluated gate.
13
14
  *
14
- * 2. Produce calibrated conclusion confidence. Pass the same calibration
15
- * `metrics` into `buildSurveyTrustBundle({ calibration: { metrics } })` and
16
- * affirmed claims carry `conclusionConfidence.value` = the empirical
17
- * affirmation rate for their extractor/field.
15
+ * 2. `buildSurveyTrustBundle({ calibration: { experimentalConclusionValue:
16
+ * true, metrics } })` sets `conclusionConfidence.value` on affirmed claims
17
+ * to their extractor/field GROUP's affirmation rate. Every affirmed claim in
18
+ * the group gets the same base rate whatever its own confidence, so it is
19
+ * not a per-claim probability. Without the experimental flag no value is set.
18
20
  *
19
- * Calibration is advisory (ADR 0003 §4): the threshold is a suggestion the
20
- * operator wires into policy, and the produced value never changes claim status.
21
+ * Calibration is advisory (ADR 0003 §4): it never changes claim status.
21
22
  *
22
23
  * Run: `node dist/examples/calibrated-auto-accept.js`
23
24
  */
@@ -31,15 +32,15 @@ const BATCH_AT = "2026-07-01T00:00:00.000Z";
31
32
  * almost always affirmed by reviewers and low-confidence ones were mostly
32
33
  * rejected — the pattern that makes an empirical threshold meaningful.
33
34
  */
34
- function buildReviewHistory() {
35
+ function buildReviewHistory(samplesPerDecile) {
35
36
  const extractions = [];
36
37
  const candidateSets = [];
37
38
  const reviewOutcomes = [];
38
- let n = 0;
39
+ let seq = 0;
39
40
  const addSamples = (count, confidence, affirmed) => {
40
41
  for (let i = 0; i < count; i += 1) {
41
- const key = `h-${n}`;
42
- n += 1;
42
+ const key = `h-${seq}`;
43
+ seq += 1;
43
44
  extractions.push({
44
45
  id: `${key}-ext`,
45
46
  sourceId: `${key}-src`,
@@ -67,33 +68,34 @@ function buildReviewHistory() {
67
68
  });
68
69
  }
69
70
  };
70
- addSamples(10, 0.95, 10); // top decile: all affirmed
71
- addSamples(10, 0.85, 10); // 0.8–0.9: all affirmed
72
- addSamples(10, 0.75, 5); // 0.7–0.8: half affirmed → below target, ends the run
73
- addSamples(10, 0.55, 1); // 0.5–0.6: mostly rejected
71
+ const n = samplesPerDecile;
72
+ addSamples(n, 0.95, n); // top decile: all affirmed
73
+ addSamples(n, 0.85, n - 1); // 0.8–0.9: all but one affirmed
74
+ addSamples(n, 0.75, Math.round(n / 2)); // 0.7–0.8: half affirmed → ends the run
75
+ addSamples(n, 0.55, Math.round(n / 10)); // 0.5–0.6: mostly rejected
74
76
  return { reviewOutcomes, candidateSets, extractions };
75
77
  }
76
78
  export function runCalibratedAutoAccept() {
77
79
  // (1) Derive the empirical calibration curve over the review history.
78
- const metrics = deriveCalibration(buildReviewHistory(), {
79
- targetAccuracy: 0.9, // the accuracy the auto-accept threshold must clear
80
- minBinSamples: 5, // a decile needs this many samples to ground the threshold
81
- });
80
+ const options = { targetAccuracy: 0.9 }; // default 30-sample floor per decile
81
+ const metrics = deriveCalibration(buildReviewHistory(60), options);
82
82
  const suggestedThreshold = metrics.overall.suggestedThreshold;
83
+ const sparseHistoryThreshold = deriveCalibration(buildReviewHistory(10), options).overall.suggestedThreshold;
83
84
  const group = metrics.byExtractorField.find((g) => g.extractor === EXTRACTOR && g.field === FIELD);
84
- // This is the number you feed into your producer profile's auto-accept policy:
85
- // surveySchemaMapping(context, extractor, { autoAcceptMinConfidence: suggestedThreshold })
86
- // Proposals at/above it were empirically affirmed often enough to auto-accept.
87
- // (2) Produce calibrated conclusion confidence on a new batch of affirmed claims.
85
+ // (2) Attach the group affirmation rate to a new batch of affirmed claims.
88
86
  const input = new SurveyInputBuilder({ source: "example-consumer:calibrated", generatedAt: BATCH_AT })
89
87
  .addObservation(affirmedObservation("entity-1", 0.85))
90
88
  .addObservation(affirmedObservation("entity-2", 0.92))
91
89
  .build();
92
90
  // Prefer metrics computed over history (not just this batch), so a claim's own
93
91
  // outcome does not feed its own value.
94
- const bundle = buildSurveyTrustBundle(input, { calibration: { metrics, minSamples: 20 } });
92
+ const bundle = buildSurveyTrustBundle(input, {
93
+ calibration: { experimentalConclusionValue: true, metrics, minSamples: 20 },
94
+ });
95
95
  const producedValues = bundle.claims.map((c) => c.conclusionConfidence?.value);
96
- return { suggestedThreshold, groupAccuracy: group?.empiricalAccuracy, producedValues };
96
+ const defaultValues = buildSurveyTrustBundle(input, { calibration: { metrics, minSamples: 20 } })
97
+ .claims.map((c) => c.conclusionConfidence?.value);
98
+ return { suggestedThreshold, sparseHistoryThreshold, groupAccuracy: group?.empiricalAccuracy, producedValues, defaultValues };
97
99
  }
98
100
  function affirmedObservation(subjectId, confidence) {
99
101
  return {
@@ -128,8 +130,10 @@ function affirmedObservation(subjectId, confidence) {
128
130
  if (process.argv[1]?.endsWith("calibrated-auto-accept.js")) {
129
131
  const result = runCalibratedAutoAccept();
130
132
  console.log(JSON.stringify({
131
- suggestedAutoAcceptThreshold: result.suggestedThreshold,
132
- empiricalAffirmationRate: result.groupAccuracy,
133
- producedConclusionConfidenceValues: result.producedValues,
133
+ experimentalSuggestedThreshold: result.suggestedThreshold,
134
+ sparseHistorySuggestedThreshold: result.sparseHistoryThreshold,
135
+ groupAffirmationRate: result.groupAccuracy,
136
+ experimentalConclusionValues: result.producedValues,
137
+ valuesWithoutOptIn: result.defaultValues,
134
138
  }, null, 2));
135
139
  }
@@ -9,10 +9,16 @@
9
9
  * from extractor X was affirmed 17/20 times" — plus a calibration gap and an
10
10
  * empirically-grounded auto-accept threshold suggestion.
11
11
  *
12
- * ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration INFORMS policy — the
13
- * suggested threshold an operator MAY wire into `autoAcceptMinConfidence` — it
14
- * never decides a claim or mutates a status. Projected claims carry status
15
- * "proposed", exactly like every other producer proposal.
12
+ * EXPERIMENTAL. The curve is a descriptive summary of past review outcomes, not
13
+ * a validated probability model. `suggestedThreshold` is withheld unless every
14
+ * contributing bin clears a sample floor AND a one-sided 95% Wilson lower bound
15
+ * on its accuracy clears the target (#279); even then it is not an auto-accept
16
+ * gate — do not wire it into `autoAcceptMinConfidence` without your own
17
+ * evaluation.
18
+ *
19
+ * ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration never decides a claim
20
+ * or mutates a status. Projected claims carry status "proposed", exactly like
21
+ * every other producer proposal.
16
22
  *
17
23
  * Machine auto-accepts are EXCLUDED by default: an auto-accepted outcome is the
18
24
  * threshold accepting its own guess, so counting it as a "correct" label would
@@ -49,15 +55,19 @@ export interface DeriveCalibrationOptions {
49
55
  /** Number of equal-width confidence bins over [0,1]. Default 10 (deciles). */
50
56
  readonly binCount?: number;
51
57
  /**
52
- * The empirical accuracy the suggested threshold must clear. Default 0.95.
53
- * A `suggestedThreshold` is the lowest bin lower-bound at/above which every
54
- * populated bin's empirical accuracy meets this target.
58
+ * The accuracy the suggested threshold must clear. Default 0.95. A
59
+ * `suggestedThreshold` is the lowest bin lower-bound at/above which every bin
60
+ * has ≥ `minBinSamples` samples and a one-sided 95% Wilson lower confidence
61
+ * bound on its accuracy (`CalibrationBin.accuracyLowerBound`) that meets this
62
+ * target — the point estimate alone is not enough.
55
63
  */
56
64
  readonly targetAccuracy?: number;
57
65
  /**
58
66
  * A bin needs at least this many samples to count toward `suggestedThreshold`
59
- * (both to qualify and to disqualify). Default 1. Raise it to avoid grounding
60
- * a threshold on a bin with too little evidence.
67
+ * (an under-sampled bin ends the qualifying run). Default
68
+ * {@link DEFAULT_MIN_BIN_SAMPLES} (30). The Wilson bound applies on top of this
69
+ * floor, so an all-affirmed bin needs more samples than the floor when the
70
+ * target is high (≈52 all-affirmed samples for the default 0.95 target).
61
71
  */
62
72
  readonly minBinSamples?: number;
63
73
  /**
@@ -95,6 +105,13 @@ export interface CalibrationBin {
95
105
  readonly empiricalAccuracy: number | undefined;
96
106
  /** Mean predicted confidence of samples in the bin; undefined when empty. */
97
107
  readonly meanPredictedConfidence: number | undefined;
108
+ /**
109
+ * One-sided 95% Wilson lower confidence bound on the bin's accuracy; undefined
110
+ * when the bin is empty. `suggestedThreshold` requires this bound, not the
111
+ * point estimate, to meet `targetAccuracy`. Optional in the type so metrics
112
+ * built by hand before this field existed still type-check.
113
+ */
114
+ readonly accuracyLowerBound?: number;
98
115
  }
99
116
  /** A calibration rollup for one extractor (and optionally one field). */
100
117
  export interface CalibrationGroup {
@@ -116,11 +133,12 @@ export interface CalibrationGroup {
116
133
  /** Per-bin empirical accuracy, ascending by lowerBound. */
117
134
  readonly bins: readonly CalibrationBin[];
118
135
  /**
119
- * Lowest bin lowerBound at/above which every populated bin (≥ minBinSamples)
120
- * meets `targetAccuracy`, scanning the top-contiguous run of qualifying bins.
121
- * undefined when no bin qualifies — the data does not yet support an empirical
122
- * auto-accept threshold at that target. ADVISORY: an operator wires this into
123
- * `autoAcceptMinConfidence`; calibration never sets it.
136
+ * EXPERIMENTAL. Lowest bin lowerBound of the top-contiguous run of bins that
137
+ * each have ≥ minBinSamples samples and an `accuracyLowerBound` ≥
138
+ * `targetAccuracy`. undefined when no bin qualifies — the data does not
139
+ * support an empirical threshold at that target. Each contributing bin reports
140
+ * its `sampleCount` and `accuracyLowerBound`. This is a descriptive summary,
141
+ * not a validated auto-accept gate; calibration never sets any policy.
124
142
  */
125
143
  readonly suggestedThreshold: number | undefined;
126
144
  }
@@ -145,20 +163,29 @@ export interface CalibrationMetrics {
145
163
  /**
146
164
  * Derives extractor/field confidence calibration from review outcomes.
147
165
  *
166
+ * EXPERIMENTAL — see the module note.
167
+ *
148
168
  * Each reviewed candidate set contributes one labeled sample: the confidence of
149
- * the SYSTEM-proposed candidate (`CandidateSet.selectedCandidateId`) as the
150
- * prediction, and whether the human review affirmed that proposed value as the
151
- * label. A sample is skipped when it carries no human label or no prediction:
169
+ * the SYSTEM-proposed candidate as the prediction, and whether the human review
170
+ * affirmed that proposed value as the label.
171
+ *
172
+ * The proposed candidate is identified by its producer role, never by
173
+ * `CandidateSet.selectedCandidateId` (which records the reviewer's pick on the
174
+ * builder and canonical review paths, #279): the single candidate whose
175
+ * `metadata.candidateRole` or `metadata.role` is `"proposed"`, or the only
176
+ * candidate of a one-candidate set (whatever its role). A sample is skipped (and counted
177
+ * in `skippedCount`) when it carries no human label or no prediction:
152
178
  *
153
179
  * - status "proposed" (not yet reviewed);
154
180
  * - resolution "could_not_confirm" (no human correctness label);
155
181
  * - a machine auto-accept, unless `includeAutoAccepted` is set;
156
- * - no `selectedCandidateId`, or the selected candidate / its confidence is
157
- * missing or non-finite (no prediction to calibrate).
182
+ * - the proposed candidate cannot be determined (several candidates without
183
+ * exactly one `"proposed"` role), or its confidence is missing or non-finite.
158
184
  *
159
185
  * A sample is "correct" when the outcome status is verified/assumed AND the
160
- * reviewer did not switch to a different candidate; "incorrect" when the status
161
- * is rejected or the reviewer overrode the proposed candidate.
186
+ * reviewer's pick (`ReviewOutcome.candidateId`, else `selectedCandidateId`) is
187
+ * the proposed candidate; "incorrect" when the status is rejected or the
188
+ * reviewer picked a different candidate (for example, kept the current value).
162
189
  */
163
190
  export declare function deriveCalibration(input: CalibrationInput, options?: DeriveCalibrationOptions): CalibrationMetrics;
164
191
  /** A single projected calibration claim triple. */
@@ -9,10 +9,16 @@
9
9
  * from extractor X was affirmed 17/20 times" — plus a calibration gap and an
10
10
  * empirically-grounded auto-accept threshold suggestion.
11
11
  *
12
- * ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration INFORMS policy — the
13
- * suggested threshold an operator MAY wire into `autoAcceptMinConfidence` — it
14
- * never decides a claim or mutates a status. Projected claims carry status
15
- * "proposed", exactly like every other producer proposal.
12
+ * EXPERIMENTAL. The curve is a descriptive summary of past review outcomes, not
13
+ * a validated probability model. `suggestedThreshold` is withheld unless every
14
+ * contributing bin clears a sample floor AND a one-sided 95% Wilson lower bound
15
+ * on its accuracy clears the target (#279); even then it is not an auto-accept
16
+ * gate — do not wire it into `autoAcceptMinConfidence` without your own
17
+ * evaluation.
18
+ *
19
+ * ADVISORY ONLY (ADR 0003 §4, proposals-only). Calibration never decides a claim
20
+ * or mutates a status. Projected claims carry status "proposed", exactly like
21
+ * every other producer proposal.
16
22
  *
17
23
  * Machine auto-accepts are EXCLUDED by default: an auto-accepted outcome is the
18
24
  * threshold accepting its own guess, so counting it as a "correct" label would
@@ -24,27 +30,39 @@
24
30
  import { AUTO_ACCEPT_ACTOR } from "./producer-profile.js";
25
31
  const DEFAULT_BIN_COUNT = 10;
26
32
  const DEFAULT_TARGET_ACCURACY = 0.95;
27
- const DEFAULT_MIN_BIN_SAMPLES = 1;
33
+ /** Default per-bin sample floor for `suggestedThreshold` (#279; was 1). */
34
+ const DEFAULT_MIN_BIN_SAMPLES = 30;
35
+ /** z for a one-sided 95% lower confidence bound. */
36
+ const ONE_SIDED_95_Z = 1.6448536269514722;
28
37
  // ---------------------------------------------------------------------------
29
38
  // deriveCalibration
30
39
  // ---------------------------------------------------------------------------
31
40
  /**
32
41
  * Derives extractor/field confidence calibration from review outcomes.
33
42
  *
43
+ * EXPERIMENTAL — see the module note.
44
+ *
34
45
  * Each reviewed candidate set contributes one labeled sample: the confidence of
35
- * the SYSTEM-proposed candidate (`CandidateSet.selectedCandidateId`) as the
36
- * prediction, and whether the human review affirmed that proposed value as the
37
- * label. A sample is skipped when it carries no human label or no prediction:
46
+ * the SYSTEM-proposed candidate as the prediction, and whether the human review
47
+ * affirmed that proposed value as the label.
48
+ *
49
+ * The proposed candidate is identified by its producer role, never by
50
+ * `CandidateSet.selectedCandidateId` (which records the reviewer's pick on the
51
+ * builder and canonical review paths, #279): the single candidate whose
52
+ * `metadata.candidateRole` or `metadata.role` is `"proposed"`, or the only
53
+ * candidate of a one-candidate set (whatever its role). A sample is skipped (and counted
54
+ * in `skippedCount`) when it carries no human label or no prediction:
38
55
  *
39
56
  * - status "proposed" (not yet reviewed);
40
57
  * - resolution "could_not_confirm" (no human correctness label);
41
58
  * - a machine auto-accept, unless `includeAutoAccepted` is set;
42
- * - no `selectedCandidateId`, or the selected candidate / its confidence is
43
- * missing or non-finite (no prediction to calibrate).
59
+ * - the proposed candidate cannot be determined (several candidates without
60
+ * exactly one `"proposed"` role), or its confidence is missing or non-finite.
44
61
  *
45
62
  * A sample is "correct" when the outcome status is verified/assumed AND the
46
- * reviewer did not switch to a different candidate; "incorrect" when the status
47
- * is rejected or the reviewer overrode the proposed candidate.
63
+ * reviewer's pick (`ReviewOutcome.candidateId`, else `selectedCandidateId`) is
64
+ * the proposed candidate; "incorrect" when the status is rejected or the
65
+ * reviewer picked a different candidate (for example, kept the current value).
48
66
  */
49
67
  export function deriveCalibration(input, options = {}) {
50
68
  const binCount = normalizeBinCount(options.binCount);
@@ -52,11 +70,6 @@ export function deriveCalibration(input, options = {}) {
52
70
  const minBinSamples = options.minBinSamples ?? DEFAULT_MIN_BIN_SAMPLES;
53
71
  const includeAutoAccepted = options.includeAutoAccepted ?? false;
54
72
  const candidateSetById = new Map(input.candidateSets.map((cs) => [cs.id, cs]));
55
- const candidateById = new Map();
56
- for (const cs of input.candidateSets) {
57
- for (const c of cs.candidates)
58
- candidateById.set(c.id, c);
59
- }
60
73
  const extractionById = new Map(input.extractions.map((e) => [e.id, e]));
61
74
  if (options.windowDays !== undefined && options.now === undefined) {
62
75
  throw new RangeError("deriveCalibration: `now` is required when `windowDays` is set (windowing is by reviewedAt).");
@@ -78,7 +91,7 @@ export function deriveCalibration(input, options = {}) {
78
91
  continue;
79
92
  }
80
93
  }
81
- const sample = toSample(outcome, candidateSetById, candidateById, extractionById, includeAutoAccepted);
94
+ const sample = toSample(outcome, candidateSetById, extractionById, includeAutoAccepted);
82
95
  if (sample === undefined) {
83
96
  skippedCount++;
84
97
  continue;
@@ -129,7 +142,7 @@ export function deriveCalibration(input, options = {}) {
129
142
  // ---------------------------------------------------------------------------
130
143
  // Internal: sample construction
131
144
  // ---------------------------------------------------------------------------
132
- function toSample(outcome, candidateSetById, candidateById, extractionById, includeAutoAccepted) {
145
+ function toSample(outcome, candidateSetById, extractionById, includeAutoAccepted) {
133
146
  // No human label yet.
134
147
  if (outcome.status === "proposed")
135
148
  return undefined;
@@ -139,13 +152,11 @@ function toSample(outcome, candidateSetById, candidateById, extractionById, incl
139
152
  const candidateSet = candidateSetById.get(outcome.candidateSetId);
140
153
  if (candidateSet === undefined)
141
154
  return undefined;
142
- // The prediction is the SYSTEM-proposed candidate's confidence.
143
- const proposedId = candidateSet.selectedCandidateId;
144
- if (proposedId === undefined)
145
- return undefined;
146
- const proposed = candidateById.get(proposedId);
155
+ // The prediction is the SYSTEM-proposed candidate's confidence, found by role.
156
+ const proposed = proposedCandidateOf(candidateSet);
147
157
  if (proposed === undefined)
148
158
  return undefined;
159
+ const proposedId = proposed.id;
149
160
  const extraction = proposed.extractionId ? extractionById.get(proposed.extractionId) : undefined;
150
161
  const rawConfidence = proposed.confidence ?? extraction?.confidence;
151
162
  if (rawConfidence === undefined || !Number.isFinite(rawConfidence))
@@ -159,10 +170,10 @@ function toSample(outcome, candidateSetById, candidateById, extractionById, incl
159
170
  correct = false;
160
171
  }
161
172
  else {
162
- // verified | assumed — an override to a different candidate means the
163
- // proposed value did NOT stand.
164
- const overrode = outcome.candidateId !== undefined && outcome.candidateId !== proposedId;
165
- correct = !overrode;
173
+ // verified | assumed — the proposed value stood only if the reviewer's pick
174
+ // is the proposed candidate.
175
+ const pickedId = outcome.candidateId ?? candidateSet.selectedCandidateId;
176
+ correct = pickedId === undefined || pickedId === proposedId;
166
177
  }
167
178
  return {
168
179
  extractor,
@@ -174,6 +185,24 @@ function toSample(outcome, candidateSetById, candidateById, extractionById, incl
174
185
  reviewedAt: outcome.reviewedAt,
175
186
  };
176
187
  }
188
+ /**
189
+ * The candidate whose confidence is the prediction. Returns undefined when it
190
+ * cannot be determined, so the outcome is skipped rather than labeled against
191
+ * the reviewer's own pick.
192
+ *
193
+ * A one-candidate set is sampled whatever the candidate's role ("computed",
194
+ * "source-version", "current", free-form roles, or none): with one candidate
195
+ * there is no pick to confuse with a proposal, and the review either affirmed
196
+ * or rejected that candidate's value. Only a multi-candidate set needs a
197
+ * single "proposed" role marker.
198
+ */
199
+ function proposedCandidateOf(candidateSet) {
200
+ if (candidateSet.candidates.length === 1)
201
+ return candidateSet.candidates[0];
202
+ const roleOf = (c) => c.metadata?.candidateRole ?? c.metadata?.role;
203
+ const proposed = candidateSet.candidates.filter((c) => roleOf(c) === "proposed");
204
+ return proposed.length === 1 ? proposed[0] : undefined;
205
+ }
177
206
  // ---------------------------------------------------------------------------
178
207
  // Internal: group + bin computation
179
208
  // ---------------------------------------------------------------------------
@@ -219,14 +248,24 @@ function computeBins(samples, binCount) {
219
248
  correctCount: b.correct,
220
249
  empiricalAccuracy: b.n > 0 ? round(b.correct / b.n) : undefined,
221
250
  meanPredictedConfidence: b.n > 0 ? round(b.sum / b.n) : undefined,
251
+ ...(b.n > 0 ? { accuracyLowerBound: round(wilsonLowerBound(b.correct, b.n)) } : {}),
222
252
  }));
223
253
  }
254
+ /** One-sided 95% Wilson score lower bound for `correct` successes out of `n`. */
255
+ function wilsonLowerBound(correct, n) {
256
+ const z = ONE_SIDED_95_Z;
257
+ const p = correct / n;
258
+ const z2 = z * z;
259
+ const centre = p + z2 / (2 * n);
260
+ const margin = z * Math.sqrt((p * (1 - p)) / n + z2 / (4 * n * n));
261
+ return Math.max(0, (centre - margin) / (1 + z2 / n));
262
+ }
224
263
  /**
225
264
  * The suggested threshold is the lowerBound of the lowest bin in the
226
- * top-contiguous run of bins that each (a) have ≥ minBinSamples and (b) meet
227
- * targetAccuracy. Scanning from the highest bin down, a populated bin that
228
- * fails the target — or an under-sampled bin we cannot vouch for — ends the run.
229
- * undefined when even the top populated bin does not qualify.
265
+ * top-contiguous run of bins that each (a) have ≥ minBinSamples and (b) have a
266
+ * one-sided 95% Wilson lower bound on accuracy ≥ targetAccuracy. Scanning from
267
+ * the highest bin down, a bin that fails the bound — or an under-sampled bin we
268
+ * cannot vouch for — ends the run. undefined when the top bin does not qualify.
230
269
  */
231
270
  function computeSuggestedThreshold(bins, targetAccuracy, minBinSamples) {
232
271
  let threshold;
@@ -234,7 +273,7 @@ function computeSuggestedThreshold(bins, targetAccuracy, minBinSamples) {
234
273
  const bin = bins[i];
235
274
  if (bin.sampleCount < minBinSamples)
236
275
  break;
237
- if (bin.empiricalAccuracy === undefined || bin.empiricalAccuracy < targetAccuracy)
276
+ if (bin.sampleCount === 0 || wilsonLowerBound(bin.correctCount, bin.sampleCount) < targetAccuracy)
238
277
  break;
239
278
  threshold = bin.lowerBound;
240
279
  }
@@ -5,7 +5,9 @@
5
5
  * Routes:
6
6
  * GET / HTML shell that mounts the workbench
7
7
  * GET /api/session Current session state (snapshot + replayed events)
8
- * POST /api/events Append review session events (same contract as MCP server)
8
+ * POST /api/events Append review session events to the stored log (same
9
+ * validation as MCP server), compare-and-swap on the
10
+ * revision the client last read
9
11
  * GET /api/stream SSE stream: emits "update" events when the session file changes
10
12
  * GET /api/health Health check
11
13
  * GET /dist/* Compiled assets served from the dist tree (traversal-safe)