@kontourai/survey 1.10.0 → 1.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -40,6 +40,7 @@ export declare const publicDirectoryReviewItemExample: {
40
40
  confidence: number;
41
41
  extractor: string;
42
42
  extractedAt: string;
43
+ model?: undefined;
43
44
  };
44
45
  claimTarget: {
45
46
  claimId: string;
@@ -91,6 +92,7 @@ export declare const publicDirectoryReviewItemExample: {
91
92
  target: string;
92
93
  confidence: number;
93
94
  extractor: string;
95
+ model: string;
94
96
  extractedAt: string;
95
97
  };
96
98
  claimTarget: {
@@ -92,6 +92,7 @@ export const publicDirectoryReviewItemExample = {
92
92
  target: "availabilityStatus",
93
93
  confidence: 0.82,
94
94
  extractor: "example-crawl",
95
+ model: "example-extraction-model-2026-05",
95
96
  extractedAt: "2026-05-31T15:00:00.000Z",
96
97
  },
97
98
  claimTarget: {
@@ -0,0 +1,29 @@
1
+ /**
2
+ * Worked example: wiring confidence calibration into a downstream consumer.
3
+ *
4
+ * A consumer that owns human review outcomes can close the confidence loop in two
5
+ * places, both shipped in @kontourai/survey (1.10.0):
6
+ *
7
+ * 1. Ground the auto-accept threshold. `deriveCalibration` over a history of
8
+ * review outcomes yields `suggestedThreshold` — the lowest confidence at
9
+ * which the extractor's proposals were empirically affirmed often enough.
10
+ * Feed that number into a producer profile's `autoAcceptMinConfidence`
11
+ * (see `SchemaMappingOptions.autoAcceptMinConfidence` /
12
+ * `InquiryMappingOptions.minConfidence`) instead of hand-picking it.
13
+ *
14
+ * 2. Produce calibrated conclusion confidence. Pass the same calibration
15
+ * `metrics` into `buildSurveyTrustBundle({ calibration: { metrics } })` and
16
+ * affirmed claims carry `conclusionConfidence.value` = the empirical
17
+ * affirmation rate for their extractor/field.
18
+ *
19
+ * Calibration is advisory (ADR 0003 §4): the threshold is a suggestion the
20
+ * operator wires into policy, and the produced value never changes claim status.
21
+ *
22
+ * Run: `node dist/examples/calibrated-auto-accept.js`
23
+ */
24
+ export interface CalibratedAutoAcceptResult {
25
+ readonly suggestedThreshold: number | undefined;
26
+ readonly groupAccuracy: number | undefined;
27
+ readonly producedValues: ReadonlyArray<number | undefined>;
28
+ }
29
+ export declare function runCalibratedAutoAccept(): CalibratedAutoAcceptResult;
@@ -0,0 +1,135 @@
1
+ /**
2
+ * Worked example: wiring confidence calibration into a downstream consumer.
3
+ *
4
+ * A consumer that owns human review outcomes can close the confidence loop in two
5
+ * places, both shipped in @kontourai/survey (1.10.0):
6
+ *
7
+ * 1. Ground the auto-accept threshold. `deriveCalibration` over a history of
8
+ * review outcomes yields `suggestedThreshold` — the lowest confidence at
9
+ * which the extractor's proposals were empirically affirmed often enough.
10
+ * Feed that number into a producer profile's `autoAcceptMinConfidence`
11
+ * (see `SchemaMappingOptions.autoAcceptMinConfidence` /
12
+ * `InquiryMappingOptions.minConfidence`) instead of hand-picking it.
13
+ *
14
+ * 2. Produce calibrated conclusion confidence. Pass the same calibration
15
+ * `metrics` into `buildSurveyTrustBundle({ calibration: { metrics } })` and
16
+ * affirmed claims carry `conclusionConfidence.value` = the empirical
17
+ * affirmation rate for their extractor/field.
18
+ *
19
+ * Calibration is advisory (ADR 0003 §4): the threshold is a suggestion the
20
+ * operator wires into policy, and the produced value never changes claim status.
21
+ *
22
+ * Run: `node dist/examples/calibrated-auto-accept.js`
23
+ */
24
+ import { buildSurveyTrustBundle, deriveCalibration, SurveyInputBuilder, } from "../src/index.js";
25
+ const EXTRACTOR = "example-extractor";
26
+ const FIELD = "registrationStatus";
27
+ const HISTORY_AT = "2026-06-01T00:00:00.000Z";
28
+ const BATCH_AT = "2026-07-01T00:00:00.000Z";
29
+ /**
30
+ * A synthetic history: for this extractor/field, high-confidence proposals were
31
+ * almost always affirmed by reviewers and low-confidence ones were mostly
32
+ * rejected — the pattern that makes an empirical threshold meaningful.
33
+ */
34
+ function buildReviewHistory() {
35
+ const extractions = [];
36
+ const candidateSets = [];
37
+ const reviewOutcomes = [];
38
+ let n = 0;
39
+ const addSamples = (count, confidence, affirmed) => {
40
+ for (let i = 0; i < count; i += 1) {
41
+ const key = `h-${n}`;
42
+ n += 1;
43
+ extractions.push({
44
+ id: `${key}-ext`,
45
+ sourceId: `${key}-src`,
46
+ target: FIELD,
47
+ value: "ACTIVE",
48
+ confidence,
49
+ extractor: EXTRACTOR,
50
+ extractedAt: HISTORY_AT,
51
+ });
52
+ const candidate = { id: `${key}-cand`, extractionId: `${key}-ext`, value: "ACTIVE", confidence };
53
+ candidateSets.push({
54
+ id: `${key}-cs`,
55
+ target: FIELD,
56
+ status: "resolved",
57
+ selectedCandidateId: candidate.id,
58
+ candidates: [candidate],
59
+ });
60
+ reviewOutcomes.push({
61
+ id: `${key}-ro`,
62
+ candidateSetId: `${key}-cs`,
63
+ candidateId: candidate.id,
64
+ status: i < affirmed ? "verified" : "rejected",
65
+ actor: "example-reviewer",
66
+ reviewedAt: HISTORY_AT,
67
+ });
68
+ }
69
+ };
70
+ addSamples(10, 0.95, 10); // top decile: all affirmed
71
+ addSamples(10, 0.85, 10); // 0.8–0.9: all affirmed
72
+ addSamples(10, 0.75, 5); // 0.7–0.8: half affirmed → below target, ends the run
73
+ addSamples(10, 0.55, 1); // 0.5–0.6: mostly rejected
74
+ return { reviewOutcomes, candidateSets, extractions };
75
+ }
76
+ export function runCalibratedAutoAccept() {
77
+ // (1) Derive the empirical calibration curve over the review history.
78
+ const metrics = deriveCalibration(buildReviewHistory(), {
79
+ targetAccuracy: 0.9, // the accuracy the auto-accept threshold must clear
80
+ minBinSamples: 5, // a decile needs this many samples to ground the threshold
81
+ });
82
+ const suggestedThreshold = metrics.overall.suggestedThreshold;
83
+ const group = metrics.byExtractorField.find((g) => g.extractor === EXTRACTOR && g.field === FIELD);
84
+ // This is the number you feed into your producer profile's auto-accept policy:
85
+ // surveySchemaMapping(context, extractor, { autoAcceptMinConfidence: suggestedThreshold })
86
+ // Proposals at/above it were empirically affirmed often enough to auto-accept.
87
+ // (2) Produce calibrated conclusion confidence on a new batch of affirmed claims.
88
+ const input = new SurveyInputBuilder({ source: "example-consumer:calibrated", generatedAt: BATCH_AT })
89
+ .addObservation(affirmedObservation("entity-1", 0.85))
90
+ .addObservation(affirmedObservation("entity-2", 0.92))
91
+ .build();
92
+ // Prefer metrics computed over history (not just this batch), so a claim's own
93
+ // outcome does not feed its own value.
94
+ const bundle = buildSurveyTrustBundle(input, { calibration: { metrics, minSamples: 20 } });
95
+ const producedValues = bundle.claims.map((c) => c.conclusionConfidence?.value);
96
+ return { suggestedThreshold, groupAccuracy: group?.empiricalAccuracy, producedValues };
97
+ }
98
+ function affirmedObservation(subjectId, confidence) {
99
+ return {
100
+ id: `example.${subjectId}.${FIELD}.current`,
101
+ rawSource: {
102
+ kind: "api-record",
103
+ sourceRef: `records://${subjectId}/registry`,
104
+ observedAt: BATCH_AT,
105
+ locatorScheme: "structured-field",
106
+ },
107
+ extraction: {
108
+ target: FIELD,
109
+ value: "ACTIVE",
110
+ confidence,
111
+ locator: "json:$.registrationStatus",
112
+ extractor: EXTRACTOR,
113
+ extractedAt: BATCH_AT,
114
+ },
115
+ reviewOutcome: { status: "verified", actor: "example-reviewer", reviewedAt: BATCH_AT },
116
+ claim: {
117
+ subjectType: "public-record.entity",
118
+ subjectId,
119
+ facet: "public-record.profile",
120
+ claimType: "public-data.field",
121
+ fieldOrBehavior: FIELD,
122
+ impactLevel: "medium",
123
+ collectedBy: EXTRACTOR,
124
+ },
125
+ };
126
+ }
127
+ // Run standalone (not when imported by a test).
128
+ if (process.argv[1]?.endsWith("calibrated-auto-accept.js")) {
129
+ const result = runCalibratedAutoAccept();
130
+ console.log(JSON.stringify({
131
+ suggestedAutoAcceptThreshold: result.suggestedThreshold,
132
+ empiricalAffirmationRate: result.groupAccuracy,
133
+ producedConclusionConfidenceValues: result.producedValues,
134
+ }, null, 2));
135
+ }
@@ -110,7 +110,8 @@ function proposedReviewCandidate(proposal, field, diff, selectedRole) {
110
110
  sourceRef: diff.sourceUrl ?? proposal.sourceUrl,
111
111
  observedAt: proposal.createdAt,
112
112
  excerpt: diff.excerpt ?? `Proposed ${field} value from downstream extraction.`,
113
- extractor: proposal.extractionModel,
113
+ extractor: "downstream-directory-extractor",
114
+ model: proposal.extractionModel,
114
115
  extractedAt: proposal.createdAt,
115
116
  confidence: diff.confidence,
116
117
  sourceRank: selectedRole === "proposed" ? 1 : 2,
@@ -193,6 +194,7 @@ function candidateExtraction(args, extractionId) {
193
194
  target: args.field,
194
195
  confidence: args.confidence,
195
196
  extractor: args.extractor,
197
+ ...(args.model ? { model: args.model } : {}),
196
198
  extractedAt: args.extractedAt,
197
199
  };
198
200
  }
@@ -35,7 +35,16 @@ export interface ExtractionReference {
35
35
  extractionId?: string;
36
36
  target: string;
37
37
  confidence?: number;
38
+ /** The extraction tool/pipeline that produced this candidate (e.g. a crawler or parser). */
38
39
  extractor?: string;
40
+ /**
41
+ * The model or model-version that generated this candidate, when the producer
42
+ * knows it (e.g. an LLM id or a dated extraction-model tag). Distinct from
43
+ * `extractor` (the tool): a reviewer wants to know *which model* proposed a
44
+ * value for trust/calibration. Producer-provided provenance; Survey only
45
+ * carries and displays it.
46
+ */
47
+ model?: string;
39
48
  extractedAt?: string;
40
49
  }
41
50
  export interface ClaimTargetHint {
@@ -130,6 +139,15 @@ export interface ReviewDecisionSpec {
130
139
  * their own admissible block. */
131
140
  authorizing?: ReviewAuthorizing;
132
141
  projection?: SurveyRecordProjectionHint;
142
+ /**
143
+ * Reviewer-edited override for the proposed candidate's value, captured when the
144
+ * reviewer edits the inline proposed-value editor before choosing "Use proposed".
145
+ * Only meaningful when the decision selects the proposed candidate. Downstream
146
+ * consumers should read the effective value as `editedValue ?? <selected candidate value>`
147
+ * rather than assuming the candidate's original value was applied verbatim.
148
+ * Additive/optional: absent means the candidate's original value was used unchanged.
149
+ */
150
+ editedValue?: unknown;
133
151
  }
134
152
  export interface ReviewDecisionStatus {
135
153
  appliedToClaimIds?: string[];
@@ -9,6 +9,12 @@ export interface ReviewWorkbenchState {
9
9
  readonly decision?: ReviewWorkbenchDecision;
10
10
  readonly reviewedAt: string;
11
11
  readonly actorId: string;
12
+ /**
13
+ * Reviewer-edited override for the item's proposed value (inline edit in the
14
+ * field-diff card). Additive/optional: undefined means no edit was made and the
15
+ * proposed candidate's original value applies.
16
+ */
17
+ readonly editedValue?: unknown;
12
18
  }
13
19
  export interface ReviewQueueSessionState {
14
20
  readonly items: readonly ReviewItem[];
@@ -17,6 +23,13 @@ export interface ReviewQueueSessionState {
17
23
  readonly decisionsByItemName: Readonly<Record<string, ReviewWorkbenchDecision>>;
18
24
  readonly reviewedAt: string;
19
25
  readonly actorId: string;
26
+ /**
27
+ * Reviewer-edited overrides for proposed values, keyed by ReviewItem name.
28
+ * Additive/optional: a session built before this field existed behaves exactly
29
+ * as before (every lookup resolves to undefined, meaning "use the candidate's
30
+ * original value").
31
+ */
32
+ readonly editedValuesByItemName?: Readonly<Record<string, unknown>>;
20
33
  }
21
34
  export interface ReviewSessionSummary {
22
35
  readonly accepted: number;
@@ -53,8 +66,25 @@ export declare function deriveQueueRowStatus(item: ReviewItem, session: ReviewQu
53
66
  export declare function nextUnresolvedItemName(session: ReviewQueueSessionState): string | undefined;
54
67
  export declare function reviewSessionSummary(session: ReviewQueueSessionState): ReviewSessionSummary;
55
68
  export declare function candidateForDecision(item: ReviewItem, decision: ReviewWorkbenchDecision): ReviewCandidate;
69
+ /**
70
+ * The value that should actually be applied for a decision: the reviewer's inline
71
+ * edit when one was made for an accept-proposed decision, otherwise the selected
72
+ * candidate's original value. Consumers reading `ReviewWorkbenchResult` should
73
+ * prefer `effectiveValue`/`effectiveDisplayValue`, which are already computed with
74
+ * this rule; this helper exists for callers deriving the value from raw session
75
+ * state directly.
76
+ */
77
+ export declare function effectiveValueForDecision(item: ReviewItem, decision: ReviewWorkbenchDecision, editedValue?: unknown): unknown;
56
78
  export declare function selectedCandidateRole(state: ReviewWorkbenchState): ReviewCandidate["role"] | undefined;
57
79
  export declare function buildReviewSessionResource(session: ReviewQueueSessionState, events?: readonly ReviewSessionEvent[], sessionName?: string): ReviewSession;
58
80
  export declare function buildReviewSessionEvents(session: ReviewQueueSessionState, sessionName?: string): ReviewSessionEvent[];
59
81
  export declare function replayReviewSessionEvents(startState: ReviewQueueSessionState, events: readonly ReviewSessionEvent[]): ReviewQueueSessionState;
82
+ /**
83
+ * Detects the explicit "clear this ReviewItem's decision" replay signal (emitted
84
+ * by the workbench's "Change" / undo control): a decision event whose
85
+ * `data.workbenchDecision` is the literal `null` sentinel, as opposed to `undefined`
86
+ * (no decision info present — event is ignored by replay, same as before this
87
+ * feature existed).
88
+ */
89
+ export declare function isClearedWorkbenchDecisionEvent(event: ReviewSessionEvent): boolean;
60
90
  export declare function buildReviewSessionEvent(session: ReviewQueueSessionState, spec: Omit<ReviewSessionEventSpec, "actor">): ReviewSessionEvent;
@@ -37,6 +37,7 @@ export function initialReviewQueueSessionState(items = reviewWorkbenchQueueExamp
37
37
  activeItemName: items[0]?.metadata.name ?? "",
38
38
  notesByItemName: {},
39
39
  decisionsByItemName: {},
40
+ editedValuesByItemName: {},
40
41
  reviewedAt: "2026-06-04T00:00:00.000Z",
41
42
  actorId: "review-workbench-operator",
42
43
  };
@@ -47,6 +48,7 @@ export function currentReviewWorkbenchState(session) {
47
48
  item,
48
49
  note: session.notesByItemName[item.metadata.name] ?? "",
49
50
  decision: session.decisionsByItemName[item.metadata.name],
51
+ editedValue: session.editedValuesByItemName?.[item.metadata.name],
50
52
  reviewedAt: session.reviewedAt,
51
53
  actorId: session.actorId,
52
54
  };
@@ -117,6 +119,18 @@ export function candidateForDecision(item, decision) {
117
119
  }
118
120
  return candidate;
119
121
  }
122
+ /**
123
+ * The value that should actually be applied for a decision: the reviewer's inline
124
+ * edit when one was made for an accept-proposed decision, otherwise the selected
125
+ * candidate's original value. Consumers reading `ReviewWorkbenchResult` should
126
+ * prefer `effectiveValue`/`effectiveDisplayValue`, which are already computed with
127
+ * this rule; this helper exists for callers deriving the value from raw session
128
+ * state directly.
129
+ */
130
+ export function effectiveValueForDecision(item, decision, editedValue) {
131
+ const candidate = candidateForDecision(item, decision);
132
+ return decision === "accept-proposed" && editedValue !== undefined ? editedValue : candidate.value;
133
+ }
120
134
  export function selectedCandidateRole(state) {
121
135
  if (!state.decision) {
122
136
  return undefined;
@@ -238,6 +252,10 @@ export function replayReviewSessionEvents(startState, events) {
238
252
  }
239
253
  if ((event.spec.eventType === "decision-changed" || event.spec.eventType === "decision-submitted")
240
254
  && event.spec.reviewItemName) {
255
+ if (isClearedWorkbenchDecisionEvent(event)) {
256
+ const { [event.spec.reviewItemName]: _removed, ...remainingDecisions } = session.decisionsByItemName;
257
+ return { ...session, decisionsByItemName: remainingDecisions };
258
+ }
241
259
  const decision = workbenchDecisionFromEvent(event);
242
260
  return decision
243
261
  ? {
@@ -252,6 +270,18 @@ export function replayReviewSessionEvents(startState, events) {
252
270
  return session;
253
271
  }, startState);
254
272
  }
273
+ /**
274
+ * Detects the explicit "clear this ReviewItem's decision" replay signal (emitted
275
+ * by the workbench's "Change" / undo control): a decision event whose
276
+ * `data.workbenchDecision` is the literal `null` sentinel, as opposed to `undefined`
277
+ * (no decision info present — event is ignored by replay, same as before this
278
+ * feature existed).
279
+ */
280
+ export function isClearedWorkbenchDecisionEvent(event) {
281
+ return event.spec.data !== undefined
282
+ && "workbenchDecision" in event.spec.data
283
+ && event.spec.data.workbenchDecision === null;
284
+ }
255
285
  export function buildReviewSessionEvent(session, spec) {
256
286
  return {
257
287
  apiVersion: reviewResourceApiVersion,
@@ -1,4 +1,4 @@
1
- import { candidateForDecision, workbenchDecisionDefinitions, } from "./review-queue-session.js";
1
+ import { candidateForDecision, isClearedWorkbenchDecisionEvent, workbenchDecisionDefinitions, } from "./review-queue-session.js";
2
2
  export function validateReviewSessionEventsForSnapshot(snapshot, events) {
3
3
  const itemsByName = new Map(snapshot.items.map((item) => [item.metadata.name, item]));
4
4
  const sequenceIssues = validateEventSequence(events);
@@ -35,7 +35,12 @@ export function validateReviewSessionEventsForSnapshot(snapshot, events) {
35
35
  message: `ReviewSessionEvent ${event.metadata.name} is a decision event but does not reference a ReviewItem.`,
36
36
  });
37
37
  }
38
- if (event.spec.eventType === "decision-changed" || event.spec.eventType === "decision-submitted") {
38
+ if ((event.spec.eventType === "decision-changed" || event.spec.eventType === "decision-submitted")
39
+ && isClearedWorkbenchDecisionEvent(event)) {
40
+ // Explicit "clear this ReviewItem's decision" signal (undo). No candidate/status
41
+ // expectations apply — the event carries no selected candidate.
42
+ }
43
+ else if (event.spec.eventType === "decision-changed" || event.spec.eventType === "decision-submitted") {
39
44
  const decision = replayableWorkbenchDecision(event.spec.data?.workbenchDecision);
40
45
  if (!decision) {
41
46
  issues.push({