@kontourai/survey 4.0.0 → 5.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -35,12 +35,12 @@ export declare const publicDirectoryReviewItemExample: {
35
35
  excerpt: string;
36
36
  };
37
37
  extraction: {
38
+ model?: undefined;
38
39
  extractionId: string;
39
40
  target: string;
40
41
  confidence: number;
41
42
  extractor: string;
42
43
  extractedAt: string;
43
- model?: undefined;
44
44
  };
45
45
  claimTarget: {
46
46
  claimId: string;
@@ -61,13 +61,13 @@ export declare const publicDirectoryReviewItemExample: {
61
61
  claimId: string;
62
62
  };
63
63
  producer: {
64
+ proposalId?: undefined;
65
+ oldValue?: undefined;
64
66
  sourceAuthority: {
65
67
  authorityClass: string;
66
68
  declaredBy: string;
67
69
  scope: string;
68
70
  };
69
- proposalId?: undefined;
70
- oldValue?: undefined;
71
71
  };
72
72
  } | {
73
73
  id: string;
@@ -24,6 +24,11 @@ export function prepareFacilityCredentialServerApply(input) {
24
24
  if (!currentCandidate || JSON.stringify(currentCandidate.value) !== JSON.stringify(input.currentRecord.credential)) {
25
25
  return { ok: false, message: "Current credential no longer matches the review session snapshot." };
26
26
  }
27
+ // A decision that selects no candidate (reject-all or could-not-confirm on a
28
+ // conflict) has no value to apply.
29
+ if (result.selectedCandidateId === undefined) {
30
+ return { ok: false, message: "Review result selects no candidate, so there is nothing to apply." };
31
+ }
27
32
  if (input.currentRecord.appliedReviewItemNames.includes(result.reviewItemName)) {
28
33
  return { ok: false, message: "Review result was already applied." };
29
34
  }
@@ -1,6 +1,6 @@
1
1
  import { SURVEY_INPUT_CONTRACT_VERSION } from "./types.js";
2
2
  import { canonicalJson } from "./review-workbench/canonical.js";
3
- import { workbenchDecisionDefinitions } from "./review-workbench/review-queue-session.js";
3
+ import { decisionSelectsNoCandidate, workbenchDecisionDefinitions } from "./review-workbench/review-queue-session.js";
4
4
  /**
5
5
  * Projects server-applied review records into the complete SurveyInput consumed
6
6
  * by buildSurveyTrustBundle. The ReviewItem and ReviewWorkbenchResult are the
@@ -33,15 +33,24 @@ export function buildCanonicalReviewedTrustInput(options) {
33
33
  throw new Error(`ReviewItem ${item.metadata.name} has no canonical server-applied result.`);
34
34
  }
35
35
  assertCanonicalResult(item, result);
36
+ // Reject-all or could-not-confirm on a conflict selects no candidate: the
37
+ // set, the review outcome and the claim then name none of its values.
38
+ const selectsNone = decisionSelectsNoCandidate(item, result.decision);
39
+ const rejectsAll = selectsNone && result.decision === "reject-proposed";
40
+ const rejectedRole = workbenchDecisionDefinitions[result.decision].candidateRole;
41
+ const rejectAllReason = result.rationale?.trim() || "Every proposed value for this claim was rejected.";
36
42
  const candidates = item.spec.candidates.map((candidate) => {
37
- const records = projectCandidate(item, candidate);
43
+ const records = projectCandidate(item, rejectsAll && candidate.role === rejectedRole
44
+ ? { ...candidate, rejectionReason: candidate.rejectionReason ?? rejectAllReason }
45
+ : candidate);
38
46
  addConsistent(rawSources, records.rawSource, "raw source");
39
47
  addConsistent(extractions, records.extraction, "extraction");
40
48
  return records.candidate;
41
49
  });
42
- const selected = item.spec.candidates.find((candidate) => candidate.id === result.selectedCandidateId);
43
- const selectedRecordId = selected.projection?.candidateId ?? selected.id;
44
- const candidateSetId = selected.projection?.candidateSetId
50
+ const selected = selectsNone ? undefined : item.spec.candidates.find((candidate) => candidate.id === result.selectedCandidateId);
51
+ const selectedRecordId = selected ? selected.projection?.candidateId ?? selected.id : undefined;
52
+ // Every candidate carries the same candidate-set id (checked below).
53
+ const candidateSetId = (selected ?? item.spec.candidates[0])?.projection?.candidateSetId
45
54
  ?? item.spec.projection?.candidateSetId
46
55
  ?? `${item.metadata.name}.candidates`;
47
56
  if (candidates.some((candidate) => candidate.metadata?.candidateSetId !== candidateSetId)) {
@@ -56,10 +65,10 @@ export function buildCanonicalReviewedTrustInput(options) {
56
65
  ? { metadata: Object.fromEntries(Object.entries(metadata).filter(([key]) => key !== "candidateSetId")) }
57
66
  : {}),
58
67
  })),
59
- selectedCandidateId: selectedRecordId,
68
+ ...(selectedRecordId !== undefined ? { selectedCandidateId: selectedRecordId } : {}),
60
69
  status: result.decision === "could-not-confirm"
61
70
  ? (item.spec.candidateSetStatus ?? "needs-review")
62
- : "resolved",
71
+ : rejectsAll ? "rejected" : "resolved",
63
72
  ...(result.rationale ?? item.spec.rationale
64
73
  ? { rationale: result.rationale ?? item.spec.rationale }
65
74
  : {}),
@@ -67,12 +76,12 @@ export function buildCanonicalReviewedTrustInput(options) {
67
76
  addConsistent(candidateSets, candidateSet, "candidate set");
68
77
  const decision = result.reviewDecision.spec;
69
78
  const reviewOutcomeId = decision.projection?.reviewOutcomeId
70
- ?? selected.projection?.reviewOutcomeId
79
+ ?? (selected ?? item.spec).projection?.reviewOutcomeId
71
80
  ?? `${item.metadata.name}.${result.decision}.review-outcome`;
72
81
  const reviewOutcome = {
73
82
  id: reviewOutcomeId,
74
83
  candidateSetId,
75
- candidateId: selectedRecordId,
84
+ ...(selectedRecordId !== undefined ? { candidateId: selectedRecordId } : {}),
76
85
  status: result.status,
77
86
  ...(decision.resolution ? { resolution: decision.resolution } : {}),
78
87
  ...(decision.resolutionReason ? { resolutionReason: decision.resolutionReason } : {}),
@@ -90,34 +99,42 @@ export function buildCanonicalReviewedTrustInput(options) {
90
99
  },
91
100
  };
92
101
  addConsistent(reviewOutcomes, reviewOutcome, "review outcome");
93
- const hint = selected.claimTarget;
102
+ // Every candidate names the same claim target (checked by assertCanonicalResult).
103
+ const hint = (selected ?? item.spec.candidates[0]).claimTarget;
94
104
  assertSingleProjectionId("claim", item.metadata.name, [
95
105
  decision.projection?.claimId,
96
106
  item.spec.projection?.claimId,
97
107
  ...item.spec.candidates.flatMap((candidate) => [candidate.projection?.claimId, candidate.claimTarget.claimId]),
98
108
  ]);
99
109
  const claimId = decision.projection?.claimId
100
- ?? selected.projection?.claimId
110
+ ?? selected?.projection?.claimId
101
111
  ?? item.spec.projection?.claimId
102
112
  ?? hint.claimId
103
113
  ?? `${item.metadata.name}.claim`;
104
114
  const claim = {
105
115
  id: claimId,
106
116
  candidateSetId,
107
- candidateId: selectedRecordId,
117
+ ...(selectedRecordId !== undefined ? { candidateId: selectedRecordId } : {}),
108
118
  subjectType: hint.subjectType,
109
119
  subjectId: hint.subjectId,
110
120
  facet: hint.facet,
111
121
  claimType: hint.claimType,
112
122
  fieldOrBehavior: hint.fieldOrBehavior,
113
- value: result.effectiveValue,
114
- status: result.status,
123
+ // Surface requires a claim value; a claim that selects none of its
124
+ // candidates carries null, and buildSurveyTrustBundle lists every value.
125
+ value: selectsNone ? null : result.effectiveValue,
126
+ // Could-not-confirm keeps the pre-review posture, and a conflicting or
127
+ // escalated candidate set is disputed before any review (see
128
+ // docs/decisions/could-not-confirm.md), never merely proposed.
129
+ status: result.decision === "could-not-confirm" && (candidateSet.status === "conflict" || candidateSet.status === "escalated")
130
+ ? "disputed"
131
+ : result.status,
115
132
  impactLevel: hint.impactLevel,
116
133
  updatedAt: decision.reviewedAt ?? options.generatedAt,
117
134
  ...(hint.evidenceType ? { evidenceType: hint.evidenceType } : {}),
118
135
  ...(hint.evidenceMethod ? { evidenceMethod: hint.evidenceMethod } : {}),
119
136
  ...(hint.derivedFrom ? { derivedFrom: [...hint.derivedFrom] } : {}),
120
- collectedBy: hint.collectedBy ?? selected.extraction.extractor ?? options.source,
137
+ collectedBy: hint.collectedBy ?? sharedExtractor(selected ? [selected] : item.spec.candidates) ?? options.source,
121
138
  ...(decision.actor?.id ? { actor: decision.actor.id } : {}),
122
139
  };
123
140
  addConsistent(claims, claim, "claim target");
@@ -137,24 +154,38 @@ export function buildCanonicalReviewedTrustInput(options) {
137
154
  };
138
155
  }
139
156
  function assertCanonicalResult(item, result) {
140
- const selected = item.spec.candidates.find((candidate) => candidate.id === result.selectedCandidateId);
141
- if (!selected) {
142
- throw new Error(`Review result ${result.reviewItemName} selects an unknown candidate.`);
143
- }
144
- if (canonicalJson(selected) !== canonicalJson(result.selectedCandidate)) {
145
- throw new Error(`Review result ${result.reviewItemName} selected candidate does not match its canonical ReviewItem.`);
146
- }
147
- const unselected = item.spec.candidates.filter((candidate) => candidate.id !== selected.id);
148
- if (canonicalJson(unselected) !== canonicalJson(result.unselectedCandidates)) {
149
- throw new Error(`Review result ${result.reviewItemName} unselected candidates do not match its canonical ReviewItem.`);
157
+ const selectsNone = decisionSelectsNoCandidate(item, result.decision);
158
+ let selected;
159
+ if (selectsNone) {
160
+ if (result.selectedCandidate !== undefined || result.selectedCandidateId !== undefined || result.selectedCandidateRole !== undefined
161
+ || result.selectedValue !== undefined || result.selectedDisplayValue !== undefined
162
+ || result.effectiveValue !== undefined || result.effectiveDisplayValue !== undefined || result.editedValue !== undefined) {
163
+ throw new Error(`Review result ${result.reviewItemName} names a selected value, but its ${result.decision} decision selects no candidate.`);
164
+ }
165
+ if (canonicalJson(item.spec.candidates) !== canonicalJson(result.unselectedCandidates)) {
166
+ throw new Error(`Review result ${result.reviewItemName} unselected candidates do not match its canonical ReviewItem.`);
167
+ }
150
168
  }
151
- if (result.selectedCandidateRole !== selected.role || canonicalJson(result.selectedValue) !== canonicalJson(selected.value)) {
152
- throw new Error(`Review result ${result.reviewItemName} selected identity does not match its canonical ReviewItem.`);
169
+ else {
170
+ selected = item.spec.candidates.find((candidate) => candidate.id === result.selectedCandidateId);
171
+ if (!selected) {
172
+ throw new Error(`Review result ${result.reviewItemName} selects an unknown candidate.`);
173
+ }
174
+ if (canonicalJson(selected) !== canonicalJson(result.selectedCandidate)) {
175
+ throw new Error(`Review result ${result.reviewItemName} selected candidate does not match its canonical ReviewItem.`);
176
+ }
177
+ const unselected = item.spec.candidates.filter((candidate) => candidate.id !== selected.id);
178
+ if (canonicalJson(unselected) !== canonicalJson(result.unselectedCandidates)) {
179
+ throw new Error(`Review result ${result.reviewItemName} unselected candidates do not match its canonical ReviewItem.`);
180
+ }
181
+ if (result.selectedCandidateRole !== selected.role || canonicalJson(result.selectedValue) !== canonicalJson(selected.value)) {
182
+ throw new Error(`Review result ${result.reviewItemName} selected identity does not match its canonical ReviewItem.`);
183
+ }
153
184
  }
154
185
  const decision = result.reviewDecision.spec;
155
186
  const definition = workbenchDecisionDefinitions[result.decision];
156
187
  if (decision.reviewItemName !== item.metadata.name
157
- || decision.candidateId !== result.selectedCandidateId
188
+ || decision.candidateId !== selected?.id
158
189
  || decision.status !== result.status
159
190
  || decision.status !== definition.status
160
191
  || decision.rationale !== result.rationale
@@ -164,14 +195,17 @@ function assertCanonicalResult(item, result) {
164
195
  if ((result.decision === "could-not-confirm") !== (decision.resolution === "could_not_confirm")) {
165
196
  throw new Error(`Review result ${result.reviewItemName} contradicts its canonical review resolution.`);
166
197
  }
167
- const expectedEffective = result.editedValue !== undefined && result.decision === "accept-proposed"
168
- ? result.editedValue
169
- : selected.value;
170
- if (canonicalJson(expectedEffective) !== canonicalJson(result.effectiveValue)) {
171
- throw new Error(`Review result ${result.reviewItemName} effective value is not canonical.`);
198
+ if (selected) {
199
+ const expectedEffective = result.editedValue !== undefined && result.decision === "accept-proposed"
200
+ ? result.editedValue
201
+ : selected.value;
202
+ if (canonicalJson(expectedEffective) !== canonicalJson(result.effectiveValue)) {
203
+ throw new Error(`Review result ${result.reviewItemName} effective value is not canonical.`);
204
+ }
172
205
  }
206
+ const reference = selected ?? item.spec.candidates[0];
173
207
  for (const candidate of item.spec.candidates) {
174
- if (canonicalJson(claimTargetIdentity(candidate.claimTarget)) !== canonicalJson(claimTargetIdentity(selected.claimTarget))) {
208
+ if (canonicalJson(claimTargetIdentity(candidate.claimTarget)) !== canonicalJson(claimTargetIdentity(reference.claimTarget))) {
175
209
  throw new Error(`ReviewItem ${item.metadata.name} candidates carry conflicting claim targets.`);
176
210
  }
177
211
  }
@@ -263,3 +297,8 @@ function assertSingleProjectionId(label, itemName, values) {
263
297
  throw new Error(`ReviewItem ${itemName} carries conflicting ${label} projection ids.`);
264
298
  }
265
299
  }
300
+ /** The extractor every candidate shares, or undefined when they differ. */
301
+ function sharedExtractor(candidates) {
302
+ const extractors = new Set(candidates.map((candidate) => candidate.extraction.extractor));
303
+ return extractors.size === 1 ? [...extractors][0] : undefined;
304
+ }
@@ -38,7 +38,12 @@ export interface PortableExtractionEvidenceMatch {
38
38
  export interface PortableExtractionProposal {
39
39
  fieldPath: string;
40
40
  candidateValue: unknown;
41
- confidence: number;
41
+ /**
42
+ * The proposer's own uncalibrated self-report in `0..1`, absent when the
43
+ * proposer reported none. Survey carries it only when present and never
44
+ * substitutes a default.
45
+ */
46
+ confidence?: number;
42
47
  provenance: {
43
48
  excerpt: string;
44
49
  locator: string;
@@ -71,6 +76,25 @@ export type PortablePreparedArtifactState = {
71
76
  actualDigest: string;
72
77
  actualContentLength: number;
73
78
  };
79
+ /**
80
+ * Why a run stopped short. The first four are early stops; the last three mean
81
+ * a dispatched chunk was not (fully) read or answered, and such an envelope
82
+ * names the affected ranges in `result.coverage`.
83
+ */
84
+ export type PortableExtractionPartialReason = "cancelled" | "max-provider-calls" | "max-total-tokens" | "max-chunks" | "provider-failure" | "content-truncated" | "output-truncated";
85
+ /**
86
+ * One prepared-text range (UTF-16 offsets, the space of `chars:` locators) and
87
+ * whether it was read and answered. `reason` is present exactly when `status`
88
+ * is `unread`, and then covers exactly the unread span.
89
+ */
90
+ export interface PortableExtractionCoverageEntry {
91
+ /** 1-based chunk number. */
92
+ chunk: number;
93
+ start: number;
94
+ end: number;
95
+ status: "complete" | "unread" | "output-truncated";
96
+ reason?: "provider-failure" | "content-truncated" | "missing-tool-call" | "not-dispatched";
97
+ }
74
98
  export interface PortableExtractionResultEnvelope {
75
99
  format: typeof portableExtractionResultFormat;
76
100
  version: typeof portableExtractionResultVersion;
@@ -90,7 +114,7 @@ export interface PortableExtractionResultEnvelope {
90
114
  status: "success";
91
115
  } | {
92
116
  status: "partial";
93
- reason: "cancelled" | "max-provider-calls" | "max-total-tokens" | "max-chunks";
117
+ reason: PortableExtractionPartialReason;
94
118
  } | {
95
119
  status: "failure";
96
120
  category: "invalid-config" | "invalid-task" | "preparation" | "provider" | "unexpected";
@@ -104,11 +128,13 @@ export interface PortableExtractionResultEnvelope {
104
128
  providerCalls: number;
105
129
  totalTokensUsed: number;
106
130
  partial?: {
107
- reason: "cancelled" | "max-provider-calls" | "max-total-tokens" | "max-chunks";
131
+ reason: PortableExtractionPartialReason;
108
132
  completedChunks: number;
109
133
  remainingChunks: number;
110
134
  tokenOvershoot?: number;
111
135
  };
136
+ /** Per-chunk read coverage, ordered by `start`; ranges may overlap. Requires `preparedArtifact`. */
137
+ coverage?: PortableExtractionCoverageEntry[];
112
138
  /** `code` is the upstream error code, informational only; `kind` stays authoritative. */
113
139
  providerFailures?: Array<{
114
140
  provider: string;
@@ -143,6 +169,13 @@ export interface ExtractionEnvelopeImportOptions {
143
169
  /** Survey meaning is supplied at the boundary; it is not added to the upstream wire contract. */
144
170
  claimTarget: (proposal: PortableExtractionProposal, index: number) => ClaimTargetHint;
145
171
  }
172
+ /**
173
+ * Why an import is `unresolved`: its prepared artifact did not resolve, or the
174
+ * extraction itself produced nothing reviewable. `extraction-failed` is a
175
+ * failure outcome; `extraction-incomplete` is a partial outcome that proposed
176
+ * nothing, so no candidate can carry the reason. Either way the import is not
177
+ * a complete run that found no values.
178
+ */
146
179
  export type ExtractionEnvelopeImportDiagnostic = {
147
180
  kind: "artifact-unavailable";
148
181
  status: "unavailable" | "storage-error" | "identity-mismatch" | "invalid-artifact";
@@ -154,6 +187,15 @@ export type ExtractionEnvelopeImportDiagnostic = {
154
187
  expectedDigest: string;
155
188
  actualDigest: string;
156
189
  message: string;
190
+ } | {
191
+ kind: "extraction-failed";
192
+ category: string;
193
+ code: string;
194
+ message: string;
195
+ } | {
196
+ kind: "extraction-incomplete";
197
+ reason: PortableExtractionPartialReason;
198
+ message: string;
157
199
  };
158
200
  export interface ExtractionEnvelopeImport {
159
201
  apiVersion: typeof extractionEnvelopeImportApiVersion;
@@ -182,6 +224,20 @@ export interface ExtractionEnvelopeResolutionIdentity {
182
224
  }
183
225
  /** Parse an untrusted upstream document and create Survey's durable import projection. */
184
226
  export declare function importExtractionEnvelope(serialized: string | PortableExtractionResultEnvelope, options: ExtractionEnvelopeImportOptions): ExtractionEnvelopeImportResult;
227
+ /**
228
+ * One ReviewItem per claim slot: every proposal whose claim target names the
229
+ * same claim (subject, facet, claim type, field or behavior, and claim id when
230
+ * one is set) at the same `pathIndices` is one candidate set, so two values for
231
+ * one claim can never be accepted as two separate verified claims. The slot is
232
+ * the claim the proposal would project to, not the upstream `fieldPath` (two
233
+ * field paths mapped to one claim are one slot) and not the value.
234
+ *
235
+ * Within a slot there is one candidate per distinct canonical value; further
236
+ * proposals of an already-seen value are recorded on that candidate as
237
+ * `sameValueProposals`. Two or more distinct values make the set `conflict`.
238
+ * Distinct `pathIndices` (array items) or distinct claim ids are separate slots,
239
+ * which is how a producer declares a multi-valued field.
240
+ */
185
241
  export declare function buildReviewItemsFromExtractionEnvelopeImport(record: ExtractionEnvelopeImport): ReviewItem[];
186
242
  export declare function exportExtractionEnvelopeImport(record: ExtractionEnvelopeImport): string;
187
243
  export declare function reimportExtractionEnvelope(serialized: string): ExtractionEnvelopeImport;
@@ -36,11 +36,48 @@ export function importExtractionEnvelope(serialized, options) {
36
36
  };
37
37
  return { record, reviewItems: buildReviewItemsFromExtractionEnvelopeImport(record) };
38
38
  }
39
+ /**
40
+ * One ReviewItem per claim slot: every proposal whose claim target names the
41
+ * same claim (subject, facet, claim type, field or behavior, and claim id when
42
+ * one is set) at the same `pathIndices` is one candidate set, so two values for
43
+ * one claim can never be accepted as two separate verified claims. The slot is
44
+ * the claim the proposal would project to, not the upstream `fieldPath` (two
45
+ * field paths mapped to one claim are one slot) and not the value.
46
+ *
47
+ * Within a slot there is one candidate per distinct canonical value; further
48
+ * proposals of an already-seen value are recorded on that candidate as
49
+ * `sameValueProposals`. Two or more distinct values make the set `conflict`.
50
+ * Distinct `pathIndices` (array items) or distinct claim ids are separate slots,
51
+ * which is how a producer declares a multi-valued field.
52
+ */
39
53
  export function buildReviewItemsFromExtractionEnvelopeImport(record) {
40
54
  validateImport(record);
41
55
  if (record.status.state !== "grounded")
42
56
  return [];
43
- return record.spec.envelope.result.proposals.map((proposal, index) => buildReviewItem(record, proposal, index));
57
+ return claimSlotGroups(record).map((group) => buildReviewItem(record, group));
58
+ }
59
+ function claimSlotGroups(record) {
60
+ const groups = new Map();
61
+ record.spec.envelope.result.proposals.forEach((proposal, index) => {
62
+ const target = record.spec.claimTargets[index];
63
+ const slot = {
64
+ subjectType: target.subjectType, subjectId: target.subjectId, facet: target.facet, claimType: target.claimType,
65
+ fieldOrBehavior: target.fieldOrBehavior, claimId: target.claimId ?? null, pathIndices: proposal.pathIndices ?? null,
66
+ };
67
+ const key = canonicalJson(slot);
68
+ const group = groups.get(key);
69
+ if (!group) {
70
+ groups.set(key, { slot, members: [{ proposal, index }] });
71
+ return;
72
+ }
73
+ const first = group.members[0].index;
74
+ // One claim cannot carry two impact levels or evidence descriptions.
75
+ if (canonicalJson(record.spec.claimTargets[first]) !== canonicalJson(target)) {
76
+ throw new Error(`Proposals ${first} and ${index} map to the same claim with different claim targets.`);
77
+ }
78
+ group.members.push({ proposal, index });
79
+ });
80
+ return [...groups.values()];
44
81
  }
45
82
  export function exportExtractionEnvelopeImport(record) {
46
83
  validateImport(record);
@@ -66,15 +103,50 @@ export function createExtractionEnvelopeResolutionIdentity(record, proposalIndex
66
103
  const nonce = globalThis.crypto.randomUUID();
67
104
  return { evidenceId: `survey.extraction.${base}.resolution-evidence.${nonce}`, eventId: `survey.extraction.${base}.resolution-event.${nonce}` };
68
105
  }
69
- function buildReviewItem(record, proposal, index) {
106
+ function buildReviewItem(record, group) {
107
+ const envelope = record.spec.envelope;
108
+ const byValue = new Map();
109
+ for (const member of group.members) {
110
+ const key = canonicalJson(member.proposal.candidateValue);
111
+ byValue.set(key, [...(byValue.get(key) ?? []), member]);
112
+ }
113
+ const candidates = [...byValue.values()].map((members) => buildCandidate(record, members));
114
+ const lead = group.members[0].proposal;
115
+ const valueType = lead.valueType ?? inferValueType(lead.candidateValue);
116
+ const identity = identityHash({ producerNamespace: record.metadata.producerNamespace, importName: record.metadata.name, source: envelope.source,
117
+ preparedArtifact: envelope.result.preparedArtifact, pdfLayout: envelope.result.pdfLayout, runId: envelope.result.runId, claimSlot: group.slot });
118
+ return {
119
+ apiVersion: reviewResourceApiVersion, kind: "ReviewItem",
120
+ metadata: { name: `extraction-envelope.${identity}`, producer: { "survey.kontourai.io/extraction-envelope": {
121
+ importName: record.metadata.name,
122
+ evidenceId: `survey.extraction.${identityHash(evidenceInputs(record, lead))}.source-evidence`,
123
+ proposalIndices: group.members.map((member) => member.index),
124
+ source: envelope.source,
125
+ ...(envelope.result.preparedArtifact ? { preparedArtifact: envelope.result.preparedArtifact } : {}),
126
+ } } },
127
+ spec: {
128
+ target: lead.fieldPath, candidates,
129
+ candidateSetStatus: candidates.length > 1 ? "conflict" : "needs-review",
130
+ valueDescriptor: { type: valueType }, editable: false,
131
+ },
132
+ status: { observedCandidateCount: candidates.length },
133
+ };
134
+ }
135
+ /** One candidate for one distinct value; `members` are that value's proposals in envelope order. */
136
+ function buildCandidate(record, members) {
70
137
  const envelope = record.spec.envelope;
71
- const identity = identityHash(identityInputs(record, proposal, index));
138
+ const { proposal, index } = members[0];
139
+ const others = members.slice(1);
140
+ // A candidate commits to every proposal it stands for; a single-proposal
141
+ // candidate keeps exactly the identity it had before grouping.
142
+ const identity = identityHash(others.length === 0 ? identityInputs(record, proposal, index)
143
+ : { ...identityInputs(record, proposal, index), sameValueProposals: others.map((other) => ({ proposalIndex: other.index, proposal: other.proposal })) });
72
144
  const evidence = identityHash(evidenceInputs(record, proposal));
73
145
  const target = record.spec.claimTargets[index];
74
146
  const valueType = proposal.valueType ?? inferValueType(proposal.candidateValue);
75
- const candidate = {
147
+ return {
76
148
  id: `extraction-envelope.${identity}.proposed`, role: "proposed", value: proposal.candidateValue,
77
- confidence: proposal.confidence,
149
+ ...(proposal.confidence !== undefined ? { confidence: proposal.confidence } : {}),
78
150
  source: {
79
151
  sourceRef: envelope.source.ref,
80
152
  sourceId: envelope.source.snapshotRef ?? envelope.source.ref,
@@ -87,8 +159,9 @@ function buildReviewItem(record, proposal, index) {
87
159
  extraction: {
88
160
  extractionId: `extraction-envelope.${identity}`,
89
161
  target: proposal.fieldPath,
90
- confidence: proposal.confidence,
162
+ ...(proposal.confidence !== undefined ? { confidence: proposal.confidence } : {}),
91
163
  extractor: proposal.extractor,
164
+ extractedAt: envelope.result.extractedAt,
92
165
  // The proposal's own served model when recorded; in a multi-chunk run
93
166
  // `result.model` names only the last chunk's model.
94
167
  ...(proposal.producedBy ? { model: proposal.producedBy.model } : envelope.result.model ? { model: envelope.result.model } : {}),
@@ -105,22 +178,21 @@ function buildReviewItem(record, proposal, index) {
105
178
  ...(proposal.producedBy ? { producedBy: proposal.producedBy } : {}),
106
179
  ...(proposal.evidenceMatch ? { evidenceMatch: proposal.evidenceMatch } : {}),
107
180
  occurrence: proposal.provenance.occurrence,
181
+ ...(others.length ? { sameValueProposals: others.map((other) => ({
182
+ proposalIndex: other.index,
183
+ evidenceId: `survey.extraction.${identityHash(evidenceInputs(record, other.proposal))}.source-evidence`,
184
+ locator: other.proposal.provenance.locator,
185
+ excerpt: other.proposal.provenance.excerpt,
186
+ })) } : {}),
108
187
  attempt: { id: envelope.result.runId, providerCalls: envelope.result.providerCalls },
109
188
  ...(envelope.result.warningClassifications ? { warnings: envelope.result.warningClassifications } : {}),
110
189
  outcome: envelope.result.outcome,
190
+ // A partial run may have left this field's other values unread: the
191
+ // reason and the per-chunk coverage travel with every candidate.
192
+ ...(envelope.result.partial ? { partial: envelope.result.partial } : {}),
193
+ ...(envelope.result.coverage ? { coverage: envelope.result.coverage } : {}),
111
194
  } },
112
195
  };
113
- return {
114
- apiVersion: reviewResourceApiVersion, kind: "ReviewItem",
115
- metadata: { name: `extraction-envelope.${identity}`, producer: { "survey.kontourai.io/extraction-envelope": {
116
- importName: record.metadata.name,
117
- evidenceId: `survey.extraction.${evidence}.source-evidence`,
118
- source: envelope.source,
119
- ...(envelope.result.preparedArtifact ? { preparedArtifact: envelope.result.preparedArtifact } : {}),
120
- } } },
121
- spec: { target: proposal.fieldPath, candidates: [candidate], candidateSetStatus: "needs-review", valueDescriptor: { type: valueType }, editable: false },
122
- status: { observedCandidateCount: 1 },
123
- };
124
196
  }
125
197
  function identityInputs(record, proposal, index) {
126
198
  return { producerNamespace: record.metadata.producerNamespace, importName: record.metadata.name, source: record.spec.envelope.source,
@@ -136,6 +208,9 @@ function evidenceInputs(record, proposal) {
136
208
  provenance: proposal.provenance };
137
209
  }
138
210
  function diagnosticsFor(envelope) {
211
+ return [...artifactDiagnostics(envelope), ...outcomeDiagnostics(envelope)];
212
+ }
213
+ function artifactDiagnostics(envelope) {
139
214
  const state = envelope.result.preparedArtifactState;
140
215
  if (!state || state.status === "available")
141
216
  return [];
@@ -147,6 +222,21 @@ function diagnosticsFor(envelope) {
147
222
  ...(state.status === "invalid-artifact" ? { artifactRef: state.canonicalRef } : { artifactRef: state.requestedRef }),
148
223
  message: `Prepared artifact resolution is ${state.status}.` }];
149
224
  }
225
+ /**
226
+ * A failed run, or a partial run with no proposal, must not look like a
227
+ * complete run that found nothing. A partial run with proposals stays
228
+ * grounded: its reason and coverage travel on every candidate.
229
+ */
230
+ function outcomeDiagnostics(envelope) {
231
+ const outcome = envelope.result.outcome;
232
+ if (outcome.status === "failure")
233
+ return [{ kind: "extraction-failed", category: outcome.category, code: outcome.code,
234
+ message: `Extraction failed (${outcome.category}/${outcome.code}); no usable answer was recorded for this source, so the import has no candidates.` }];
235
+ if (outcome.status === "partial" && envelope.result.proposals.length === 0)
236
+ return [{ kind: "extraction-incomplete", reason: outcome.reason,
237
+ message: `Extraction stopped short (${outcome.reason}) without proposing any value; unread text may hold values.` }];
238
+ return [];
239
+ }
150
240
  function validateImport(value) {
151
241
  jsonSafe(value, "Extraction envelope import");
152
242
  const record = obj(value, "Extraction envelope import");
@@ -189,7 +279,7 @@ function validateEnvelope(input) {
189
279
  if (source.snapshotRef !== undefined)
190
280
  safeReference(source.snapshotRef, "source.snapshotRef");
191
281
  const r = obj(e.result, "result");
192
- exact(r, ["proposals", "provider", "runId", "raw", "outcome", "extractedAt", "providerCalls", "totalTokensUsed"], "result", ["model", "warningClassifications", "partial", "providerFailures", "taskDigest", "exampleDigests", "pdfPageOffsets", "pdfLayout", "ocrDerived", "preparedArtifact", "preparedArtifactState"]);
282
+ exact(r, ["proposals", "provider", "runId", "raw", "outcome", "extractedAt", "providerCalls", "totalTokensUsed"], "result", ["model", "warningClassifications", "partial", "coverage", "providerFailures", "taskDigest", "exampleDigests", "pdfPageOffsets", "pdfLayout", "ocrDerived", "preparedArtifact", "preparedArtifactState"]);
193
283
  stableIdentity(r.provider, "result.provider");
194
284
  if (r.model !== undefined)
195
285
  stableIdentity(r.model, "result.model");
@@ -224,6 +314,9 @@ function validateEnvelope(input) {
224
314
  : artifact === undefined
225
315
  ? (() => { throw new Error("result.pdfLayout requires result.preparedArtifact."); })()
226
316
  : validatePortablePdfLayout(r.pdfLayout, artifact.contentLength);
317
+ if (r.coverage !== undefined)
318
+ validateCoverage(r.coverage, artifact);
319
+ validateCoverageAgreement(r.outcome, r.coverage);
227
320
  const state = r.preparedArtifactState === undefined ? undefined : validateArtifactState(r.preparedArtifactState, artifact);
228
321
  if (source.snapshotRef !== undefined && artifact?.sourceSnapshotRef !== undefined && source.snapshotRef !== artifact.sourceSnapshotRef)
229
322
  throw new Error("Source snapshot identity mismatch.");
@@ -232,10 +325,12 @@ function validateEnvelope(input) {
232
325
  }
233
326
  function validateProposal(input, index, contentLength) {
234
327
  const p = obj(input, `proposal[${index}]`);
235
- exact(p, ["fieldPath", "candidateValue", "confidence", "provenance", "extractor"], `proposal[${index}]`, ["pathIndices", "inferenceType", "valueType", "enumValues", "producedBy", "evidenceMatch"]);
328
+ exact(p, ["fieldPath", "candidateValue", "provenance", "extractor"], `proposal[${index}]`, ["confidence", "pathIndices", "inferenceType", "valueType", "enumValues", "producedBy", "evidenceMatch"]);
236
329
  wireNonEmpty(p.fieldPath, "proposal.fieldPath");
237
330
  stableIdentity(p.extractor, "proposal.extractor");
238
- finite(p.confidence, "proposal.confidence", 0, 1);
331
+ // Absent means the proposer reported none; `null` is not absent and is rejected.
332
+ if (Object.hasOwn(p, "confidence"))
333
+ finite(p.confidence, "proposal.confidence", 0, 1);
239
334
  const provenance = obj(p.provenance, "proposal.provenance");
240
335
  exact(provenance, ["excerpt", "locator", "occurrence"], "proposal.provenance");
241
336
  wireNonEmpty(provenance.excerpt, "proposal.provenance.excerpt");
@@ -323,6 +418,45 @@ else if (o.status === "failure") {
323
418
  else
324
419
  throw new Error("outcome status is invalid."); if (o.status !== "partial" && partial !== undefined)
325
420
  throw new Error("partial requires partial outcome."); }
421
+ function validateCoverage(input, artifact) {
422
+ if (!artifact)
423
+ throw new Error("result.coverage requires result.preparedArtifact.");
424
+ const entries = array(input, "result.coverage");
425
+ entries.forEach((value, index) => {
426
+ const subject = `result.coverage[${index}]`;
427
+ const c = obj(value, subject);
428
+ exact(c, ["chunk", "start", "end", "status"], subject, ["reason"]);
429
+ integer(c.chunk, `${subject}.chunk`);
430
+ if (c.chunk === 0)
431
+ throw new Error(`${subject}.chunk must be positive.`);
432
+ integer(c.start, `${subject}.start`);
433
+ integer(c.end, `${subject}.end`);
434
+ if (c.end <= c.start)
435
+ throw new Error(`${subject} must have start < end.`);
436
+ if (c.end > artifact.contentLength)
437
+ throw new Error(`${subject}.end exceeds the prepared artifact contentLength.`);
438
+ if (!COVERAGE_STATUSES.has(c.status))
439
+ throw new Error(`${subject}.status is invalid.`);
440
+ if ((c.status === "unread") !== (c.reason !== undefined))
441
+ throw new Error(`${subject}.reason is required exactly when status is unread.`);
442
+ if (c.reason !== undefined && !COVERAGE_REASONS.has(c.reason))
443
+ throw new Error(`${subject}.reason is invalid.`);
444
+ // Chunks overlap by design, so ranges may overlap; only the order is fixed.
445
+ if (index > 0 && c.start < (entries[index - 1].start))
446
+ throw new Error("result.coverage entries must be ordered by start.");
447
+ });
448
+ }
449
+ /** The outcome and the coverage must tell the same story about unread text. */
450
+ function validateCoverageAgreement(outcome, coverage) {
451
+ const lost = coverage?.some((entry) => entry.status !== "complete") ?? false;
452
+ if (outcome.status === "success" && lost)
453
+ throw new Error("result.coverage names unread or unanswered text, but the outcome is success.");
454
+ // A loss reason means a dispatched chunk lost text; a never-dispatched range
455
+ // is an early stop, not that loss.
456
+ const dispatchedLoss = coverage?.some((entry) => entry.status !== "complete" && entry.reason !== "not-dispatched") ?? false;
457
+ if (outcome.status === "partial" && LOSS_PARTIAL.has(outcome.reason) && !dispatchedLoss)
458
+ throw new Error(`partial reason ${outcome.reason} requires a result.coverage entry for a dispatched chunk that was not read or answered.`);
459
+ }
326
460
  function validateWarning(v) { const w = obj(v, "warning"); exact(w, ["category", "code"], "warning"); if (!WARNING_CATEGORIES.has(w.category))
327
461
  throw new Error("warning category invalid."); stableIdentity(w.code, "warning.code"); }
328
462
  function validateFailure(v) { const f = obj(v, "providerFailure"); exact(f, ["provider", "kind", "retryable"], "providerFailure", ["code"]); stableIdentity(f.provider, "failure.provider"); if (!FAILURE_KINDS.has(f.kind) || typeof f.retryable !== "boolean")
@@ -447,7 +581,11 @@ function isWellFormedUnicode(value) { for (let index = 0; index < value.length;
447
581
  } return true; }
448
582
  const RAW_SOURCE_KINDS = new Set(["uploaded-document", "web-page", "api-record", "manual-entry", "policy-standard", "inquiry-question", "agent-utterance", "system-schema"]);
449
583
  const VALUE_TYPES = new Set(["string", "number", "boolean", "date", "enum", "array", "object"]);
450
- const PARTIAL = new Set(["cancelled", "max-provider-calls", "max-total-tokens", "max-chunks"]);
584
+ /** Partial reasons meaning a dispatched chunk was not (fully) read or answered. */
585
+ const LOSS_PARTIAL = new Set(["provider-failure", "content-truncated", "output-truncated"]);
586
+ const PARTIAL = new Set(["cancelled", "max-provider-calls", "max-total-tokens", "max-chunks", ...LOSS_PARTIAL]);
587
+ const COVERAGE_STATUSES = new Set(["complete", "unread", "output-truncated"]);
588
+ const COVERAGE_REASONS = new Set(["provider-failure", "content-truncated", "missing-tool-call", "not-dispatched"]);
451
589
  const FAILURE_CATEGORIES = new Set(["invalid-config", "invalid-task", "preparation", "provider", "unexpected"]);
452
590
  const WARNING_CATEGORIES = new Set(["provider", "normalization", "preparation", "limit", "storage", "content", "other"]);
453
591
  const FAILURE_KINDS = new Set(["authentication", "rate-limit", "timeout", "invalid-request", "unavailable", "unknown"]);