@kontourai/survey 0.4.24 → 0.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/README.md +10 -10
  2. package/dist/example-data/corrected-document-candidates.d.ts +2 -0
  3. package/dist/{fixtures → example-data}/corrected-document-candidates.js +3 -3
  4. package/dist/{fixtures → example-data}/downstream-public-directory-proposal.d.ts +1 -1
  5. package/dist/{fixtures → example-data}/downstream-public-directory-proposal.js +1 -1
  6. package/dist/{fixtures → example-data}/public-directory-review-resource.d.ts +2 -2
  7. package/dist/{fixtures → example-data}/public-directory-review-resource.js +2 -2
  8. package/dist/example-data/public-field-review.d.ts +2 -0
  9. package/dist/{fixtures → example-data}/public-field-review.js +2 -2
  10. package/dist/{fixtures → example-data}/regulated-document-review-resource.d.ts +1 -1
  11. package/dist/{fixtures → example-data}/regulated-document-review-resource.js +3 -3
  12. package/dist/examples/public-field-observation.js +4 -4
  13. package/dist/examples/review-workbench/downstream-public-directory-adapter.d.ts +1 -1
  14. package/dist/examples/review-workbench/facility-credential-consumer.d.ts +2 -2
  15. package/dist/examples/review-workbench/facility-credential-consumer.js +8 -8
  16. package/dist/examples/review-workbench/server-apply-consumer.d.ts +1 -1
  17. package/dist/examples/review-workbench/server-apply-consumer.js +3 -3
  18. package/dist/src/agent-utterance.d.ts +166 -0
  19. package/dist/src/agent-utterance.js +373 -0
  20. package/dist/src/anthropic.d.ts +104 -0
  21. package/dist/src/anthropic.js +383 -0
  22. package/dist/src/index.d.ts +8 -2
  23. package/dist/src/index.js +4 -1
  24. package/dist/src/inquiry-mapping.d.ts +256 -0
  25. package/dist/src/inquiry-mapping.js +385 -0
  26. package/dist/src/review-workbench/review-queue-session.js +3 -3
  27. package/dist/src/review-workbench/review-surface-preview.js +1 -1
  28. package/dist/src/review-workbench/review-workbench-data.d.ts +4 -4
  29. package/dist/src/review-workbench/review-workbench-data.js +14 -14
  30. package/dist/src/schema-mapping.d.ts +196 -0
  31. package/dist/src/schema-mapping.js +486 -0
  32. package/dist/src/to-surface.d.ts +3 -3
  33. package/dist/src/to-surface.js +1 -1
  34. package/dist/src/types.d.ts +2 -2
  35. package/package.json +20 -7
  36. package/dist/fixtures/corrected-document-candidates.d.ts +0 -2
  37. package/dist/fixtures/public-field-review.d.ts +0 -2
@@ -0,0 +1,196 @@
1
+ /**
2
+ * Schema-mapping producer profile — EVIDENCED-ONTOLOGY layer.
3
+ *
4
+ * Every other semantic-layer mapping is unaudited config. Here every mapping
5
+ * shows its work: proposals carry schema-doc evidence, confidence, and
6
+ * rationale; they flow through Survey's candidate → review machinery before
7
+ * any mapping is accepted. Nothing here silently decides (ADR 0003 §4).
8
+ *
9
+ * The wedge: in every other semantic layer the mapping is unaudited config;
10
+ * here every mapping shows its work.
11
+ *
12
+ * Integration pattern:
13
+ * 1. Call surveySchemaMapping(context, extractor, options) to produce a
14
+ * SurveyInput from two or more system schemas.
15
+ * 2. Review the CandidateSets (kind: "needs-review" or "conflict").
16
+ * 3. Call mappingReviewToSurface(reviewedMappings) to project accepted
17
+ * mappings into a TrustBundle where each accepted mapping is BOTH:
18
+ * (a) a Claim (subjectType "system-field", fieldOrBehavior "maps-to")
19
+ * (b) an IdentityLink with relation/conversion and mappingClaimId
20
+ * so that resolveInquiry can resolve across systems with weakest-link
21
+ * capping.
22
+ */
23
+ import type { TrustBundle } from "@kontourai/surface";
24
+ import type { Candidate, CandidateSet, ReviewOutcome, SurveyInput } from "./types.js";
25
+ /**
26
+ * A stable reference to one field within one system's schema.
27
+ */
28
+ export interface SystemFieldRef {
29
+ /** The system identifier (e.g. "crm", "erp", "salesforce"). */
30
+ system: string;
31
+ /** The entity/table/resource name within that system. */
32
+ entity: string;
33
+ /** The field/column/attribute name within that entity. */
34
+ field: string;
35
+ /**
36
+ * Optional structural locator within a schema document (e.g.
37
+ * "json:$.definitions.Contact.properties.phoneNumber").
38
+ */
39
+ locator?: string;
40
+ }
41
+ /**
42
+ * A proposed link between two SystemFieldRefs, with evidence from schema
43
+ * documents or profiling output.
44
+ *
45
+ * This is the "show your work" record: every mapping carries excerpts from
46
+ * the relevant schema docs, a confidence score, a rationale, and the name
47
+ * of the extractor that produced the proposal. Nothing is accepted until
48
+ * it flows through review.
49
+ */
50
+ export interface MappingProposalRecord {
51
+ id: string;
52
+ /** The source field being mapped. */
53
+ sourceField: SystemFieldRef;
54
+ /** The target field being mapped. */
55
+ targetField: SystemFieldRef;
56
+ /**
57
+ * Semantic relation between the two fields.
58
+ * "equivalent" — same real-world fact, same unit.
59
+ * "subsumes" — the source field contains / is a superset of the target.
60
+ * "converts" — related by a numeric unit/scale conversion; supply conversion.
61
+ */
62
+ relation: "equivalent" | "subsumes" | "converts";
63
+ /**
64
+ * Numeric conversion parameters. Only meaningful when relation = "converts".
65
+ * target_value = source_value * factor + offset
66
+ */
67
+ conversion?: {
68
+ factor?: number;
69
+ offset?: number;
70
+ note?: string;
71
+ };
72
+ /**
73
+ * Schema document excerpts that support the proposal.
74
+ * One entry per system that the extractor consulted.
75
+ */
76
+ evidence: Array<{
77
+ system: string;
78
+ excerpt: string;
79
+ }>;
80
+ /** Extractor confidence in the mapping (0–1). */
81
+ confidence: number;
82
+ /** Human-readable rationale for the proposal. */
83
+ rationale: string;
84
+ /** Name of the SchemaMappingExtractor that produced this proposal. */
85
+ proposedBy: string;
86
+ /** ISO 8601 timestamp. */
87
+ proposedAt: string;
88
+ }
89
+ /**
90
+ * Pluggable interface for proposing field mappings across system schemas.
91
+ *
92
+ * Implementations may be deterministic (like referenceSchemaExtractor below),
93
+ * embedding-based, or LLM-backed — but they are always proposers: their output
94
+ * carries full provenance and goes through review before it counts.
95
+ *
96
+ * Supports both synchronous and async implementations.
97
+ */
98
+ export interface SchemaMappingExtractor {
99
+ name: string;
100
+ extract(context: {
101
+ systems: Array<{
102
+ system: string;
103
+ schemaText: string;
104
+ }>;
105
+ }): MappingProposalRecord[] | Promise<MappingProposalRecord[]>;
106
+ }
107
+ export interface SchemaMappingOptions {
108
+ /**
109
+ * If set, proposals at or above this confidence threshold are auto-accepted
110
+ * as "assumed" (mirrors applyAutoAcceptPolicy in inquiry-mapping).
111
+ * Conflicting proposals are never auto-accepted.
112
+ */
113
+ autoAcceptMinConfidence?: number;
114
+ /** ISO 8601 timestamp; defaults to new Date().toISOString(). */
115
+ generatedAt?: string;
116
+ /** Identifies the Survey producer run. */
117
+ source?: string;
118
+ }
119
+ /**
120
+ * Run the extractor against the provided system schemas and project the
121
+ * resulting proposals into the standard Survey chain:
122
+ *
123
+ * RawSource (kind "system-schema") per system
124
+ * → Extraction per proposal
125
+ * → Candidate per proposal
126
+ * → CandidateSet per field pair
127
+ * status "conflict" when proposals disagree about the same field pair
128
+ * status "needs-review" otherwise
129
+ *
130
+ * If options.autoAcceptMinConfidence is set, non-conflicting proposals above
131
+ * the threshold gain a ReviewOutcome with status "assumed".
132
+ *
133
+ * The returned SurveyInput is ready for buildSurveyTrustBundle (for provenance
134
+ * storage) or for mappingReviewToSurface after human review.
135
+ */
136
+ export declare function surveySchemaMapping(context: {
137
+ systems: Array<{
138
+ system: string;
139
+ schemaText: string;
140
+ }>;
141
+ }, extractor: SchemaMappingExtractor, options?: SchemaMappingOptions): Promise<{
142
+ surveyInput: SurveyInput;
143
+ proposals: MappingProposalRecord[];
144
+ candidateSets: CandidateSet[];
145
+ }>;
146
+ /**
147
+ * A reviewed (accepted or rejected) mapping record: the CandidateSet, the
148
+ * selected candidate, and the review outcome.
149
+ */
150
+ export interface ReviewedMapping {
151
+ pairKey: string;
152
+ candidateSet: CandidateSet;
153
+ /** The selected candidate (winner of review). */
154
+ selectedCandidate: Candidate;
155
+ reviewOutcome: ReviewOutcome;
156
+ /** Original proposal record. */
157
+ proposal: MappingProposalRecord;
158
+ }
159
+ /**
160
+ * Project reviewed (accepted) mappings into a TrustBundle.
161
+ *
162
+ * For each accepted mapping, the bundle contains BOTH:
163
+ * (a) a Claim: subjectType "system-field", fieldOrBehavior "maps-to"
164
+ * whose status reflects the review outcome; disputing this claim caps
165
+ * the identity-link answer via weakest-link (see mappingClaimId).
166
+ * (b) an IdentityLink: links the source and target system-field subjects
167
+ * with relation/conversion, and sets mappingClaimId to the claim id.
168
+ *
169
+ * This means resolveInquiry can traverse the link when asked about system B's
170
+ * field while only system A's claim exists in the bundle, with the mapping
171
+ * claim's status as the weakest-link ceiling.
172
+ *
173
+ * Rejected mappings are omitted from the bundle (but can be retained for
174
+ * audit by calling buildSurveyTrustBundle on the original SurveyInput).
175
+ */
176
+ export declare function mappingReviewToSurface(reviewedMappings: ReviewedMapping[], options?: {
177
+ source?: string;
178
+ generatedAt?: string;
179
+ }): TrustBundle;
180
+ /**
181
+ * Reference SchemaMappingExtractor for tests.
182
+ *
183
+ * REFERENCE IMPLEMENTATION ONLY — not suitable for production matching.
184
+ *
185
+ * Matching strategy: two fields are proposed as "equivalent" when they share
186
+ * the same field name (case-insensitive) and the same inferred type token
187
+ * ("string", "number", "boolean", "date", or absent) across two system schemas.
188
+ *
189
+ * Schema text format expected by this reference extractor:
190
+ * Each non-empty line is: "<entity>.<field>[:<type>]"
191
+ * e.g. "Contact.email:string" or "Account.revenue:number"
192
+ *
193
+ * This format is intentionally simple and transparent so tests can be
194
+ * deterministic without depending on any schema parsing library.
195
+ */
196
+ export declare const referenceSchemaExtractor: SchemaMappingExtractor;
@@ -0,0 +1,486 @@
1
+ /**
2
+ * Schema-mapping producer profile — EVIDENCED-ONTOLOGY layer.
3
+ *
4
+ * Every other semantic-layer mapping is unaudited config. Here every mapping
5
+ * shows its work: proposals carry schema-doc evidence, confidence, and
6
+ * rationale; they flow through Survey's candidate → review machinery before
7
+ * any mapping is accepted. Nothing here silently decides (ADR 0003 §4).
8
+ *
9
+ * The wedge: in every other semantic layer the mapping is unaudited config;
10
+ * here every mapping shows its work.
11
+ *
12
+ * Integration pattern:
13
+ * 1. Call surveySchemaMapping(context, extractor, options) to produce a
14
+ * SurveyInput from two or more system schemas.
15
+ * 2. Review the CandidateSets (kind: "needs-review" or "conflict").
16
+ * 3. Call mappingReviewToSurface(reviewedMappings) to project accepted
17
+ * mappings into a TrustBundle where each accepted mapping is BOTH:
18
+ * (a) a Claim (subjectType "system-field", fieldOrBehavior "maps-to")
19
+ * (b) an IdentityLink with relation/conversion and mappingClaimId
20
+ * so that resolveInquiry can resolve across systems with weakest-link
21
+ * capping.
22
+ */
23
+ import { buildSurveyTrustBundle } from "./to-surface.js";
24
+ // ---------------------------------------------------------------------------
25
+ // Canonical pair key
26
+ // ---------------------------------------------------------------------------
27
+ /** Stable key for a field pair (order-independent alphabetic sort). */
28
+ function fieldPairKey(a, b) {
29
+ const aKey = `${a.system}::${a.entity}::${a.field}`;
30
+ const bKey = `${b.system}::${b.entity}::${b.field}`;
31
+ return aKey <= bKey ? `${aKey}|${bKey}` : `${bKey}|${aKey}`;
32
+ }
33
+ /** Canonical mapping claim subject id. */
34
+ function mappingSubjectId(a, b) {
35
+ return fieldPairKey(a, b);
36
+ }
37
+ // ---------------------------------------------------------------------------
38
+ // surveySchemaMapping
39
+ // ---------------------------------------------------------------------------
40
+ /**
41
+ * Run the extractor against the provided system schemas and project the
42
+ * resulting proposals into the standard Survey chain:
43
+ *
44
+ * RawSource (kind "system-schema") per system
45
+ * → Extraction per proposal
46
+ * → Candidate per proposal
47
+ * → CandidateSet per field pair
48
+ * status "conflict" when proposals disagree about the same field pair
49
+ * status "needs-review" otherwise
50
+ *
51
+ * If options.autoAcceptMinConfidence is set, non-conflicting proposals above
52
+ * the threshold gain a ReviewOutcome with status "assumed".
53
+ *
54
+ * The returned SurveyInput is ready for buildSurveyTrustBundle (for provenance
55
+ * storage) or for mappingReviewToSurface after human review.
56
+ */
57
+ export async function surveySchemaMapping(context, extractor, options = {}) {
58
+ const generatedAt = options.generatedAt ?? new Date().toISOString();
59
+ const source = options.source ?? `schema-mapping:${extractor.name}`;
60
+ const proposals = await Promise.resolve(extractor.extract(context));
61
+ // One RawSource per system schema
62
+ const rawSources = context.systems.map((s) => ({
63
+ id: `schema-mapping.source.${s.system}`,
64
+ kind: "system-schema",
65
+ sourceRef: `system-schema://${s.system}`,
66
+ observedAt: generatedAt,
67
+ locatorScheme: "structured-field",
68
+ inlineText: s.schemaText,
69
+ metadata: { system: s.system },
70
+ }));
71
+ const sourceById = new Map(rawSources.map((r) => [r.metadata?.system, r]));
72
+ // Group proposals by canonical field-pair key to detect conflicts
73
+ const proposalsByPair = new Map();
74
+ for (const proposal of proposals) {
75
+ const key = fieldPairKey(proposal.sourceField, proposal.targetField);
76
+ proposalsByPair.set(key, [...(proposalsByPair.get(key) ?? []), proposal]);
77
+ }
78
+ const extractions = [];
79
+ const candidateSets = [];
80
+ const reviewOutcomes = [];
81
+ const claims = [];
82
+ for (const [pairKey, pairProposals] of proposalsByPair) {
83
+ const first = pairProposals[0];
84
+ const subjectId = mappingSubjectId(first.sourceField, first.targetField);
85
+ const candidateSetId = `schema-mapping.candidate-set.${pairKey}`;
86
+ const claimId = `schema-mapping.claim.${pairKey}`;
87
+ // Detect conflict: proposals disagree on relation
88
+ const relations = new Set(pairProposals.map((p) => p.relation));
89
+ const status = relations.size > 1 ? "conflict" : "needs-review";
90
+ const candidates = pairProposals.map((proposal) => {
91
+ // Use the source system's RawSource for this extraction
92
+ const rawSource = sourceById.get(proposal.sourceField.system) ?? rawSources[0];
93
+ const extractionId = `schema-mapping.extraction.${proposal.id}`;
94
+ const extraction = {
95
+ id: extractionId,
96
+ sourceId: rawSource.id,
97
+ target: `${proposal.sourceField.entity}.${proposal.sourceField.field}:maps-to:${proposal.targetField.entity}.${proposal.targetField.field}`,
98
+ value: {
99
+ relation: proposal.relation,
100
+ targetField: proposal.targetField,
101
+ conversion: proposal.conversion,
102
+ },
103
+ confidence: proposal.confidence,
104
+ locator: proposal.sourceField.locator ?? `structured-field:${proposal.sourceField.entity}.${proposal.sourceField.field}`,
105
+ excerpt: proposal.evidence.map((e) => `[${e.system}] ${e.excerpt}`).join(" | "),
106
+ extractor: proposal.proposedBy,
107
+ extractedAt: proposal.proposedAt,
108
+ metadata: {
109
+ schemaMappingProposal: {
110
+ proposalId: proposal.id,
111
+ sourceField: proposal.sourceField,
112
+ targetField: proposal.targetField,
113
+ relation: proposal.relation,
114
+ conversion: proposal.conversion,
115
+ evidence: proposal.evidence,
116
+ confidence: proposal.confidence,
117
+ rationale: proposal.rationale,
118
+ },
119
+ },
120
+ };
121
+ extractions.push(extraction);
122
+ return {
123
+ id: `schema-mapping.candidate.${proposal.id}`,
124
+ extractionId,
125
+ value: {
126
+ relation: proposal.relation,
127
+ targetField: proposal.targetField,
128
+ conversion: proposal.conversion,
129
+ },
130
+ confidence: proposal.confidence,
131
+ metadata: {
132
+ schemaMappingProposal: {
133
+ proposalId: proposal.id,
134
+ sourceField: proposal.sourceField,
135
+ targetField: proposal.targetField,
136
+ relation: proposal.relation,
137
+ conversion: proposal.conversion,
138
+ evidence: proposal.evidence,
139
+ confidence: proposal.confidence,
140
+ rationale: proposal.rationale,
141
+ proposedBy: proposal.proposedBy,
142
+ proposedAt: proposal.proposedAt,
143
+ },
144
+ },
145
+ };
146
+ });
147
+ const candidateSet = {
148
+ id: candidateSetId,
149
+ target: `schema-mapping:${pairKey}`,
150
+ candidates,
151
+ selectedCandidateId: status !== "conflict" ? candidates[0]?.id : undefined,
152
+ status,
153
+ rationale: status === "conflict"
154
+ ? `Proposals disagree on relation for pair ${pairKey}: ${[...relations].join(", ")}`
155
+ : `${candidates.length} proposal(s) agree on relation "${first.relation}" for pair ${pairKey}.`,
156
+ metadata: {
157
+ schemaMapping: {
158
+ pairKey,
159
+ sourceField: first.sourceField,
160
+ targetField: first.targetField,
161
+ },
162
+ },
163
+ };
164
+ candidateSets.push(candidateSet);
165
+ // Auto-accept policy: non-conflicting proposals above threshold → assumed
166
+ let autoReviewStatus;
167
+ if (status !== "conflict" && options.autoAcceptMinConfidence !== undefined) {
168
+ const topConfidence = Math.max(...pairProposals.map((p) => p.confidence));
169
+ if (topConfidence >= options.autoAcceptMinConfidence) {
170
+ autoReviewStatus = "assumed";
171
+ }
172
+ }
173
+ const selectedCandidate = candidateSet.selectedCandidateId
174
+ ? candidates.find((c) => c.id === candidateSet.selectedCandidateId)
175
+ : candidates[0];
176
+ if (autoReviewStatus && selectedCandidate) {
177
+ const reviewId = `schema-mapping.review.${pairKey}`;
178
+ reviewOutcomes.push({
179
+ id: reviewId,
180
+ candidateSetId,
181
+ candidateId: selectedCandidate.id,
182
+ status: autoReviewStatus,
183
+ actor: "auto-accept-policy",
184
+ reviewedAt: generatedAt,
185
+ rationale: `Auto-accepted: confidence ${selectedCandidate.confidence} >= threshold ${options.autoAcceptMinConfidence}`,
186
+ withinComfortZone: true,
187
+ });
188
+ }
189
+ // Project to ClaimTarget (subjectType "system-field", fieldOrBehavior "maps-to")
190
+ if (selectedCandidate) {
191
+ const review = reviewOutcomes.find((r) => r.candidateSetId === candidateSetId);
192
+ const claimStatus = review
193
+ ? review.status
194
+ : (status === "conflict" ? "disputed" : undefined);
195
+ const claimTarget = {
196
+ id: claimId,
197
+ candidateSetId,
198
+ candidateId: selectedCandidate.id,
199
+ subjectType: "system-field",
200
+ subjectId: subjectId,
201
+ surface: "schema-mapping.profile",
202
+ claimType: "schema-mapping.field-link",
203
+ fieldOrBehavior: "maps-to",
204
+ value: {
205
+ relation: first.relation,
206
+ sourceField: first.sourceField,
207
+ targetField: first.targetField,
208
+ conversion: first.conversion,
209
+ },
210
+ ...(claimStatus ? { status: claimStatus } : {}),
211
+ impactLevel: "medium",
212
+ collectedBy: extractor.name,
213
+ metadata: {
214
+ survey: {
215
+ schemaMapping: {
216
+ pairKey,
217
+ sourceField: first.sourceField,
218
+ targetField: first.targetField,
219
+ relation: first.relation,
220
+ conversion: first.conversion,
221
+ },
222
+ },
223
+ },
224
+ };
225
+ claims.push(claimTarget);
226
+ }
227
+ }
228
+ const surveyInput = {
229
+ source,
230
+ generatedAt,
231
+ rawSources,
232
+ extractions,
233
+ candidateSets,
234
+ reviewOutcomes,
235
+ claims,
236
+ };
237
+ return { surveyInput, proposals, candidateSets };
238
+ }
239
+ // ---------------------------------------------------------------------------
240
+ // mappingReviewToSurface
241
+ // ---------------------------------------------------------------------------
242
+ /**
243
+ * Project reviewed (accepted) mappings into a TrustBundle.
244
+ *
245
+ * For each accepted mapping, the bundle contains BOTH:
246
+ * (a) a Claim: subjectType "system-field", fieldOrBehavior "maps-to"
247
+ * whose status reflects the review outcome; disputing this claim caps
248
+ * the identity-link answer via weakest-link (see mappingClaimId).
249
+ * (b) an IdentityLink: links the source and target system-field subjects
250
+ * with relation/conversion, and sets mappingClaimId to the claim id.
251
+ *
252
+ * This means resolveInquiry can traverse the link when asked about system B's
253
+ * field while only system A's claim exists in the bundle, with the mapping
254
+ * claim's status as the weakest-link ceiling.
255
+ *
256
+ * Rejected mappings are omitted from the bundle (but can be retained for
257
+ * audit by calling buildSurveyTrustBundle on the original SurveyInput).
258
+ */
259
+ export function mappingReviewToSurface(reviewedMappings, options = {}) {
260
+ const generatedAt = options.generatedAt ?? new Date().toISOString();
261
+ const source = options.source ?? "schema-mapping.reviewed";
262
+ // Filter to accepted mappings only
263
+ const accepted = reviewedMappings.filter((m) => m.reviewOutcome.status === "verified" || m.reviewOutcome.status === "assumed");
264
+ // Build a SurveyInput for the accepted mappings so buildSurveyTrustBundle
265
+ // handles the full projection (provenance, events, evidence).
266
+ const rawSources = [];
267
+ const seenSources = new Set();
268
+ for (const rm of accepted) {
269
+ const meta = rm.selectedCandidate.metadata?.schemaMappingProposal;
270
+ const sourceField = meta?.sourceField ?? rm.proposal.sourceField;
271
+ const sourceId = `schema-mapping.source.${sourceField.system}`;
272
+ if (!seenSources.has(sourceId)) {
273
+ seenSources.add(sourceId);
274
+ rawSources.push({
275
+ id: sourceId,
276
+ kind: "system-schema",
277
+ sourceRef: `system-schema://${sourceField.system}`,
278
+ observedAt: generatedAt,
279
+ locatorScheme: "structured-field",
280
+ inlineText: (meta?.evidence ?? rm.proposal.evidence)
281
+ .filter((e) => e.system === sourceField.system)
282
+ .map((e) => e.excerpt)
283
+ .join("\n"),
284
+ metadata: { system: sourceField.system },
285
+ });
286
+ }
287
+ }
288
+ const extractions = [];
289
+ const candidateSets = [];
290
+ const reviewOutcomes = [];
291
+ const claims = [];
292
+ for (const rm of accepted) {
293
+ const meta = rm.selectedCandidate.metadata?.schemaMappingProposal;
294
+ const sourceField = meta?.sourceField ?? rm.proposal.sourceField;
295
+ const targetField = meta?.targetField ?? rm.proposal.targetField;
296
+ const relation = (meta?.relation ?? rm.proposal.relation);
297
+ const conversion = meta?.conversion ?? rm.proposal.conversion;
298
+ const evidence = meta?.evidence ?? rm.proposal.evidence;
299
+ const confidence = meta?.confidence ?? rm.proposal.confidence;
300
+ const proposedAt = meta?.proposedAt ?? rm.proposal.proposedAt;
301
+ const proposedBy = meta?.proposedBy ?? rm.proposal.proposedBy;
302
+ const rawSourceId = `schema-mapping.source.${sourceField.system}`;
303
+ const extractionId = `${rm.selectedCandidate.extractionId}`;
304
+ const candidateSetId = rm.candidateSet.id;
305
+ const claimId = `schema-mapping.claim.${rm.pairKey}`;
306
+ const subjectId = mappingSubjectId(sourceField, targetField);
307
+ const extraction = {
308
+ id: extractionId,
309
+ sourceId: rawSourceId,
310
+ target: `${sourceField.entity}.${sourceField.field}:maps-to:${targetField.entity}.${targetField.field}`,
311
+ value: { relation, targetField, conversion },
312
+ confidence,
313
+ locator: sourceField.locator ?? `structured-field:${sourceField.entity}.${sourceField.field}`,
314
+ excerpt: evidence.map((e) => `[${e.system}] ${e.excerpt}`).join(" | "),
315
+ extractor: proposedBy,
316
+ extractedAt: proposedAt,
317
+ metadata: {
318
+ schemaMappingProposal: {
319
+ sourceField,
320
+ targetField,
321
+ relation,
322
+ conversion,
323
+ evidence,
324
+ confidence,
325
+ },
326
+ },
327
+ };
328
+ extractions.push(extraction);
329
+ // Rebuild the candidate set with exactly the accepted candidate
330
+ const candidateSet = {
331
+ ...rm.candidateSet,
332
+ id: candidateSetId,
333
+ candidates: [rm.selectedCandidate],
334
+ selectedCandidateId: rm.selectedCandidate.id,
335
+ status: "resolved",
336
+ };
337
+ candidateSets.push(candidateSet);
338
+ reviewOutcomes.push(rm.reviewOutcome);
339
+ const claimTarget = {
340
+ id: claimId,
341
+ candidateSetId,
342
+ candidateId: rm.selectedCandidate.id,
343
+ subjectType: "system-field",
344
+ subjectId,
345
+ surface: "schema-mapping.profile",
346
+ claimType: "schema-mapping.field-link",
347
+ fieldOrBehavior: "maps-to",
348
+ value: { relation, sourceField, targetField, conversion },
349
+ status: rm.reviewOutcome.status,
350
+ impactLevel: "medium",
351
+ collectedBy: proposedBy,
352
+ actor: rm.reviewOutcome.actor,
353
+ metadata: {
354
+ survey: {
355
+ schemaMapping: {
356
+ pairKey: rm.pairKey,
357
+ sourceField,
358
+ targetField,
359
+ relation,
360
+ conversion,
361
+ },
362
+ },
363
+ },
364
+ };
365
+ claims.push(claimTarget);
366
+ }
367
+ const surveyInput = {
368
+ source,
369
+ generatedAt,
370
+ rawSources,
371
+ extractions,
372
+ candidateSets,
373
+ reviewOutcomes,
374
+ claims,
375
+ };
376
+ const bundle = buildSurveyTrustBundle(surveyInput);
377
+ // Attach IdentityLinks: one per accepted mapping.
378
+ // Each link ties the source-system-field subject to the target-system-field
379
+ // subject and back-references the mapping claim via mappingClaimId.
380
+ const identityLinks = [];
381
+ for (const rm of accepted) {
382
+ const meta = rm.selectedCandidate.metadata?.schemaMappingProposal;
383
+ const sourceField = meta?.sourceField ?? rm.proposal.sourceField;
384
+ const targetField = meta?.targetField ?? rm.proposal.targetField;
385
+ const relation = (meta?.relation ?? rm.proposal.relation);
386
+ const conversion = (meta?.conversion ?? rm.proposal.conversion);
387
+ const claimId = `schema-mapping.claim.${rm.pairKey}`;
388
+ const link = {
389
+ id: `schema-mapping.link.${rm.pairKey}`,
390
+ subjects: [
391
+ { subjectType: "system-field", subjectId: `${sourceField.system}::${sourceField.entity}::${sourceField.field}` },
392
+ { subjectType: "system-field", subjectId: `${targetField.system}::${targetField.entity}::${targetField.field}` },
393
+ ],
394
+ reason: `Schema mapping: ${sourceField.system}.${sourceField.entity}.${sourceField.field} ${relation} ${targetField.system}.${targetField.entity}.${targetField.field}`,
395
+ attestedBy: rm.reviewOutcome.actor,
396
+ relation,
397
+ ...(conversion ? { conversion } : {}),
398
+ mappingClaimId: claimId,
399
+ };
400
+ identityLinks.push(link);
401
+ }
402
+ return {
403
+ ...bundle,
404
+ identityLinks: identityLinks.length > 0 ? identityLinks : undefined,
405
+ };
406
+ }
407
+ // ---------------------------------------------------------------------------
408
+ // Reference extractor (deterministic, for tests — not for production use)
409
+ // ---------------------------------------------------------------------------
410
+ /**
411
+ * Reference SchemaMappingExtractor for tests.
412
+ *
413
+ * REFERENCE IMPLEMENTATION ONLY — not suitable for production matching.
414
+ *
415
+ * Matching strategy: two fields are proposed as "equivalent" when they share
416
+ * the same field name (case-insensitive) and the same inferred type token
417
+ * ("string", "number", "boolean", "date", or absent) across two system schemas.
418
+ *
419
+ * Schema text format expected by this reference extractor:
420
+ * Each non-empty line is: "<entity>.<field>[:<type>]"
421
+ * e.g. "Contact.email:string" or "Account.revenue:number"
422
+ *
423
+ * This format is intentionally simple and transparent so tests can be
424
+ * deterministic without depending on any schema parsing library.
425
+ */
426
+ export const referenceSchemaExtractor = {
427
+ name: "reference-schema-extractor",
428
+ extract(context) {
429
+ const now = new Date().toISOString();
430
+ const parsed = context.systems.map(({ system, schemaText }) => {
431
+ const fields = [];
432
+ for (const line of schemaText.split("\n")) {
433
+ const trimmed = line.trim();
434
+ if (!trimmed || trimmed.startsWith("#"))
435
+ continue;
436
+ // Format: entity.field[:type]
437
+ const colonIdx = trimmed.indexOf(":");
438
+ const namepart = colonIdx >= 0 ? trimmed.slice(0, colonIdx) : trimmed;
439
+ const type = colonIdx >= 0 ? trimmed.slice(colonIdx + 1).trim().toLowerCase() : "";
440
+ const dotIdx = namepart.indexOf(".");
441
+ if (dotIdx < 0)
442
+ continue;
443
+ const entity = namepart.slice(0, dotIdx).trim();
444
+ const field = namepart.slice(dotIdx + 1).trim();
445
+ if (!entity || !field)
446
+ continue;
447
+ fields.push({ entity, field, type, locator: `structured-field:${entity}.${field}` });
448
+ }
449
+ return { system, fields };
450
+ });
451
+ const proposals = [];
452
+ // Compare all pairs of systems
453
+ for (let i = 0; i < parsed.length; i++) {
454
+ for (let j = i + 1; j < parsed.length; j++) {
455
+ const sysA = parsed[i];
456
+ const sysB = parsed[j];
457
+ for (const fa of sysA.fields) {
458
+ for (const fb of sysB.fields) {
459
+ // Exact-name match (case-insensitive)
460
+ if (fa.field.toLowerCase() !== fb.field.toLowerCase())
461
+ continue;
462
+ // Type match (if both specified, they must be the same)
463
+ if (fa.type && fb.type && fa.type !== fb.type)
464
+ continue;
465
+ const confidence = fa.type && fb.type && fa.type === fb.type ? 0.9 : 0.75;
466
+ proposals.push({
467
+ id: `ref-schema-prop.${sysA.system}.${fa.entity}.${fa.field}__${sysB.system}.${fb.entity}.${fb.field}.${Date.now()}`,
468
+ sourceField: { system: sysA.system, entity: fa.entity, field: fa.field, locator: fa.locator },
469
+ targetField: { system: sysB.system, entity: fb.entity, field: fb.field, locator: fb.locator },
470
+ relation: "equivalent",
471
+ evidence: [
472
+ { system: sysA.system, excerpt: `${fa.entity}.${fa.field}${fa.type ? `:${fa.type}` : ""}` },
473
+ { system: sysB.system, excerpt: `${fb.entity}.${fb.field}${fb.type ? `:${fb.type}` : ""}` },
474
+ ],
475
+ confidence,
476
+ rationale: `Reference extractor: exact field-name match "${fa.field}"${fa.type && fb.type ? ` with matching type "${fa.type}"` : ""} across ${sysA.system} and ${sysB.system}.`,
477
+ proposedBy: "reference-schema-extractor",
478
+ proposedAt: now,
479
+ });
480
+ }
481
+ }
482
+ }
483
+ }
484
+ return proposals;
485
+ },
486
+ };