@kontourai/survey 0.4.24 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -10
- package/dist/example-data/corrected-document-candidates.d.ts +2 -0
- package/dist/{fixtures → example-data}/corrected-document-candidates.js +3 -3
- package/dist/{fixtures → example-data}/downstream-public-directory-proposal.d.ts +1 -1
- package/dist/{fixtures → example-data}/downstream-public-directory-proposal.js +1 -1
- package/dist/{fixtures → example-data}/public-directory-review-resource.d.ts +2 -2
- package/dist/{fixtures → example-data}/public-directory-review-resource.js +2 -2
- package/dist/example-data/public-field-review.d.ts +2 -0
- package/dist/{fixtures → example-data}/public-field-review.js +2 -2
- package/dist/{fixtures → example-data}/regulated-document-review-resource.d.ts +1 -1
- package/dist/{fixtures → example-data}/regulated-document-review-resource.js +3 -3
- package/dist/examples/public-field-observation.js +4 -4
- package/dist/examples/review-workbench/downstream-public-directory-adapter.d.ts +1 -1
- package/dist/examples/review-workbench/facility-credential-consumer.d.ts +2 -2
- package/dist/examples/review-workbench/facility-credential-consumer.js +8 -8
- package/dist/examples/review-workbench/server-apply-consumer.d.ts +1 -1
- package/dist/examples/review-workbench/server-apply-consumer.js +3 -3
- package/dist/src/agent-utterance.d.ts +166 -0
- package/dist/src/agent-utterance.js +373 -0
- package/dist/src/anthropic.d.ts +104 -0
- package/dist/src/anthropic.js +383 -0
- package/dist/src/index.d.ts +8 -2
- package/dist/src/index.js +4 -1
- package/dist/src/inquiry-mapping.d.ts +256 -0
- package/dist/src/inquiry-mapping.js +385 -0
- package/dist/src/review-workbench/review-queue-session.js +3 -3
- package/dist/src/review-workbench/review-surface-preview.js +1 -1
- package/dist/src/review-workbench/review-workbench-data.d.ts +4 -4
- package/dist/src/review-workbench/review-workbench-data.js +14 -14
- package/dist/src/schema-mapping.d.ts +196 -0
- package/dist/src/schema-mapping.js +486 -0
- package/dist/src/to-surface.d.ts +3 -3
- package/dist/src/to-surface.js +1 -1
- package/dist/src/types.d.ts +2 -2
- package/package.json +20 -7
- package/dist/fixtures/corrected-document-candidates.d.ts +0 -2
- package/dist/fixtures/public-field-review.d.ts +0 -2
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Schema-mapping producer profile — EVIDENCED-ONTOLOGY layer.
|
|
3
|
+
*
|
|
4
|
+
* Every other semantic-layer mapping is unaudited config. Here every mapping
|
|
5
|
+
* shows its work: proposals carry schema-doc evidence, confidence, and
|
|
6
|
+
* rationale; they flow through Survey's candidate → review machinery before
|
|
7
|
+
* any mapping is accepted. Nothing here silently decides (ADR 0003 §4).
|
|
8
|
+
*
|
|
9
|
+
* The wedge: in every other semantic layer the mapping is unaudited config;
|
|
10
|
+
* here every mapping shows its work.
|
|
11
|
+
*
|
|
12
|
+
* Integration pattern:
|
|
13
|
+
* 1. Call surveySchemaMapping(context, extractor, options) to produce a
|
|
14
|
+
* SurveyInput from two or more system schemas.
|
|
15
|
+
* 2. Review the CandidateSets (kind: "needs-review" or "conflict").
|
|
16
|
+
* 3. Call mappingReviewToSurface(reviewedMappings) to project accepted
|
|
17
|
+
* mappings into a TrustBundle where each accepted mapping is BOTH:
|
|
18
|
+
* (a) a Claim (subjectType "system-field", fieldOrBehavior "maps-to")
|
|
19
|
+
* (b) an IdentityLink with relation/conversion and mappingClaimId
|
|
20
|
+
* so that resolveInquiry can resolve across systems with weakest-link
|
|
21
|
+
* capping.
|
|
22
|
+
*/
|
|
23
|
+
import type { TrustBundle } from "@kontourai/surface";
|
|
24
|
+
import type { Candidate, CandidateSet, ReviewOutcome, SurveyInput } from "./types.js";
|
|
25
|
+
/**
|
|
26
|
+
* A stable reference to one field within one system's schema.
|
|
27
|
+
*/
|
|
28
|
+
export interface SystemFieldRef {
|
|
29
|
+
/** The system identifier (e.g. "crm", "erp", "salesforce"). */
|
|
30
|
+
system: string;
|
|
31
|
+
/** The entity/table/resource name within that system. */
|
|
32
|
+
entity: string;
|
|
33
|
+
/** The field/column/attribute name within that entity. */
|
|
34
|
+
field: string;
|
|
35
|
+
/**
|
|
36
|
+
* Optional structural locator within a schema document (e.g.
|
|
37
|
+
* "json:$.definitions.Contact.properties.phoneNumber").
|
|
38
|
+
*/
|
|
39
|
+
locator?: string;
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* A proposed link between two SystemFieldRefs, with evidence from schema
|
|
43
|
+
* documents or profiling output.
|
|
44
|
+
*
|
|
45
|
+
* This is the "show your work" record: every mapping carries excerpts from
|
|
46
|
+
* the relevant schema docs, a confidence score, a rationale, and the name
|
|
47
|
+
* of the extractor that produced the proposal. Nothing is accepted until
|
|
48
|
+
* it flows through review.
|
|
49
|
+
*/
|
|
50
|
+
export interface MappingProposalRecord {
|
|
51
|
+
id: string;
|
|
52
|
+
/** The source field being mapped. */
|
|
53
|
+
sourceField: SystemFieldRef;
|
|
54
|
+
/** The target field being mapped. */
|
|
55
|
+
targetField: SystemFieldRef;
|
|
56
|
+
/**
|
|
57
|
+
* Semantic relation between the two fields.
|
|
58
|
+
* "equivalent" — same real-world fact, same unit.
|
|
59
|
+
* "subsumes" — the source field contains / is a superset of the target.
|
|
60
|
+
* "converts" — related by a numeric unit/scale conversion; supply conversion.
|
|
61
|
+
*/
|
|
62
|
+
relation: "equivalent" | "subsumes" | "converts";
|
|
63
|
+
/**
|
|
64
|
+
* Numeric conversion parameters. Only meaningful when relation = "converts".
|
|
65
|
+
* target_value = source_value * factor + offset
|
|
66
|
+
*/
|
|
67
|
+
conversion?: {
|
|
68
|
+
factor?: number;
|
|
69
|
+
offset?: number;
|
|
70
|
+
note?: string;
|
|
71
|
+
};
|
|
72
|
+
/**
|
|
73
|
+
* Schema document excerpts that support the proposal.
|
|
74
|
+
* One entry per system that the extractor consulted.
|
|
75
|
+
*/
|
|
76
|
+
evidence: Array<{
|
|
77
|
+
system: string;
|
|
78
|
+
excerpt: string;
|
|
79
|
+
}>;
|
|
80
|
+
/** Extractor confidence in the mapping (0–1). */
|
|
81
|
+
confidence: number;
|
|
82
|
+
/** Human-readable rationale for the proposal. */
|
|
83
|
+
rationale: string;
|
|
84
|
+
/** Name of the SchemaMappingExtractor that produced this proposal. */
|
|
85
|
+
proposedBy: string;
|
|
86
|
+
/** ISO 8601 timestamp. */
|
|
87
|
+
proposedAt: string;
|
|
88
|
+
}
|
|
89
|
+
/**
|
|
90
|
+
* Pluggable interface for proposing field mappings across system schemas.
|
|
91
|
+
*
|
|
92
|
+
* Implementations may be deterministic (like referenceSchemaExtractor below),
|
|
93
|
+
* embedding-based, or LLM-backed — but they are always proposers: their output
|
|
94
|
+
* carries full provenance and goes through review before it counts.
|
|
95
|
+
*
|
|
96
|
+
* Supports both synchronous and async implementations.
|
|
97
|
+
*/
|
|
98
|
+
export interface SchemaMappingExtractor {
|
|
99
|
+
name: string;
|
|
100
|
+
extract(context: {
|
|
101
|
+
systems: Array<{
|
|
102
|
+
system: string;
|
|
103
|
+
schemaText: string;
|
|
104
|
+
}>;
|
|
105
|
+
}): MappingProposalRecord[] | Promise<MappingProposalRecord[]>;
|
|
106
|
+
}
|
|
107
|
+
export interface SchemaMappingOptions {
|
|
108
|
+
/**
|
|
109
|
+
* If set, proposals at or above this confidence threshold are auto-accepted
|
|
110
|
+
* as "assumed" (mirrors applyAutoAcceptPolicy in inquiry-mapping).
|
|
111
|
+
* Conflicting proposals are never auto-accepted.
|
|
112
|
+
*/
|
|
113
|
+
autoAcceptMinConfidence?: number;
|
|
114
|
+
/** ISO 8601 timestamp; defaults to new Date().toISOString(). */
|
|
115
|
+
generatedAt?: string;
|
|
116
|
+
/** Identifies the Survey producer run. */
|
|
117
|
+
source?: string;
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* Run the extractor against the provided system schemas and project the
|
|
121
|
+
* resulting proposals into the standard Survey chain:
|
|
122
|
+
*
|
|
123
|
+
* RawSource (kind "system-schema") per system
|
|
124
|
+
* → Extraction per proposal
|
|
125
|
+
* → Candidate per proposal
|
|
126
|
+
* → CandidateSet per field pair
|
|
127
|
+
* status "conflict" when proposals disagree about the same field pair
|
|
128
|
+
* status "needs-review" otherwise
|
|
129
|
+
*
|
|
130
|
+
* If options.autoAcceptMinConfidence is set, non-conflicting proposals above
|
|
131
|
+
* the threshold gain a ReviewOutcome with status "assumed".
|
|
132
|
+
*
|
|
133
|
+
* The returned SurveyInput is ready for buildSurveyTrustBundle (for provenance
|
|
134
|
+
* storage) or for mappingReviewToSurface after human review.
|
|
135
|
+
*/
|
|
136
|
+
export declare function surveySchemaMapping(context: {
|
|
137
|
+
systems: Array<{
|
|
138
|
+
system: string;
|
|
139
|
+
schemaText: string;
|
|
140
|
+
}>;
|
|
141
|
+
}, extractor: SchemaMappingExtractor, options?: SchemaMappingOptions): Promise<{
|
|
142
|
+
surveyInput: SurveyInput;
|
|
143
|
+
proposals: MappingProposalRecord[];
|
|
144
|
+
candidateSets: CandidateSet[];
|
|
145
|
+
}>;
|
|
146
|
+
/**
|
|
147
|
+
* A reviewed (accepted or rejected) mapping record: the CandidateSet, the
|
|
148
|
+
* selected candidate, and the review outcome.
|
|
149
|
+
*/
|
|
150
|
+
export interface ReviewedMapping {
|
|
151
|
+
pairKey: string;
|
|
152
|
+
candidateSet: CandidateSet;
|
|
153
|
+
/** The selected candidate (winner of review). */
|
|
154
|
+
selectedCandidate: Candidate;
|
|
155
|
+
reviewOutcome: ReviewOutcome;
|
|
156
|
+
/** Original proposal record. */
|
|
157
|
+
proposal: MappingProposalRecord;
|
|
158
|
+
}
|
|
159
|
+
/**
|
|
160
|
+
* Project reviewed (accepted) mappings into a TrustBundle.
|
|
161
|
+
*
|
|
162
|
+
* For each accepted mapping, the bundle contains BOTH:
|
|
163
|
+
* (a) a Claim: subjectType "system-field", fieldOrBehavior "maps-to"
|
|
164
|
+
* whose status reflects the review outcome; disputing this claim caps
|
|
165
|
+
* the identity-link answer via weakest-link (see mappingClaimId).
|
|
166
|
+
* (b) an IdentityLink: links the source and target system-field subjects
|
|
167
|
+
* with relation/conversion, and sets mappingClaimId to the claim id.
|
|
168
|
+
*
|
|
169
|
+
* This means resolveInquiry can traverse the link when asked about system B's
|
|
170
|
+
* field while only system A's claim exists in the bundle, with the mapping
|
|
171
|
+
* claim's status as the weakest-link ceiling.
|
|
172
|
+
*
|
|
173
|
+
* Rejected mappings are omitted from the bundle (but can be retained for
|
|
174
|
+
* audit by calling buildSurveyTrustBundle on the original SurveyInput).
|
|
175
|
+
*/
|
|
176
|
+
export declare function mappingReviewToSurface(reviewedMappings: ReviewedMapping[], options?: {
|
|
177
|
+
source?: string;
|
|
178
|
+
generatedAt?: string;
|
|
179
|
+
}): TrustBundle;
|
|
180
|
+
/**
|
|
181
|
+
* Reference SchemaMappingExtractor for tests.
|
|
182
|
+
*
|
|
183
|
+
* REFERENCE IMPLEMENTATION ONLY — not suitable for production matching.
|
|
184
|
+
*
|
|
185
|
+
* Matching strategy: two fields are proposed as "equivalent" when they share
|
|
186
|
+
* the same field name (case-insensitive) and the same inferred type token
|
|
187
|
+
* ("string", "number", "boolean", "date", or absent) across two system schemas.
|
|
188
|
+
*
|
|
189
|
+
* Schema text format expected by this reference extractor:
|
|
190
|
+
* Each non-empty line is: "<entity>.<field>[:<type>]"
|
|
191
|
+
* e.g. "Contact.email:string" or "Account.revenue:number"
|
|
192
|
+
*
|
|
193
|
+
* This format is intentionally simple and transparent so tests can be
|
|
194
|
+
* deterministic without depending on any schema parsing library.
|
|
195
|
+
*/
|
|
196
|
+
export declare const referenceSchemaExtractor: SchemaMappingExtractor;
|
|
@@ -0,0 +1,486 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Schema-mapping producer profile — EVIDENCED-ONTOLOGY layer.
|
|
3
|
+
*
|
|
4
|
+
* Every other semantic-layer mapping is unaudited config. Here every mapping
|
|
5
|
+
* shows its work: proposals carry schema-doc evidence, confidence, and
|
|
6
|
+
* rationale; they flow through Survey's candidate → review machinery before
|
|
7
|
+
* any mapping is accepted. Nothing here silently decides (ADR 0003 §4).
|
|
8
|
+
*
|
|
9
|
+
* The wedge: in every other semantic layer the mapping is unaudited config;
|
|
10
|
+
* here every mapping shows its work.
|
|
11
|
+
*
|
|
12
|
+
* Integration pattern:
|
|
13
|
+
* 1. Call surveySchemaMapping(context, extractor, options) to produce a
|
|
14
|
+
* SurveyInput from two or more system schemas.
|
|
15
|
+
* 2. Review the CandidateSets (kind: "needs-review" or "conflict").
|
|
16
|
+
* 3. Call mappingReviewToSurface(reviewedMappings) to project accepted
|
|
17
|
+
* mappings into a TrustBundle where each accepted mapping is BOTH:
|
|
18
|
+
* (a) a Claim (subjectType "system-field", fieldOrBehavior "maps-to")
|
|
19
|
+
* (b) an IdentityLink with relation/conversion and mappingClaimId
|
|
20
|
+
* so that resolveInquiry can resolve across systems with weakest-link
|
|
21
|
+
* capping.
|
|
22
|
+
*/
|
|
23
|
+
import { buildSurveyTrustBundle } from "./to-surface.js";
|
|
24
|
+
// ---------------------------------------------------------------------------
|
|
25
|
+
// Canonical pair key
|
|
26
|
+
// ---------------------------------------------------------------------------
|
|
27
|
+
/** Stable key for a field pair (order-independent alphabetic sort). */
|
|
28
|
+
function fieldPairKey(a, b) {
|
|
29
|
+
const aKey = `${a.system}::${a.entity}::${a.field}`;
|
|
30
|
+
const bKey = `${b.system}::${b.entity}::${b.field}`;
|
|
31
|
+
return aKey <= bKey ? `${aKey}|${bKey}` : `${bKey}|${aKey}`;
|
|
32
|
+
}
|
|
33
|
+
/** Canonical mapping claim subject id. */
|
|
34
|
+
function mappingSubjectId(a, b) {
|
|
35
|
+
return fieldPairKey(a, b);
|
|
36
|
+
}
|
|
37
|
+
// ---------------------------------------------------------------------------
|
|
38
|
+
// surveySchemaMapping
|
|
39
|
+
// ---------------------------------------------------------------------------
|
|
40
|
+
/**
|
|
41
|
+
* Run the extractor against the provided system schemas and project the
|
|
42
|
+
* resulting proposals into the standard Survey chain:
|
|
43
|
+
*
|
|
44
|
+
* RawSource (kind "system-schema") per system
|
|
45
|
+
* → Extraction per proposal
|
|
46
|
+
* → Candidate per proposal
|
|
47
|
+
* → CandidateSet per field pair
|
|
48
|
+
* status "conflict" when proposals disagree about the same field pair
|
|
49
|
+
* status "needs-review" otherwise
|
|
50
|
+
*
|
|
51
|
+
* If options.autoAcceptMinConfidence is set, non-conflicting proposals above
|
|
52
|
+
* the threshold gain a ReviewOutcome with status "assumed".
|
|
53
|
+
*
|
|
54
|
+
* The returned SurveyInput is ready for buildSurveyTrustBundle (for provenance
|
|
55
|
+
* storage) or for mappingReviewToSurface after human review.
|
|
56
|
+
*/
|
|
57
|
+
export async function surveySchemaMapping(context, extractor, options = {}) {
|
|
58
|
+
const generatedAt = options.generatedAt ?? new Date().toISOString();
|
|
59
|
+
const source = options.source ?? `schema-mapping:${extractor.name}`;
|
|
60
|
+
const proposals = await Promise.resolve(extractor.extract(context));
|
|
61
|
+
// One RawSource per system schema
|
|
62
|
+
const rawSources = context.systems.map((s) => ({
|
|
63
|
+
id: `schema-mapping.source.${s.system}`,
|
|
64
|
+
kind: "system-schema",
|
|
65
|
+
sourceRef: `system-schema://${s.system}`,
|
|
66
|
+
observedAt: generatedAt,
|
|
67
|
+
locatorScheme: "structured-field",
|
|
68
|
+
inlineText: s.schemaText,
|
|
69
|
+
metadata: { system: s.system },
|
|
70
|
+
}));
|
|
71
|
+
const sourceById = new Map(rawSources.map((r) => [r.metadata?.system, r]));
|
|
72
|
+
// Group proposals by canonical field-pair key to detect conflicts
|
|
73
|
+
const proposalsByPair = new Map();
|
|
74
|
+
for (const proposal of proposals) {
|
|
75
|
+
const key = fieldPairKey(proposal.sourceField, proposal.targetField);
|
|
76
|
+
proposalsByPair.set(key, [...(proposalsByPair.get(key) ?? []), proposal]);
|
|
77
|
+
}
|
|
78
|
+
const extractions = [];
|
|
79
|
+
const candidateSets = [];
|
|
80
|
+
const reviewOutcomes = [];
|
|
81
|
+
const claims = [];
|
|
82
|
+
for (const [pairKey, pairProposals] of proposalsByPair) {
|
|
83
|
+
const first = pairProposals[0];
|
|
84
|
+
const subjectId = mappingSubjectId(first.sourceField, first.targetField);
|
|
85
|
+
const candidateSetId = `schema-mapping.candidate-set.${pairKey}`;
|
|
86
|
+
const claimId = `schema-mapping.claim.${pairKey}`;
|
|
87
|
+
// Detect conflict: proposals disagree on relation
|
|
88
|
+
const relations = new Set(pairProposals.map((p) => p.relation));
|
|
89
|
+
const status = relations.size > 1 ? "conflict" : "needs-review";
|
|
90
|
+
const candidates = pairProposals.map((proposal) => {
|
|
91
|
+
// Use the source system's RawSource for this extraction
|
|
92
|
+
const rawSource = sourceById.get(proposal.sourceField.system) ?? rawSources[0];
|
|
93
|
+
const extractionId = `schema-mapping.extraction.${proposal.id}`;
|
|
94
|
+
const extraction = {
|
|
95
|
+
id: extractionId,
|
|
96
|
+
sourceId: rawSource.id,
|
|
97
|
+
target: `${proposal.sourceField.entity}.${proposal.sourceField.field}:maps-to:${proposal.targetField.entity}.${proposal.targetField.field}`,
|
|
98
|
+
value: {
|
|
99
|
+
relation: proposal.relation,
|
|
100
|
+
targetField: proposal.targetField,
|
|
101
|
+
conversion: proposal.conversion,
|
|
102
|
+
},
|
|
103
|
+
confidence: proposal.confidence,
|
|
104
|
+
locator: proposal.sourceField.locator ?? `structured-field:${proposal.sourceField.entity}.${proposal.sourceField.field}`,
|
|
105
|
+
excerpt: proposal.evidence.map((e) => `[${e.system}] ${e.excerpt}`).join(" | "),
|
|
106
|
+
extractor: proposal.proposedBy,
|
|
107
|
+
extractedAt: proposal.proposedAt,
|
|
108
|
+
metadata: {
|
|
109
|
+
schemaMappingProposal: {
|
|
110
|
+
proposalId: proposal.id,
|
|
111
|
+
sourceField: proposal.sourceField,
|
|
112
|
+
targetField: proposal.targetField,
|
|
113
|
+
relation: proposal.relation,
|
|
114
|
+
conversion: proposal.conversion,
|
|
115
|
+
evidence: proposal.evidence,
|
|
116
|
+
confidence: proposal.confidence,
|
|
117
|
+
rationale: proposal.rationale,
|
|
118
|
+
},
|
|
119
|
+
},
|
|
120
|
+
};
|
|
121
|
+
extractions.push(extraction);
|
|
122
|
+
return {
|
|
123
|
+
id: `schema-mapping.candidate.${proposal.id}`,
|
|
124
|
+
extractionId,
|
|
125
|
+
value: {
|
|
126
|
+
relation: proposal.relation,
|
|
127
|
+
targetField: proposal.targetField,
|
|
128
|
+
conversion: proposal.conversion,
|
|
129
|
+
},
|
|
130
|
+
confidence: proposal.confidence,
|
|
131
|
+
metadata: {
|
|
132
|
+
schemaMappingProposal: {
|
|
133
|
+
proposalId: proposal.id,
|
|
134
|
+
sourceField: proposal.sourceField,
|
|
135
|
+
targetField: proposal.targetField,
|
|
136
|
+
relation: proposal.relation,
|
|
137
|
+
conversion: proposal.conversion,
|
|
138
|
+
evidence: proposal.evidence,
|
|
139
|
+
confidence: proposal.confidence,
|
|
140
|
+
rationale: proposal.rationale,
|
|
141
|
+
proposedBy: proposal.proposedBy,
|
|
142
|
+
proposedAt: proposal.proposedAt,
|
|
143
|
+
},
|
|
144
|
+
},
|
|
145
|
+
};
|
|
146
|
+
});
|
|
147
|
+
const candidateSet = {
|
|
148
|
+
id: candidateSetId,
|
|
149
|
+
target: `schema-mapping:${pairKey}`,
|
|
150
|
+
candidates,
|
|
151
|
+
selectedCandidateId: status !== "conflict" ? candidates[0]?.id : undefined,
|
|
152
|
+
status,
|
|
153
|
+
rationale: status === "conflict"
|
|
154
|
+
? `Proposals disagree on relation for pair ${pairKey}: ${[...relations].join(", ")}`
|
|
155
|
+
: `${candidates.length} proposal(s) agree on relation "${first.relation}" for pair ${pairKey}.`,
|
|
156
|
+
metadata: {
|
|
157
|
+
schemaMapping: {
|
|
158
|
+
pairKey,
|
|
159
|
+
sourceField: first.sourceField,
|
|
160
|
+
targetField: first.targetField,
|
|
161
|
+
},
|
|
162
|
+
},
|
|
163
|
+
};
|
|
164
|
+
candidateSets.push(candidateSet);
|
|
165
|
+
// Auto-accept policy: non-conflicting proposals above threshold → assumed
|
|
166
|
+
let autoReviewStatus;
|
|
167
|
+
if (status !== "conflict" && options.autoAcceptMinConfidence !== undefined) {
|
|
168
|
+
const topConfidence = Math.max(...pairProposals.map((p) => p.confidence));
|
|
169
|
+
if (topConfidence >= options.autoAcceptMinConfidence) {
|
|
170
|
+
autoReviewStatus = "assumed";
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
const selectedCandidate = candidateSet.selectedCandidateId
|
|
174
|
+
? candidates.find((c) => c.id === candidateSet.selectedCandidateId)
|
|
175
|
+
: candidates[0];
|
|
176
|
+
if (autoReviewStatus && selectedCandidate) {
|
|
177
|
+
const reviewId = `schema-mapping.review.${pairKey}`;
|
|
178
|
+
reviewOutcomes.push({
|
|
179
|
+
id: reviewId,
|
|
180
|
+
candidateSetId,
|
|
181
|
+
candidateId: selectedCandidate.id,
|
|
182
|
+
status: autoReviewStatus,
|
|
183
|
+
actor: "auto-accept-policy",
|
|
184
|
+
reviewedAt: generatedAt,
|
|
185
|
+
rationale: `Auto-accepted: confidence ${selectedCandidate.confidence} >= threshold ${options.autoAcceptMinConfidence}`,
|
|
186
|
+
withinComfortZone: true,
|
|
187
|
+
});
|
|
188
|
+
}
|
|
189
|
+
// Project to ClaimTarget (subjectType "system-field", fieldOrBehavior "maps-to")
|
|
190
|
+
if (selectedCandidate) {
|
|
191
|
+
const review = reviewOutcomes.find((r) => r.candidateSetId === candidateSetId);
|
|
192
|
+
const claimStatus = review
|
|
193
|
+
? review.status
|
|
194
|
+
: (status === "conflict" ? "disputed" : undefined);
|
|
195
|
+
const claimTarget = {
|
|
196
|
+
id: claimId,
|
|
197
|
+
candidateSetId,
|
|
198
|
+
candidateId: selectedCandidate.id,
|
|
199
|
+
subjectType: "system-field",
|
|
200
|
+
subjectId: subjectId,
|
|
201
|
+
surface: "schema-mapping.profile",
|
|
202
|
+
claimType: "schema-mapping.field-link",
|
|
203
|
+
fieldOrBehavior: "maps-to",
|
|
204
|
+
value: {
|
|
205
|
+
relation: first.relation,
|
|
206
|
+
sourceField: first.sourceField,
|
|
207
|
+
targetField: first.targetField,
|
|
208
|
+
conversion: first.conversion,
|
|
209
|
+
},
|
|
210
|
+
...(claimStatus ? { status: claimStatus } : {}),
|
|
211
|
+
impactLevel: "medium",
|
|
212
|
+
collectedBy: extractor.name,
|
|
213
|
+
metadata: {
|
|
214
|
+
survey: {
|
|
215
|
+
schemaMapping: {
|
|
216
|
+
pairKey,
|
|
217
|
+
sourceField: first.sourceField,
|
|
218
|
+
targetField: first.targetField,
|
|
219
|
+
relation: first.relation,
|
|
220
|
+
conversion: first.conversion,
|
|
221
|
+
},
|
|
222
|
+
},
|
|
223
|
+
},
|
|
224
|
+
};
|
|
225
|
+
claims.push(claimTarget);
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
const surveyInput = {
|
|
229
|
+
source,
|
|
230
|
+
generatedAt,
|
|
231
|
+
rawSources,
|
|
232
|
+
extractions,
|
|
233
|
+
candidateSets,
|
|
234
|
+
reviewOutcomes,
|
|
235
|
+
claims,
|
|
236
|
+
};
|
|
237
|
+
return { surveyInput, proposals, candidateSets };
|
|
238
|
+
}
|
|
239
|
+
// ---------------------------------------------------------------------------
|
|
240
|
+
// mappingReviewToSurface
|
|
241
|
+
// ---------------------------------------------------------------------------
|
|
242
|
+
/**
|
|
243
|
+
* Project reviewed (accepted) mappings into a TrustBundle.
|
|
244
|
+
*
|
|
245
|
+
* For each accepted mapping, the bundle contains BOTH:
|
|
246
|
+
* (a) a Claim: subjectType "system-field", fieldOrBehavior "maps-to"
|
|
247
|
+
* whose status reflects the review outcome; disputing this claim caps
|
|
248
|
+
* the identity-link answer via weakest-link (see mappingClaimId).
|
|
249
|
+
* (b) an IdentityLink: links the source and target system-field subjects
|
|
250
|
+
* with relation/conversion, and sets mappingClaimId to the claim id.
|
|
251
|
+
*
|
|
252
|
+
* This means resolveInquiry can traverse the link when asked about system B's
|
|
253
|
+
* field while only system A's claim exists in the bundle, with the mapping
|
|
254
|
+
* claim's status as the weakest-link ceiling.
|
|
255
|
+
*
|
|
256
|
+
* Rejected mappings are omitted from the bundle (but can be retained for
|
|
257
|
+
* audit by calling buildSurveyTrustBundle on the original SurveyInput).
|
|
258
|
+
*/
|
|
259
|
+
export function mappingReviewToSurface(reviewedMappings, options = {}) {
|
|
260
|
+
const generatedAt = options.generatedAt ?? new Date().toISOString();
|
|
261
|
+
const source = options.source ?? "schema-mapping.reviewed";
|
|
262
|
+
// Filter to accepted mappings only
|
|
263
|
+
const accepted = reviewedMappings.filter((m) => m.reviewOutcome.status === "verified" || m.reviewOutcome.status === "assumed");
|
|
264
|
+
// Build a SurveyInput for the accepted mappings so buildSurveyTrustBundle
|
|
265
|
+
// handles the full projection (provenance, events, evidence).
|
|
266
|
+
const rawSources = [];
|
|
267
|
+
const seenSources = new Set();
|
|
268
|
+
for (const rm of accepted) {
|
|
269
|
+
const meta = rm.selectedCandidate.metadata?.schemaMappingProposal;
|
|
270
|
+
const sourceField = meta?.sourceField ?? rm.proposal.sourceField;
|
|
271
|
+
const sourceId = `schema-mapping.source.${sourceField.system}`;
|
|
272
|
+
if (!seenSources.has(sourceId)) {
|
|
273
|
+
seenSources.add(sourceId);
|
|
274
|
+
rawSources.push({
|
|
275
|
+
id: sourceId,
|
|
276
|
+
kind: "system-schema",
|
|
277
|
+
sourceRef: `system-schema://${sourceField.system}`,
|
|
278
|
+
observedAt: generatedAt,
|
|
279
|
+
locatorScheme: "structured-field",
|
|
280
|
+
inlineText: (meta?.evidence ?? rm.proposal.evidence)
|
|
281
|
+
.filter((e) => e.system === sourceField.system)
|
|
282
|
+
.map((e) => e.excerpt)
|
|
283
|
+
.join("\n"),
|
|
284
|
+
metadata: { system: sourceField.system },
|
|
285
|
+
});
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
const extractions = [];
|
|
289
|
+
const candidateSets = [];
|
|
290
|
+
const reviewOutcomes = [];
|
|
291
|
+
const claims = [];
|
|
292
|
+
for (const rm of accepted) {
|
|
293
|
+
const meta = rm.selectedCandidate.metadata?.schemaMappingProposal;
|
|
294
|
+
const sourceField = meta?.sourceField ?? rm.proposal.sourceField;
|
|
295
|
+
const targetField = meta?.targetField ?? rm.proposal.targetField;
|
|
296
|
+
const relation = (meta?.relation ?? rm.proposal.relation);
|
|
297
|
+
const conversion = meta?.conversion ?? rm.proposal.conversion;
|
|
298
|
+
const evidence = meta?.evidence ?? rm.proposal.evidence;
|
|
299
|
+
const confidence = meta?.confidence ?? rm.proposal.confidence;
|
|
300
|
+
const proposedAt = meta?.proposedAt ?? rm.proposal.proposedAt;
|
|
301
|
+
const proposedBy = meta?.proposedBy ?? rm.proposal.proposedBy;
|
|
302
|
+
const rawSourceId = `schema-mapping.source.${sourceField.system}`;
|
|
303
|
+
const extractionId = `${rm.selectedCandidate.extractionId}`;
|
|
304
|
+
const candidateSetId = rm.candidateSet.id;
|
|
305
|
+
const claimId = `schema-mapping.claim.${rm.pairKey}`;
|
|
306
|
+
const subjectId = mappingSubjectId(sourceField, targetField);
|
|
307
|
+
const extraction = {
|
|
308
|
+
id: extractionId,
|
|
309
|
+
sourceId: rawSourceId,
|
|
310
|
+
target: `${sourceField.entity}.${sourceField.field}:maps-to:${targetField.entity}.${targetField.field}`,
|
|
311
|
+
value: { relation, targetField, conversion },
|
|
312
|
+
confidence,
|
|
313
|
+
locator: sourceField.locator ?? `structured-field:${sourceField.entity}.${sourceField.field}`,
|
|
314
|
+
excerpt: evidence.map((e) => `[${e.system}] ${e.excerpt}`).join(" | "),
|
|
315
|
+
extractor: proposedBy,
|
|
316
|
+
extractedAt: proposedAt,
|
|
317
|
+
metadata: {
|
|
318
|
+
schemaMappingProposal: {
|
|
319
|
+
sourceField,
|
|
320
|
+
targetField,
|
|
321
|
+
relation,
|
|
322
|
+
conversion,
|
|
323
|
+
evidence,
|
|
324
|
+
confidence,
|
|
325
|
+
},
|
|
326
|
+
},
|
|
327
|
+
};
|
|
328
|
+
extractions.push(extraction);
|
|
329
|
+
// Rebuild the candidate set with exactly the accepted candidate
|
|
330
|
+
const candidateSet = {
|
|
331
|
+
...rm.candidateSet,
|
|
332
|
+
id: candidateSetId,
|
|
333
|
+
candidates: [rm.selectedCandidate],
|
|
334
|
+
selectedCandidateId: rm.selectedCandidate.id,
|
|
335
|
+
status: "resolved",
|
|
336
|
+
};
|
|
337
|
+
candidateSets.push(candidateSet);
|
|
338
|
+
reviewOutcomes.push(rm.reviewOutcome);
|
|
339
|
+
const claimTarget = {
|
|
340
|
+
id: claimId,
|
|
341
|
+
candidateSetId,
|
|
342
|
+
candidateId: rm.selectedCandidate.id,
|
|
343
|
+
subjectType: "system-field",
|
|
344
|
+
subjectId,
|
|
345
|
+
surface: "schema-mapping.profile",
|
|
346
|
+
claimType: "schema-mapping.field-link",
|
|
347
|
+
fieldOrBehavior: "maps-to",
|
|
348
|
+
value: { relation, sourceField, targetField, conversion },
|
|
349
|
+
status: rm.reviewOutcome.status,
|
|
350
|
+
impactLevel: "medium",
|
|
351
|
+
collectedBy: proposedBy,
|
|
352
|
+
actor: rm.reviewOutcome.actor,
|
|
353
|
+
metadata: {
|
|
354
|
+
survey: {
|
|
355
|
+
schemaMapping: {
|
|
356
|
+
pairKey: rm.pairKey,
|
|
357
|
+
sourceField,
|
|
358
|
+
targetField,
|
|
359
|
+
relation,
|
|
360
|
+
conversion,
|
|
361
|
+
},
|
|
362
|
+
},
|
|
363
|
+
},
|
|
364
|
+
};
|
|
365
|
+
claims.push(claimTarget);
|
|
366
|
+
}
|
|
367
|
+
const surveyInput = {
|
|
368
|
+
source,
|
|
369
|
+
generatedAt,
|
|
370
|
+
rawSources,
|
|
371
|
+
extractions,
|
|
372
|
+
candidateSets,
|
|
373
|
+
reviewOutcomes,
|
|
374
|
+
claims,
|
|
375
|
+
};
|
|
376
|
+
const bundle = buildSurveyTrustBundle(surveyInput);
|
|
377
|
+
// Attach IdentityLinks: one per accepted mapping.
|
|
378
|
+
// Each link ties the source-system-field subject to the target-system-field
|
|
379
|
+
// subject and back-references the mapping claim via mappingClaimId.
|
|
380
|
+
const identityLinks = [];
|
|
381
|
+
for (const rm of accepted) {
|
|
382
|
+
const meta = rm.selectedCandidate.metadata?.schemaMappingProposal;
|
|
383
|
+
const sourceField = meta?.sourceField ?? rm.proposal.sourceField;
|
|
384
|
+
const targetField = meta?.targetField ?? rm.proposal.targetField;
|
|
385
|
+
const relation = (meta?.relation ?? rm.proposal.relation);
|
|
386
|
+
const conversion = (meta?.conversion ?? rm.proposal.conversion);
|
|
387
|
+
const claimId = `schema-mapping.claim.${rm.pairKey}`;
|
|
388
|
+
const link = {
|
|
389
|
+
id: `schema-mapping.link.${rm.pairKey}`,
|
|
390
|
+
subjects: [
|
|
391
|
+
{ subjectType: "system-field", subjectId: `${sourceField.system}::${sourceField.entity}::${sourceField.field}` },
|
|
392
|
+
{ subjectType: "system-field", subjectId: `${targetField.system}::${targetField.entity}::${targetField.field}` },
|
|
393
|
+
],
|
|
394
|
+
reason: `Schema mapping: ${sourceField.system}.${sourceField.entity}.${sourceField.field} ${relation} ${targetField.system}.${targetField.entity}.${targetField.field}`,
|
|
395
|
+
attestedBy: rm.reviewOutcome.actor,
|
|
396
|
+
relation,
|
|
397
|
+
...(conversion ? { conversion } : {}),
|
|
398
|
+
mappingClaimId: claimId,
|
|
399
|
+
};
|
|
400
|
+
identityLinks.push(link);
|
|
401
|
+
}
|
|
402
|
+
return {
|
|
403
|
+
...bundle,
|
|
404
|
+
identityLinks: identityLinks.length > 0 ? identityLinks : undefined,
|
|
405
|
+
};
|
|
406
|
+
}
|
|
407
|
+
// ---------------------------------------------------------------------------
|
|
408
|
+
// Reference extractor (deterministic, for tests — not for production use)
|
|
409
|
+
// ---------------------------------------------------------------------------
|
|
410
|
+
/**
|
|
411
|
+
* Reference SchemaMappingExtractor for tests.
|
|
412
|
+
*
|
|
413
|
+
* REFERENCE IMPLEMENTATION ONLY — not suitable for production matching.
|
|
414
|
+
*
|
|
415
|
+
* Matching strategy: two fields are proposed as "equivalent" when they share
|
|
416
|
+
* the same field name (case-insensitive) and the same inferred type token
|
|
417
|
+
* ("string", "number", "boolean", "date", or absent) across two system schemas.
|
|
418
|
+
*
|
|
419
|
+
* Schema text format expected by this reference extractor:
|
|
420
|
+
* Each non-empty line is: "<entity>.<field>[:<type>]"
|
|
421
|
+
* e.g. "Contact.email:string" or "Account.revenue:number"
|
|
422
|
+
*
|
|
423
|
+
* This format is intentionally simple and transparent so tests can be
|
|
424
|
+
* deterministic without depending on any schema parsing library.
|
|
425
|
+
*/
|
|
426
|
+
export const referenceSchemaExtractor = {
|
|
427
|
+
name: "reference-schema-extractor",
|
|
428
|
+
extract(context) {
|
|
429
|
+
const now = new Date().toISOString();
|
|
430
|
+
const parsed = context.systems.map(({ system, schemaText }) => {
|
|
431
|
+
const fields = [];
|
|
432
|
+
for (const line of schemaText.split("\n")) {
|
|
433
|
+
const trimmed = line.trim();
|
|
434
|
+
if (!trimmed || trimmed.startsWith("#"))
|
|
435
|
+
continue;
|
|
436
|
+
// Format: entity.field[:type]
|
|
437
|
+
const colonIdx = trimmed.indexOf(":");
|
|
438
|
+
const namepart = colonIdx >= 0 ? trimmed.slice(0, colonIdx) : trimmed;
|
|
439
|
+
const type = colonIdx >= 0 ? trimmed.slice(colonIdx + 1).trim().toLowerCase() : "";
|
|
440
|
+
const dotIdx = namepart.indexOf(".");
|
|
441
|
+
if (dotIdx < 0)
|
|
442
|
+
continue;
|
|
443
|
+
const entity = namepart.slice(0, dotIdx).trim();
|
|
444
|
+
const field = namepart.slice(dotIdx + 1).trim();
|
|
445
|
+
if (!entity || !field)
|
|
446
|
+
continue;
|
|
447
|
+
fields.push({ entity, field, type, locator: `structured-field:${entity}.${field}` });
|
|
448
|
+
}
|
|
449
|
+
return { system, fields };
|
|
450
|
+
});
|
|
451
|
+
const proposals = [];
|
|
452
|
+
// Compare all pairs of systems
|
|
453
|
+
for (let i = 0; i < parsed.length; i++) {
|
|
454
|
+
for (let j = i + 1; j < parsed.length; j++) {
|
|
455
|
+
const sysA = parsed[i];
|
|
456
|
+
const sysB = parsed[j];
|
|
457
|
+
for (const fa of sysA.fields) {
|
|
458
|
+
for (const fb of sysB.fields) {
|
|
459
|
+
// Exact-name match (case-insensitive)
|
|
460
|
+
if (fa.field.toLowerCase() !== fb.field.toLowerCase())
|
|
461
|
+
continue;
|
|
462
|
+
// Type match (if both specified, they must be the same)
|
|
463
|
+
if (fa.type && fb.type && fa.type !== fb.type)
|
|
464
|
+
continue;
|
|
465
|
+
const confidence = fa.type && fb.type && fa.type === fb.type ? 0.9 : 0.75;
|
|
466
|
+
proposals.push({
|
|
467
|
+
id: `ref-schema-prop.${sysA.system}.${fa.entity}.${fa.field}__${sysB.system}.${fb.entity}.${fb.field}.${Date.now()}`,
|
|
468
|
+
sourceField: { system: sysA.system, entity: fa.entity, field: fa.field, locator: fa.locator },
|
|
469
|
+
targetField: { system: sysB.system, entity: fb.entity, field: fb.field, locator: fb.locator },
|
|
470
|
+
relation: "equivalent",
|
|
471
|
+
evidence: [
|
|
472
|
+
{ system: sysA.system, excerpt: `${fa.entity}.${fa.field}${fa.type ? `:${fa.type}` : ""}` },
|
|
473
|
+
{ system: sysB.system, excerpt: `${fb.entity}.${fb.field}${fb.type ? `:${fb.type}` : ""}` },
|
|
474
|
+
],
|
|
475
|
+
confidence,
|
|
476
|
+
rationale: `Reference extractor: exact field-name match "${fa.field}"${fa.type && fb.type ? ` with matching type "${fa.type}"` : ""} across ${sysA.system} and ${sysB.system}.`,
|
|
477
|
+
proposedBy: "reference-schema-extractor",
|
|
478
|
+
proposedAt: now,
|
|
479
|
+
});
|
|
480
|
+
}
|
|
481
|
+
}
|
|
482
|
+
}
|
|
483
|
+
}
|
|
484
|
+
return proposals;
|
|
485
|
+
},
|
|
486
|
+
};
|