@kontourai/survey 2.5.0 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -1
- package/dist/examples/calibrated-auto-accept.d.ts +22 -15
- package/dist/examples/calibrated-auto-accept.js +40 -36
- package/dist/src/agent-utterance.d.ts +87 -11
- package/dist/src/agent-utterance.js +135 -44
- package/dist/src/calibration.d.ts +48 -21
- package/dist/src/calibration.js +72 -33
- package/dist/src/console/review-console-server.d.ts +3 -1
- package/dist/src/console/review-console-server.js +203 -50
- package/dist/src/extraction-envelope.d.ts +22 -0
- package/dist/src/extraction-envelope.js +25 -4
- package/dist/src/index.d.ts +9 -8
- package/dist/src/index.js +2 -2
- package/dist/src/inquiry-mapping.d.ts +15 -1
- package/dist/src/inquiry-mapping.js +10 -2
- package/dist/src/mcp/review-mcp.js +219 -279
- package/dist/src/producer-profile.d.ts +41 -2
- package/dist/src/producer-profile.js +29 -2
- package/dist/src/review-session-file.d.ts +64 -0
- package/dist/src/review-session-file.js +320 -0
- package/dist/src/review-workbench/edited-value.d.ts +70 -0
- package/dist/src/review-workbench/edited-value.js +147 -0
- package/dist/src/review-workbench/review-presentation.d.ts +44 -0
- package/dist/src/review-workbench/review-presentation.js +49 -0
- package/dist/src/review-workbench/review-queue-session.js +8 -1
- package/dist/src/review-workbench/review-session-replay.d.ts +35 -1
- package/dist/src/review-workbench/review-session-replay.js +77 -0
- package/dist/src/review-workbench/review-workbench.d.ts +8 -4
- package/dist/src/review-workbench/review-workbench.js +19 -7
- package/dist/src/review-workbench/server-review-session.d.ts +3 -1
- package/dist/src/review-workbench/server-review-session.js +1 -0
- package/dist/src/reviewed-candidate-resolution.js +13 -7
- package/dist/src/schema-mapping.d.ts +23 -0
- package/dist/src/schema-mapping.js +30 -20
- package/dist/src/to-surface.d.ts +30 -6
- package/dist/src/to-surface.js +196 -18
- package/dist/src/types.d.ts +39 -0
- package/package.json +8 -4
package/README.md
CHANGED
|
@@ -40,6 +40,10 @@ The Review Workbench rendering a real example queue — current vs proposed valu
|
|
|
40
40
|
npm install @kontourai/survey @kontourai/surface
|
|
41
41
|
```
|
|
42
42
|
|
|
43
|
+
Survey supports `@kontourai/surface` 2.13 and later 2.x, and 3.x, and CI tests
|
|
44
|
+
the lowest 2.x and the newest 3.x. Install either major and your project and
|
|
45
|
+
Survey share one Surface copy (`npm ls @kontourai/surface` shows one entry).
|
|
46
|
+
|
|
43
47
|
Requires Node.js >=22. TypeScript >=5.0 is required to compile against
|
|
44
48
|
Survey's published type declarations (`defineProductVocabulary`'s `const`
|
|
45
49
|
type parameters are TS 5.0+ syntax) — JavaScript consumers are unaffected;
|
|
@@ -203,7 +207,8 @@ host app should read as itself.
|
|
|
203
207
|
|
|
204
208
|
## Review MCP
|
|
205
209
|
|
|
206
|
-
Drive review-queue decisions from
|
|
210
|
+
Drive review-queue decisions from any MCP host. The official server runtime
|
|
211
|
+
automatically supports MCP 2026-07-28 discovery and existing legacy clients:
|
|
207
212
|
|
|
208
213
|
```sh
|
|
209
214
|
npx survey-review-mcp --session path/to/session.json
|
|
@@ -1,29 +1,36 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Worked example:
|
|
2
|
+
* Worked example: EXPERIMENTAL confidence calibration from review history.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
4
|
+
* Calibration summarizes how often reviewers affirmed an extractor's proposals.
|
|
5
|
+
* Both outputs below are experimental descriptive statistics (#279), not
|
|
6
|
+
* validated probabilities or policy:
|
|
6
7
|
*
|
|
7
|
-
* 1.
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
* (
|
|
12
|
-
*
|
|
8
|
+
* 1. `deriveCalibration` reports per-decile affirmation rates and, only when a
|
|
9
|
+
* decile has enough samples AND a one-sided 95% Wilson lower bound on its
|
|
10
|
+
* accuracy clears the target, a `suggestedThreshold`. A thin history gets
|
|
11
|
+
* no threshold at all. Do not feed `suggestedThreshold` into an auto-accept
|
|
12
|
+
* policy (`autoAcceptMinConfidence` / `minConfidence`) as-is: it is not an
|
|
13
|
+
* evaluated gate.
|
|
13
14
|
*
|
|
14
|
-
* 2.
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
15
|
+
* 2. `buildSurveyTrustBundle({ calibration: { experimentalConclusionValue:
|
|
16
|
+
* true, metrics } })` sets `conclusionConfidence.value` on affirmed claims
|
|
17
|
+
* to their extractor/field GROUP's affirmation rate. Every affirmed claim in
|
|
18
|
+
* the group gets the same base rate whatever its own confidence, so it is
|
|
19
|
+
* not a per-claim probability. Without the experimental flag no value is set.
|
|
18
20
|
*
|
|
19
|
-
* Calibration is advisory (ADR 0003 §4):
|
|
20
|
-
* operator wires into policy, and the produced value never changes claim status.
|
|
21
|
+
* Calibration is advisory (ADR 0003 §4): it never changes claim status.
|
|
21
22
|
*
|
|
22
23
|
* Run: `node dist/examples/calibrated-auto-accept.js`
|
|
23
24
|
*/
|
|
24
25
|
export interface CalibratedAutoAcceptResult {
|
|
26
|
+
/** Threshold from a 60-samples-per-decile history (enough evidence). */
|
|
25
27
|
readonly suggestedThreshold: number | undefined;
|
|
28
|
+
/** Threshold from a 10-samples-per-decile history (withheld: too thin). */
|
|
29
|
+
readonly sparseHistoryThreshold: number | undefined;
|
|
26
30
|
readonly groupAccuracy: number | undefined;
|
|
31
|
+
/** Values with the experimental opt-in: the group base rate on each claim. */
|
|
27
32
|
readonly producedValues: ReadonlyArray<number | undefined>;
|
|
33
|
+
/** Values with `calibration` enabled but no experimental opt-in: none. */
|
|
34
|
+
readonly defaultValues: ReadonlyArray<number | undefined>;
|
|
28
35
|
}
|
|
29
36
|
export declare function runCalibratedAutoAccept(): CalibratedAutoAcceptResult;
|
|
@@ -1,23 +1,24 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Worked example:
|
|
2
|
+
* Worked example: EXPERIMENTAL confidence calibration from review history.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
4
|
+
* Calibration summarizes how often reviewers affirmed an extractor's proposals.
|
|
5
|
+
* Both outputs below are experimental descriptive statistics (#279), not
|
|
6
|
+
* validated probabilities or policy:
|
|
6
7
|
*
|
|
7
|
-
* 1.
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
* (
|
|
12
|
-
*
|
|
8
|
+
* 1. `deriveCalibration` reports per-decile affirmation rates and, only when a
|
|
9
|
+
* decile has enough samples AND a one-sided 95% Wilson lower bound on its
|
|
10
|
+
* accuracy clears the target, a `suggestedThreshold`. A thin history gets
|
|
11
|
+
* no threshold at all. Do not feed `suggestedThreshold` into an auto-accept
|
|
12
|
+
* policy (`autoAcceptMinConfidence` / `minConfidence`) as-is: it is not an
|
|
13
|
+
* evaluated gate.
|
|
13
14
|
*
|
|
14
|
-
* 2.
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
15
|
+
* 2. `buildSurveyTrustBundle({ calibration: { experimentalConclusionValue:
|
|
16
|
+
* true, metrics } })` sets `conclusionConfidence.value` on affirmed claims
|
|
17
|
+
* to their extractor/field GROUP's affirmation rate. Every affirmed claim in
|
|
18
|
+
* the group gets the same base rate whatever its own confidence, so it is
|
|
19
|
+
* not a per-claim probability. Without the experimental flag no value is set.
|
|
18
20
|
*
|
|
19
|
-
* Calibration is advisory (ADR 0003 §4):
|
|
20
|
-
* operator wires into policy, and the produced value never changes claim status.
|
|
21
|
+
* Calibration is advisory (ADR 0003 §4): it never changes claim status.
|
|
21
22
|
*
|
|
22
23
|
* Run: `node dist/examples/calibrated-auto-accept.js`
|
|
23
24
|
*/
|
|
@@ -31,15 +32,15 @@ const BATCH_AT = "2026-07-01T00:00:00.000Z";
|
|
|
31
32
|
* almost always affirmed by reviewers and low-confidence ones were mostly
|
|
32
33
|
* rejected — the pattern that makes an empirical threshold meaningful.
|
|
33
34
|
*/
|
|
34
|
-
function buildReviewHistory() {
|
|
35
|
+
function buildReviewHistory(samplesPerDecile) {
|
|
35
36
|
const extractions = [];
|
|
36
37
|
const candidateSets = [];
|
|
37
38
|
const reviewOutcomes = [];
|
|
38
|
-
let
|
|
39
|
+
let seq = 0;
|
|
39
40
|
const addSamples = (count, confidence, affirmed) => {
|
|
40
41
|
for (let i = 0; i < count; i += 1) {
|
|
41
|
-
const key = `h-${
|
|
42
|
-
|
|
42
|
+
const key = `h-${seq}`;
|
|
43
|
+
seq += 1;
|
|
43
44
|
extractions.push({
|
|
44
45
|
id: `${key}-ext`,
|
|
45
46
|
sourceId: `${key}-src`,
|
|
@@ -67,33 +68,34 @@ function buildReviewHistory() {
|
|
|
67
68
|
});
|
|
68
69
|
}
|
|
69
70
|
};
|
|
70
|
-
|
|
71
|
-
addSamples(
|
|
72
|
-
addSamples(
|
|
73
|
-
addSamples(
|
|
71
|
+
const n = samplesPerDecile;
|
|
72
|
+
addSamples(n, 0.95, n); // top decile: all affirmed
|
|
73
|
+
addSamples(n, 0.85, n - 1); // 0.8–0.9: all but one affirmed
|
|
74
|
+
addSamples(n, 0.75, Math.round(n / 2)); // 0.7–0.8: half affirmed → ends the run
|
|
75
|
+
addSamples(n, 0.55, Math.round(n / 10)); // 0.5–0.6: mostly rejected
|
|
74
76
|
return { reviewOutcomes, candidateSets, extractions };
|
|
75
77
|
}
|
|
76
78
|
export function runCalibratedAutoAccept() {
|
|
77
79
|
// (1) Derive the empirical calibration curve over the review history.
|
|
78
|
-
const
|
|
79
|
-
|
|
80
|
-
minBinSamples: 5, // a decile needs this many samples to ground the threshold
|
|
81
|
-
});
|
|
80
|
+
const options = { targetAccuracy: 0.9 }; // default 30-sample floor per decile
|
|
81
|
+
const metrics = deriveCalibration(buildReviewHistory(60), options);
|
|
82
82
|
const suggestedThreshold = metrics.overall.suggestedThreshold;
|
|
83
|
+
const sparseHistoryThreshold = deriveCalibration(buildReviewHistory(10), options).overall.suggestedThreshold;
|
|
83
84
|
const group = metrics.byExtractorField.find((g) => g.extractor === EXTRACTOR && g.field === FIELD);
|
|
84
|
-
//
|
|
85
|
-
// surveySchemaMapping(context, extractor, { autoAcceptMinConfidence: suggestedThreshold })
|
|
86
|
-
// Proposals at/above it were empirically affirmed often enough to auto-accept.
|
|
87
|
-
// (2) Produce calibrated conclusion confidence on a new batch of affirmed claims.
|
|
85
|
+
// (2) Attach the group affirmation rate to a new batch of affirmed claims.
|
|
88
86
|
const input = new SurveyInputBuilder({ source: "example-consumer:calibrated", generatedAt: BATCH_AT })
|
|
89
87
|
.addObservation(affirmedObservation("entity-1", 0.85))
|
|
90
88
|
.addObservation(affirmedObservation("entity-2", 0.92))
|
|
91
89
|
.build();
|
|
92
90
|
// Prefer metrics computed over history (not just this batch), so a claim's own
|
|
93
91
|
// outcome does not feed its own value.
|
|
94
|
-
const bundle = buildSurveyTrustBundle(input, {
|
|
92
|
+
const bundle = buildSurveyTrustBundle(input, {
|
|
93
|
+
calibration: { experimentalConclusionValue: true, metrics, minSamples: 20 },
|
|
94
|
+
});
|
|
95
95
|
const producedValues = bundle.claims.map((c) => c.conclusionConfidence?.value);
|
|
96
|
-
|
|
96
|
+
const defaultValues = buildSurveyTrustBundle(input, { calibration: { metrics, minSamples: 20 } })
|
|
97
|
+
.claims.map((c) => c.conclusionConfidence?.value);
|
|
98
|
+
return { suggestedThreshold, sparseHistoryThreshold, groupAccuracy: group?.empiricalAccuracy, producedValues, defaultValues };
|
|
97
99
|
}
|
|
98
100
|
function affirmedObservation(subjectId, confidence) {
|
|
99
101
|
return {
|
|
@@ -128,8 +130,10 @@ function affirmedObservation(subjectId, confidence) {
|
|
|
128
130
|
if (process.argv[1]?.endsWith("calibrated-auto-accept.js")) {
|
|
129
131
|
const result = runCalibratedAutoAccept();
|
|
130
132
|
console.log(JSON.stringify({
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
133
|
+
experimentalSuggestedThreshold: result.suggestedThreshold,
|
|
134
|
+
sparseHistorySuggestedThreshold: result.sparseHistoryThreshold,
|
|
135
|
+
groupAffirmationRate: result.groupAccuracy,
|
|
136
|
+
experimentalConclusionValues: result.producedValues,
|
|
137
|
+
valuesWithoutOptIn: result.defaultValues,
|
|
134
138
|
}, null, 2));
|
|
135
139
|
}
|
|
@@ -16,7 +16,7 @@
|
|
|
16
16
|
* has full provenance (excerpt, span, extractor name, confidence) and is
|
|
17
17
|
* run through the Inquiry pipeline rather than treated as authoritative.
|
|
18
18
|
*/
|
|
19
|
-
import type { DerivationRule, InquiryRecord, TrustBundle } from "@kontourai/surface";
|
|
19
|
+
import type { DerivationRule, InquiryRecord, TrustBundle, TrustStatus } from "@kontourai/surface";
|
|
20
20
|
import type { CanonicalClaimTarget } from "@kontourai/surface";
|
|
21
21
|
import type { Candidate, CandidateSet, Extraction, RawSource, SurveyInput } from "./types.js";
|
|
22
22
|
import type { InquiryMapping } from "./inquiry-mapping.js";
|
|
@@ -52,17 +52,63 @@ export interface UtteranceClaimExtractor {
|
|
|
52
52
|
extract(utterance: string): ExtractedStatement[] | Promise<ExtractedStatement[]>;
|
|
53
53
|
}
|
|
54
54
|
/**
|
|
55
|
-
* Badge values for each extracted statement
|
|
56
|
-
*
|
|
57
|
-
*
|
|
58
|
-
*
|
|
59
|
-
*
|
|
60
|
-
*
|
|
61
|
-
*
|
|
62
|
-
*
|
|
63
|
-
*
|
|
55
|
+
* Badge values for each extracted statement.
|
|
56
|
+
*
|
|
57
|
+
* A badge grades THE STATEMENT, not the target. It is a function of two
|
|
58
|
+
* things: the status of the bundle's answer for the statement's canonical
|
|
59
|
+
* target, and how the statement's own asserted value compares to that
|
|
60
|
+
* answer's value.
|
|
61
|
+
*
|
|
62
|
+
* The status half of the vocabulary is Surface's `TrustStatus` verbatim —
|
|
63
|
+
* Survey does not mint parallel status terms for concepts Surface (and the
|
|
64
|
+
* Hachure core record shapes it implements) already name. Only two badge
|
|
65
|
+
* values are Survey-side additions, because they describe the
|
|
66
|
+
* statement-vs-answer relation rather than a claim's standing:
|
|
67
|
+
*
|
|
68
|
+
* - "contradicted": the bundle has an answer that would otherwise read as
|
|
69
|
+
* support ("verified", "assumed", "stale") and the statement asserts a
|
|
70
|
+
* DIFFERENT value. Mirrors the Hachure `contradiction` transparency-gap
|
|
71
|
+
* type (merge.md §7b): a value conflict is surfaced, never silently
|
|
72
|
+
* resolved in favour of one side.
|
|
73
|
+
* - "unsupported": the inquiry did not resolve to an answer at all — no
|
|
74
|
+
* mapping and no registered claim for the target. This value means "there
|
|
75
|
+
* is nothing here to compare against", and nothing else: a claim that is
|
|
76
|
+
* registered but merely awaiting review badges "proposed", and one with no
|
|
77
|
+
* evidence badges "unknown".
|
|
78
|
+
*
|
|
79
|
+
* Every other badge is the answer's `TrustStatus` passed through unchanged,
|
|
80
|
+
* so there is no fall-through path that can report a real status under a
|
|
81
|
+
* label meaning "no such claim".
|
|
82
|
+
*/
|
|
83
|
+
export type StatementBadge = TrustStatus | "contradicted" | "unsupported";
|
|
84
|
+
/**
|
|
85
|
+
* How a statement's asserted value compares to the bundle's answer value.
|
|
86
|
+
*
|
|
87
|
+
* - "agrees": both are comparable scalars and equivalent under
|
|
88
|
+
* `assertionComparisonKey`.
|
|
89
|
+
* - "contradicts": both are comparable scalars and NOT equivalent.
|
|
90
|
+
* - "not-compared": no comparison was possible — the inquiry produced no
|
|
91
|
+
* answer, or the extractor parsed no value out of the statement
|
|
92
|
+
* (`ExtractedStatement.value` absent), or one side is not a scalar.
|
|
93
|
+
* Never treated as agreement.
|
|
94
|
+
*/
|
|
95
|
+
export type StatementValueComparison = "agrees" | "contradicts" | "not-compared";
|
|
96
|
+
/**
|
|
97
|
+
* How the Extraction's `text-span:` locator was resolved for a statement.
|
|
98
|
+
*
|
|
99
|
+
* - "span": the extractor supplied an explicit character span; the locator
|
|
100
|
+
* points where the extractor said it does.
|
|
101
|
+
* - "excerpt-match": no span, but the excerpt was found verbatim in the
|
|
102
|
+
* utterance; the locator points at that occurrence.
|
|
103
|
+
* - "unanchored-fallback": no span AND the excerpt does not occur in the
|
|
104
|
+
* utterance. The locator is still well-formed (`text-span:0-<length>`) so
|
|
105
|
+
* downstream producer discipline holds, but it is a length-shaped
|
|
106
|
+
* placeholder anchored at offset 0 — it does NOT point at the excerpt, and
|
|
107
|
+
* the text it spans is unrelated prose. Anything that resolves the locator
|
|
108
|
+
* against the source MUST check this field first; a hallucinated excerpt
|
|
109
|
+
* lands here.
|
|
64
110
|
*/
|
|
65
|
-
export type
|
|
111
|
+
export type LocatorResolution = "span" | "excerpt-match" | "unanchored-fallback";
|
|
66
112
|
export interface UtteranceStatement {
|
|
67
113
|
excerpt: string;
|
|
68
114
|
span?: {
|
|
@@ -70,6 +116,29 @@ export interface UtteranceStatement {
|
|
|
70
116
|
end: number;
|
|
71
117
|
};
|
|
72
118
|
target: CanonicalClaimTarget;
|
|
119
|
+
/**
|
|
120
|
+
* The value the statement asserted, verbatim from the extractor
|
|
121
|
+
* (`ExtractedStatement.value`) — not normalized, not defaulted. Absent when
|
|
122
|
+
* the extractor parsed no value, which is exactly when `valueComparison`
|
|
123
|
+
* is "not-compared". This is the field a reader needs to see WHAT was
|
|
124
|
+
* compared against the bundle's answer.
|
|
125
|
+
*/
|
|
126
|
+
assertedValue?: unknown;
|
|
127
|
+
/** How `assertedValue` compares to `inquiryRecord.answer?.value`. */
|
|
128
|
+
valueComparison: StatementValueComparison;
|
|
129
|
+
/** Human-readable account of the comparison, naming both sides. */
|
|
130
|
+
comparisonRationale: string;
|
|
131
|
+
/**
|
|
132
|
+
* The Survey provenance records this statement produced: its Extraction,
|
|
133
|
+
* its Candidate, and the per-target Candidate Set it belongs to. The
|
|
134
|
+
* Candidate Set carries the Candidate Conflict verdict and rationale when
|
|
135
|
+
* two statements in the SAME utterance disagree about one target
|
|
136
|
+
* (`candidateSet.status === "conflict"`), which is a different signal from
|
|
137
|
+
* `valueComparison` (statement vs bundle).
|
|
138
|
+
*/
|
|
139
|
+
records: UtteranceStatementRecords;
|
|
140
|
+
/** How this statement's Extraction locator was resolved. */
|
|
141
|
+
locatorResolution: LocatorResolution;
|
|
73
142
|
inquiryRecord: InquiryRecord;
|
|
74
143
|
badge: StatementBadge;
|
|
75
144
|
}
|
|
@@ -107,6 +176,13 @@ interface UtteranceRecordsResult {
|
|
|
107
176
|
records: UtteranceStatementRecords[];
|
|
108
177
|
extractions: Extraction[];
|
|
109
178
|
candidateSets: CandidateSet[];
|
|
179
|
+
/**
|
|
180
|
+
* `locatorResolutions[idx]` is how `records[idx]`'s Extraction locator was
|
|
181
|
+
* resolved — a parallel array rather than a fourth key on
|
|
182
|
+
* `UtteranceStatementRecords`, whose shape is a pinned contract. The same
|
|
183
|
+
* value is also on `records[idx].extraction.metadata.agentUtterance`.
|
|
184
|
+
*/
|
|
185
|
+
locatorResolutions: LocatorResolution[];
|
|
110
186
|
}
|
|
111
187
|
/**
|
|
112
188
|
* Build the full set of Survey records for every extracted statement in one
|
|
@@ -59,8 +59,10 @@ function buildUtteranceExtraction(params) {
|
|
|
59
59
|
const candidateId = `${statementId}.candidate`;
|
|
60
60
|
// Compute locator — required for non-manual-entry sources
|
|
61
61
|
// (assertProducerDiscipline throws without it). Source Locator rule:
|
|
62
|
-
// span-first, excerpt-fallback —
|
|
63
|
-
|
|
62
|
+
// span-first, excerpt-fallback — locator VALUES unchanged from Slice 1;
|
|
63
|
+
// what is new is that the record now says which branch produced them, so
|
|
64
|
+
// an unanchored placeholder is never mistaken for a resolved pointer.
|
|
65
|
+
const { locator, resolution: locatorResolution } = resolveUtteranceLocator(utterance, statement);
|
|
64
66
|
const extraction = {
|
|
65
67
|
id: extractionId,
|
|
66
68
|
sourceId,
|
|
@@ -77,6 +79,7 @@ function buildUtteranceExtraction(params) {
|
|
|
77
79
|
excerpt: statement.excerpt,
|
|
78
80
|
extractorName,
|
|
79
81
|
confidence: statement.confidence,
|
|
82
|
+
locatorResolution,
|
|
80
83
|
},
|
|
81
84
|
},
|
|
82
85
|
};
|
|
@@ -93,7 +96,7 @@ function buildUtteranceExtraction(params) {
|
|
|
93
96
|
confidence: statement.confidence,
|
|
94
97
|
},
|
|
95
98
|
};
|
|
96
|
-
return { extraction, proposal };
|
|
99
|
+
return { extraction, proposal, locatorResolution };
|
|
97
100
|
}
|
|
98
101
|
/**
|
|
99
102
|
* Group extraction/proposal pairs by canonical target and project each
|
|
@@ -166,7 +169,7 @@ function groupUtteranceExtractionsByTarget(sourceId, items) {
|
|
|
166
169
|
export function buildUtteranceRecords(params) {
|
|
167
170
|
const { sourceId, utterance, extracted, extractorName, observedAt } = params;
|
|
168
171
|
const items = extracted.map((statement, idx) => {
|
|
169
|
-
const { extraction, proposal } = buildUtteranceExtraction({
|
|
172
|
+
const { extraction, proposal, locatorResolution } = buildUtteranceExtraction({
|
|
170
173
|
sourceId,
|
|
171
174
|
idx,
|
|
172
175
|
statement,
|
|
@@ -174,7 +177,7 @@ export function buildUtteranceRecords(params) {
|
|
|
174
177
|
extractorName,
|
|
175
178
|
observedAt,
|
|
176
179
|
});
|
|
177
|
-
return { statement, extraction, proposal };
|
|
180
|
+
return { statement, extraction, proposal, locatorResolution };
|
|
178
181
|
});
|
|
179
182
|
const groups = groupUtteranceExtractionsByTarget(sourceId, items);
|
|
180
183
|
const records = items.map((item) => {
|
|
@@ -186,6 +189,7 @@ export function buildUtteranceRecords(params) {
|
|
|
186
189
|
records,
|
|
187
190
|
extractions: items.map((i) => i.extraction),
|
|
188
191
|
candidateSets: [...groups.values()].map((g) => g.candidateSet),
|
|
192
|
+
locatorResolutions: items.map((i) => i.locatorResolution),
|
|
189
193
|
};
|
|
190
194
|
}
|
|
191
195
|
// ---------------------------------------------------------------------------
|
|
@@ -234,7 +238,7 @@ export function utteranceToSurveyInput(utterance, extracted, context) {
|
|
|
234
238
|
// core) — replaces the old per-statement builder call. Claims below stay
|
|
235
239
|
// one-per-statement; `record.candidateSet.id`/`record.candidate.id` may be
|
|
236
240
|
// shared across several claims when statements share a target (legal).
|
|
237
|
-
const { records, extractions, candidateSets } = buildUtteranceRecords({
|
|
241
|
+
const { records, extractions, candidateSets, locatorResolutions } = buildUtteranceRecords({
|
|
238
242
|
sourceId,
|
|
239
243
|
utterance,
|
|
240
244
|
extracted,
|
|
@@ -270,6 +274,9 @@ export function utteranceToSurveyInput(utterance, extracted, context) {
|
|
|
270
274
|
span: statement.span,
|
|
271
275
|
confidence: statement.confidence,
|
|
272
276
|
locator: record.extraction.locator,
|
|
277
|
+
// Travels with the locator so a downstream reader of the Claim
|
|
278
|
+
// never has to assume the locator resolved.
|
|
279
|
+
locatorResolution: locatorResolutions[idx],
|
|
273
280
|
},
|
|
274
281
|
},
|
|
275
282
|
},
|
|
@@ -318,13 +325,14 @@ export async function surveyAgentUtterance(utterance, extractor, context) {
|
|
|
318
325
|
};
|
|
319
326
|
// Step 2: Extract statements
|
|
320
327
|
const extracted = await Promise.resolve(extractor.extract(utterance));
|
|
321
|
-
// Batched, grouped provenance construction
|
|
322
|
-
//
|
|
323
|
-
//
|
|
324
|
-
//
|
|
325
|
-
//
|
|
326
|
-
//
|
|
327
|
-
|
|
328
|
+
// Batched, grouped provenance construction. Grouping needs every statement
|
|
329
|
+
// of a target's group present at once, so this cannot be a per-statement
|
|
330
|
+
// call. The result is now carried onto every UtteranceStatement
|
|
331
|
+
// (`records`), so the extractor's confidence, locator, Candidate and
|
|
332
|
+
// Candidate Set — including the Candidate Conflict verdict when two
|
|
333
|
+
// statements in this utterance disagree about one target — are observable
|
|
334
|
+
// in the report instead of being computed and dropped on the floor.
|
|
335
|
+
const { records, locatorResolutions } = buildUtteranceRecords({
|
|
328
336
|
sourceId,
|
|
329
337
|
utterance,
|
|
330
338
|
extracted,
|
|
@@ -333,7 +341,7 @@ export async function surveyAgentUtterance(utterance, extractor, context) {
|
|
|
333
341
|
});
|
|
334
342
|
// Step 3 & 4: Resolve each statement and build the report
|
|
335
343
|
const statements = [];
|
|
336
|
-
for (const statement of extracted) {
|
|
344
|
+
for (const [idx, statement] of extracted.entries()) {
|
|
337
345
|
// Resolve the claim
|
|
338
346
|
let inquiryRecord;
|
|
339
347
|
if (mappings && mappings.length > 0) {
|
|
@@ -358,11 +366,17 @@ export async function surveyAgentUtterance(utterance, extractor, context) {
|
|
|
358
366
|
// Resolve directly by canonical target
|
|
359
367
|
inquiryRecord = resolveByTarget(bundle, statement.target, agentId, observedAt, rules, now);
|
|
360
368
|
}
|
|
361
|
-
const
|
|
369
|
+
const comparison = compareAssertedValue(statement, inquiryRecord);
|
|
370
|
+
const badge = badgeFor(inquiryRecord, comparison.valueComparison);
|
|
362
371
|
statements.push({
|
|
363
372
|
excerpt: statement.excerpt,
|
|
364
373
|
span: statement.span,
|
|
365
374
|
target: statement.target,
|
|
375
|
+
...(hasAssertedValue(statement) ? { assertedValue: statement.value } : {}),
|
|
376
|
+
valueComparison: comparison.valueComparison,
|
|
377
|
+
comparisonRationale: comparison.rationale,
|
|
378
|
+
records: records[idx],
|
|
379
|
+
locatorResolution: locatorResolutions[idx],
|
|
366
380
|
inquiryRecord,
|
|
367
381
|
badge,
|
|
368
382
|
});
|
|
@@ -389,45 +403,122 @@ function canonicalTargetKey(target) {
|
|
|
389
403
|
function targetToQuestion(target) {
|
|
390
404
|
return `${target.subjectId} ${target.fieldOrBehavior}`;
|
|
391
405
|
}
|
|
392
|
-
|
|
406
|
+
/**
|
|
407
|
+
* Answer statuses a reader takes as support for what the statement said.
|
|
408
|
+
* These — and only these — are the statuses a contradiction must override:
|
|
409
|
+
* badging a statement "verified" when it asserts a value the verified claim
|
|
410
|
+
* denies is the exact failure this profile exists to prevent. For any other
|
|
411
|
+
* status the claim's own standing is already the more informative thing to
|
|
412
|
+
* show, and the contradiction stays legible in `valueComparison`.
|
|
413
|
+
*/
|
|
414
|
+
const SUPPORTING_STATUSES = new Set(["verified", "assumed", "stale"]);
|
|
415
|
+
function hasAssertedValue(statement) {
|
|
416
|
+
return statement.value !== undefined;
|
|
417
|
+
}
|
|
418
|
+
/**
|
|
419
|
+
* Whether a value can take part in a statement-vs-answer comparison at all.
|
|
420
|
+
* Scalars can; objects and arrays cannot, because an utterance extractor
|
|
421
|
+
* pulls a token out of prose and there is no defensible way to decide
|
|
422
|
+
* whether that token "is" a structured value. Those report "not-compared"
|
|
423
|
+
* rather than being asserted to contradict.
|
|
424
|
+
*/
|
|
425
|
+
function isComparableScalar(value) {
|
|
426
|
+
return value === null || ["string", "number", "boolean"].includes(typeof value);
|
|
427
|
+
}
|
|
428
|
+
/**
|
|
429
|
+
* The statement-vs-answer comparison key.
|
|
430
|
+
*
|
|
431
|
+
* This is deliberately NOT `utteranceEquivalenceKey`. That key compares two
|
|
432
|
+
* values from the SAME extractor, which share one type discipline, so it
|
|
433
|
+
* refuses cross-type equality on purpose (5 and "5" from one extractor
|
|
434
|
+
* really are different findings). A statement-vs-answer comparison crosses a
|
|
435
|
+
* boundary: the left side is a token the extractor pulled out of prose, the
|
|
436
|
+
* right side is a value the producer typed. Comparing those two by
|
|
437
|
+
* `typeof` would badge every true statement about a numeric or boolean field
|
|
438
|
+
* as a contradiction — a false accusation, which damages the badge exactly
|
|
439
|
+
* as much as a false green does.
|
|
440
|
+
*
|
|
441
|
+
* So scalars are compared by their canonical TEXT rendering, trimmed and
|
|
442
|
+
* lowercased. This bridges "the agent wrote 95" to "the producer stored 95"
|
|
443
|
+
* without losing a genuine disagreement: "5" and "6" still differ, and the
|
|
444
|
+
* bridge is named in `comparisonRationale` rather than applied silently.
|
|
445
|
+
*/
|
|
446
|
+
function assertionComparisonKey(value) {
|
|
447
|
+
return String(value).trim().toLowerCase();
|
|
448
|
+
}
|
|
449
|
+
/**
|
|
450
|
+
* Compare the statement's own asserted value against the bundle's answer.
|
|
451
|
+
*
|
|
452
|
+
* An absent asserted value is never treated as agreement: an extractor that
|
|
453
|
+
* parsed no value has asserted nothing to check, so it reports
|
|
454
|
+
* "not-compared".
|
|
455
|
+
*/
|
|
456
|
+
function compareAssertedValue(statement, record) {
|
|
457
|
+
const targetKey = canonicalTargetKey(statement.target);
|
|
458
|
+
const answer = record.answer;
|
|
459
|
+
if (!answer) {
|
|
460
|
+
return {
|
|
461
|
+
valueComparison: "not-compared",
|
|
462
|
+
rationale: `No answer for ${targetKey} (inquiry outcome: ${record.outcome}); nothing to compare the statement against.`,
|
|
463
|
+
};
|
|
464
|
+
}
|
|
465
|
+
if (!hasAssertedValue(statement)) {
|
|
466
|
+
return {
|
|
467
|
+
valueComparison: "not-compared",
|
|
468
|
+
rationale: `The extractor parsed no value out of this statement, so nothing was compared against the ${answer.status} answer for ${targetKey}.`,
|
|
469
|
+
};
|
|
470
|
+
}
|
|
471
|
+
if (!isComparableScalar(statement.value) || !isComparableScalar(answer.value)) {
|
|
472
|
+
return {
|
|
473
|
+
valueComparison: "not-compared",
|
|
474
|
+
rationale: `Statement value ${JSON.stringify(statement.value) ?? "undefined"} and the ${answer.status} answer ${JSON.stringify(answer.value) ?? "undefined"} for ${targetKey} are not both scalars; no comparison was attempted.`,
|
|
475
|
+
};
|
|
476
|
+
}
|
|
477
|
+
const asserted = assertionComparisonKey(statement.value);
|
|
478
|
+
const answered = assertionComparisonKey(answer.value);
|
|
479
|
+
const shown = `statement "${asserted}" vs ${answer.status} answer "${answered}" (compared as text)`;
|
|
480
|
+
return asserted === answered
|
|
481
|
+
? { valueComparison: "agrees", rationale: `Agrees for ${targetKey}: ${shown}.` }
|
|
482
|
+
: { valueComparison: "contradicts", rationale: `Contradicts for ${targetKey}: ${shown}.` };
|
|
483
|
+
}
|
|
484
|
+
/**
|
|
485
|
+
* Grade the STATEMENT: the answer's status, overridden by "contradicted"
|
|
486
|
+
* when the statement asserts something the answer denies and that answer
|
|
487
|
+
* would otherwise have read as support.
|
|
488
|
+
*/
|
|
489
|
+
function badgeFor(record, valueComparison) {
|
|
393
490
|
if (record.outcome === "unsupported")
|
|
394
491
|
return "unsupported";
|
|
395
492
|
const status = record.answer?.status;
|
|
396
493
|
if (!status)
|
|
397
494
|
return "unsupported";
|
|
398
|
-
if (
|
|
399
|
-
return "
|
|
400
|
-
|
|
401
|
-
return "assumed";
|
|
402
|
-
if (status === "stale")
|
|
403
|
-
return "stale";
|
|
404
|
-
if (status === "disputed")
|
|
405
|
-
return "disputed";
|
|
406
|
-
if (status === "rejected" || status === "superseded")
|
|
407
|
-
return "rejected";
|
|
408
|
-
return "unsupported";
|
|
495
|
+
if (valueComparison === "contradicts" && SUPPORTING_STATUSES.has(status))
|
|
496
|
+
return "contradicted";
|
|
497
|
+
return status;
|
|
409
498
|
}
|
|
410
499
|
/**
|
|
411
|
-
*
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
*
|
|
420
|
-
*
|
|
421
|
-
*
|
|
500
|
+
* The single Source Locator rule this module guarantees: span-first,
|
|
501
|
+
* excerpt-match second, unanchored placeholder last — and it always reports
|
|
502
|
+
* WHICH of the three produced the locator.
|
|
503
|
+
*
|
|
504
|
+
* The locator strings are unchanged from Slice 1, including the last branch's
|
|
505
|
+
* `text-span:0-<excerpt.length>` placeholder. That placeholder is deliberate
|
|
506
|
+
* (producer discipline requires a locator on a non-manual-entry source), but
|
|
507
|
+
* it is well-formed and therefore resolvable — it will happily span real,
|
|
508
|
+
* unrelated prose at the head of the utterance. Returning the resolution
|
|
509
|
+
* alongside it is what keeps a hallucinated excerpt from acquiring a pointer
|
|
510
|
+
* that looks exactly like a found one: the record now says the lookup failed
|
|
511
|
+
* instead of leaving the reader to re-derive it.
|
|
422
512
|
*/
|
|
423
|
-
function
|
|
424
|
-
|
|
513
|
+
function resolveUtteranceLocator(utterance, statement) {
|
|
514
|
+
if (statement.span) {
|
|
515
|
+
return { locator: `text-span:${statement.span.start}-${statement.span.end}`, resolution: "span" };
|
|
516
|
+
}
|
|
517
|
+
const idx = utterance.indexOf(statement.excerpt);
|
|
425
518
|
if (idx >= 0) {
|
|
426
|
-
return `text-span:${idx}-${idx + excerpt.length}
|
|
519
|
+
return { locator: `text-span:${idx}-${idx + statement.excerpt.length}`, resolution: "excerpt-match" };
|
|
427
520
|
}
|
|
428
|
-
|
|
429
|
-
// best-effort for span-less extractors)
|
|
430
|
-
return `text-span:0-${excerpt.length}`;
|
|
521
|
+
return { locator: `text-span:0-${statement.excerpt.length}`, resolution: "unanchored-fallback" };
|
|
431
522
|
}
|
|
432
523
|
// ---------------------------------------------------------------------------
|
|
433
524
|
// Reference extractor (deterministic, for tests — not for production use)
|