@kontourai/survey 1.9.0 → 1.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/src/to-surface.d.ts +28 -0
- package/dist/src/to-surface.js +65 -10
- package/package.json +1 -1
package/dist/src/to-surface.d.ts
CHANGED
|
@@ -1,6 +1,34 @@
|
|
|
1
1
|
import type { TrustBundle } from "@kontourai/surface";
|
|
2
|
+
import { type CalibrationMetrics } from "./calibration.js";
|
|
2
3
|
import type { SurveyInput } from "./types.js";
|
|
4
|
+
export interface SurveyCalibrationOptions {
|
|
5
|
+
/**
|
|
6
|
+
* Precomputed calibration to source the value from — typically derived over a
|
|
7
|
+
* LONGER history than the current batch (a better-grounded curve, and it avoids
|
|
8
|
+
* the mild self-reference of a claim's own review outcome feeding its value).
|
|
9
|
+
* When omitted, calibration is derived from THIS batch's review outcomes.
|
|
10
|
+
*/
|
|
11
|
+
metrics?: CalibrationMetrics;
|
|
12
|
+
/**
|
|
13
|
+
* Minimum labeled samples a group needs before its accuracy is emitted as a
|
|
14
|
+
* value. Groups below the floor leave `value` unset rather than emitting a
|
|
15
|
+
* poorly-grounded number. Default {@link DEFAULT_CALIBRATION_MIN_SAMPLES}.
|
|
16
|
+
*/
|
|
17
|
+
minSamples?: number;
|
|
18
|
+
}
|
|
3
19
|
export interface BuildSurveyTrustBundleOptions {
|
|
4
20
|
reviewProofs?: boolean;
|
|
21
|
+
/**
|
|
22
|
+
* Populate `conclusionConfidence.value` from empirical review calibration —
|
|
23
|
+
* "how often this extractor's proposals at this confidence were affirmed by a
|
|
24
|
+
* human reviewer" (the produce side of the confidence loop; see #114/#137).
|
|
25
|
+
* `true` derives calibration from this batch; an object supplies precomputed
|
|
26
|
+
* metrics and/or a `minSamples` floor. Absent → `value` stays unset and only
|
|
27
|
+
* the comfort-zone signal is carried (unchanged behavior).
|
|
28
|
+
*
|
|
29
|
+
* ADVISORY (ADR 0003 §4): this only enriches the emitted conclusion confidence;
|
|
30
|
+
* it never changes a claim's `status`.
|
|
31
|
+
*/
|
|
32
|
+
calibration?: boolean | SurveyCalibrationOptions;
|
|
5
33
|
}
|
|
6
34
|
export declare function buildSurveyTrustBundle(input: SurveyInput, options?: BuildSurveyTrustBundleOptions): TrustBundle;
|
package/dist/src/to-surface.js
CHANGED
|
@@ -1,10 +1,23 @@
|
|
|
1
1
|
import { buildReviewProofAnchor } from "./review-proof.js";
|
|
2
2
|
import { assertReviewOutcomeDiscipline } from "./producer-discipline.js";
|
|
3
|
+
import { deriveCalibration } from "./calibration.js";
|
|
4
|
+
/** Minimum labeled samples a calibration group needs before its empirical
|
|
5
|
+
* accuracy is emitted as a `conclusionConfidence.value`. */
|
|
6
|
+
const DEFAULT_CALIBRATION_MIN_SAMPLES = 20;
|
|
3
7
|
export function buildSurveyTrustBundle(input, options = {}) {
|
|
4
8
|
const rawSources = indexById(input.rawSources, "raw source");
|
|
5
9
|
const extractions = indexById(input.extractions, "extraction");
|
|
6
10
|
const candidateSets = indexById(input.candidateSets, "candidate set");
|
|
7
11
|
const reviewsByCandidateSet = groupBy(input.reviewOutcomes, (review) => review.candidateSetId);
|
|
12
|
+
const calibrationOptions = normalizeCalibrationOptions(options.calibration);
|
|
13
|
+
const calibrationMetrics = calibrationOptions
|
|
14
|
+
? (calibrationOptions.metrics ?? deriveCalibration({
|
|
15
|
+
reviewOutcomes: input.reviewOutcomes,
|
|
16
|
+
candidateSets: input.candidateSets,
|
|
17
|
+
extractions: input.extractions,
|
|
18
|
+
}))
|
|
19
|
+
: undefined;
|
|
20
|
+
const calibrationMinSamples = calibrationOptions?.minSamples ?? DEFAULT_CALIBRATION_MIN_SAMPLES;
|
|
8
21
|
const claims = [];
|
|
9
22
|
const evidence = [];
|
|
10
23
|
const events = [];
|
|
@@ -47,18 +60,35 @@ export function buildSurveyTrustBundle(input, options = {}) {
|
|
|
47
60
|
survey: buildSurveyMetadata({ projection, rawSource, extraction, candidateSet, candidate, review }),
|
|
48
61
|
},
|
|
49
62
|
};
|
|
50
|
-
// Promote the review's comfort-zone signal
|
|
51
|
-
//
|
|
63
|
+
// Promote the review's comfort-zone signal — and, when calibration is
|
|
64
|
+
// enabled, an empirically-calibrated conclusion probability — into the
|
|
65
|
+
// first-class conclusionConfidence field (Surface 2.9 / Hachure 0.14) so the
|
|
52
66
|
// signal is portable and comparable, not buried in producer metadata.
|
|
53
|
-
//
|
|
54
|
-
//
|
|
55
|
-
//
|
|
56
|
-
|
|
67
|
+
//
|
|
68
|
+
// comfortZone is CARRIED from the review. `value` is PRODUCED from empirical
|
|
69
|
+
// review calibration (#114/#137): the affirmation rate of this extractor's
|
|
70
|
+
// proposals at this confidence — a calibrated conclusion probability, distinct
|
|
71
|
+
// from the extraction-confidence ingredient in confidenceBasis.
|
|
72
|
+
//
|
|
73
|
+
// A value is produced only for an AFFIRMED conclusion (status verified/assumed)
|
|
74
|
+
// that clears the sample floor. conclusionConfidence.value is "probability the
|
|
75
|
+
// conclusion is correct"; attaching an affirmation rate to a REJECTED (or
|
|
76
|
+
// not-yet-reviewed) conclusion would assert the opposite of what the human
|
|
77
|
+
// decided, so those claims get no value.
|
|
78
|
+
const comfortZone = review?.withinComfortZone !== undefined
|
|
79
|
+
? {
|
|
80
|
+
within: review.withinComfortZone,
|
|
81
|
+
...(review.comfortZoneNote ? { reason: review.comfortZoneNote } : {}),
|
|
82
|
+
}
|
|
83
|
+
: undefined;
|
|
84
|
+
const affirmedConclusion = status === "verified" || status === "assumed";
|
|
85
|
+
const calibrated = calibrationMetrics && review && affirmedConclusion
|
|
86
|
+
? lookupCalibratedValue(calibrationMetrics, extraction.extractor, extraction.target, calibrationMinSamples)
|
|
87
|
+
: undefined;
|
|
88
|
+
if (comfortZone || calibrated) {
|
|
57
89
|
claim.conclusionConfidence = {
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
...(review.comfortZoneNote ? { reason: review.comfortZoneNote } : {}),
|
|
61
|
-
},
|
|
90
|
+
...(calibrated ? { value: calibrated.value, method: calibrated.method } : {}),
|
|
91
|
+
...(comfortZone ? { comfortZone } : {}),
|
|
62
92
|
};
|
|
63
93
|
}
|
|
64
94
|
if (options.reviewProofs && review) {
|
|
@@ -370,6 +400,31 @@ function selectCandidate(candidateSet, candidateId) {
|
|
|
370
400
|
function selectReview(reviews, candidateId) {
|
|
371
401
|
return reviews.find((review) => review.candidateId === candidateId) ?? reviews.find((review) => !review.candidateId);
|
|
372
402
|
}
|
|
403
|
+
function normalizeCalibrationOptions(calibration) {
|
|
404
|
+
if (calibration === undefined || calibration === false)
|
|
405
|
+
return undefined;
|
|
406
|
+
if (calibration === true)
|
|
407
|
+
return {};
|
|
408
|
+
return calibration;
|
|
409
|
+
}
|
|
410
|
+
/**
|
|
411
|
+
* Looks up the empirical affirmation rate for an extractor/field, preferring the
|
|
412
|
+
* finer (extractor, field) group and falling back to the extractor-level group
|
|
413
|
+
* when the field group is below the sample floor. Returns undefined when neither
|
|
414
|
+
* group clears the floor, so an ungrounded claim leaves `value` unset. The
|
|
415
|
+
* `method` records which granularity produced the value.
|
|
416
|
+
*/
|
|
417
|
+
function lookupCalibratedValue(metrics, extractor, field, minSamples) {
|
|
418
|
+
const fieldGroup = metrics.byExtractorField.find((g) => g.extractor === extractor && g.field === field);
|
|
419
|
+
if (fieldGroup && fieldGroup.sampleCount >= minSamples && fieldGroup.empiricalAccuracy !== undefined) {
|
|
420
|
+
return { value: fieldGroup.empiricalAccuracy, method: "empirical-review-calibration:extractor-field" };
|
|
421
|
+
}
|
|
422
|
+
const extractorGroup = metrics.byExtractor.find((g) => g.extractor === extractor);
|
|
423
|
+
if (extractorGroup && extractorGroup.sampleCount >= minSamples && extractorGroup.empiricalAccuracy !== undefined) {
|
|
424
|
+
return { value: extractorGroup.empiricalAccuracy, method: "empirical-review-calibration:extractor" };
|
|
425
|
+
}
|
|
426
|
+
return undefined;
|
|
427
|
+
}
|
|
373
428
|
function evidenceTypeFor(rawSource) {
|
|
374
429
|
if (rawSource.kind === "policy-standard")
|
|
375
430
|
return "policy_rule";
|
package/package.json
CHANGED