@kontourai/survey 0.5.2 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,146 @@
1
+ /**
2
+ * Oversight-quality metrics for EU AI Act Art. 14 "effective oversight" / automation-bias evidence.
3
+ *
4
+ * These metrics are INDICATORS, not proof of reviewer cognition. See the Honest Limits section
5
+ * in docs/record-contracts.md. Pace statistics can be gamed; metrics complement (not replace)
6
+ * identity signing and authorizing provenance.
7
+ *
8
+ * All computations are deterministic with injected `now`.
9
+ */
10
+ import type { Claim, Evidence, TrustBundle, VerificationEvent } from "@kontourai/surface";
11
+ import type { ReviewDecision } from "./review-resource.js";
12
+ export interface ReviewerOversightMetrics {
13
+ /** Actor id this row belongs to. */
14
+ readonly actorId: string;
15
+ /** Total decisions made by this reviewer in the window. */
16
+ readonly decisionCount: number;
17
+ /**
18
+ * Average decisions per hour, computed as decisionCount / elapsed hours between
19
+ * the first and last reviewedAt timestamp (or 0 when decisionCount < 2).
20
+ */
21
+ readonly decisionsPerHour: number;
22
+ /**
23
+ * Fraction of decisions where the reviewer chose a candidate that differs from
24
+ * the item's pre-selected (proposed) candidate — i.e. the reviewer changed the
25
+ * outcome suggested by the system. Ranges 0–1.
26
+ */
27
+ readonly overrideRate: number;
28
+ /**
29
+ * Fraction of decisions where the authorizing block carries action === "typed",
30
+ * indicating the reviewer also wrote a rationale note. Ranges 0–1.
31
+ */
32
+ readonly typedRationaleRate: number;
33
+ /**
34
+ * Median inter-decision gap in seconds. Undefined when fewer than 2 decisions.
35
+ */
36
+ readonly medianInterDecisionSeconds: number | undefined;
37
+ /**
38
+ * Fraction of presented items that were decided, when the caller supplies
39
+ * presentedCount. Omitted (undefined) otherwise.
40
+ */
41
+ readonly samplingCoverage: number | undefined;
42
+ }
43
+ export interface AggregateOversightMetrics {
44
+ /** Total decisions across all reviewers. */
45
+ readonly decisionCount: number;
46
+ /**
47
+ * Average decisions per hour across the full window (first decision to last
48
+ * decision, all reviewers combined).
49
+ */
50
+ readonly decisionsPerHour: number;
51
+ /** Override rate across all decisions. */
52
+ readonly overrideRate: number;
53
+ /** Typed-rationale rate across all decisions. */
54
+ readonly typedRationaleRate: number;
55
+ /** Median inter-decision gap in seconds across all decisions. */
56
+ readonly medianInterDecisionSeconds: number | undefined;
57
+ /**
58
+ * Fraction of presented items decided, when presentedCount was supplied.
59
+ * Omitted otherwise.
60
+ */
61
+ readonly samplingCoverage: number | undefined;
62
+ }
63
+ export interface OversightMetrics {
64
+ /** One row per unique actorId found in the decisions. */
65
+ readonly byReviewer: readonly ReviewerOversightMetrics[];
66
+ /** Aggregate across all reviewers. */
67
+ readonly aggregate: AggregateOversightMetrics;
68
+ /** ISO 8601 timestamp of the earliest decision in the window. */
69
+ readonly windowStart: string | undefined;
70
+ /** ISO 8601 timestamp of the latest decision in the window. */
71
+ readonly windowEnd: string | undefined;
72
+ /** Number of calendar days covered by the window (may be fractional). */
73
+ readonly windowDays: number | undefined;
74
+ /** Number of input decisions used for these metrics. */
75
+ readonly inputDecisionCount: number;
76
+ }
77
+ export interface DeriveOversightMetricsOptions {
78
+ /** Current time — injected for determinism. */
79
+ readonly now: Date;
80
+ /**
81
+ * Optional rolling window in days. Decisions older than `now - windowDays`
82
+ * are excluded. When omitted all supplied decisions are included.
83
+ */
84
+ readonly windowDays?: number;
85
+ /**
86
+ * Optional total number of items that were PRESENTED to reviewers (the
87
+ * denominator for samplingCoverage). When omitted, samplingCoverage is
88
+ * not computed.
89
+ */
90
+ readonly presentedCount?: number;
91
+ }
92
+ /**
93
+ * Derives per-reviewer and aggregate oversight-quality metrics from a stream
94
+ * of ReviewDecision resources.
95
+ *
96
+ * @param decisions - Array of ReviewDecision Kubernetes-style resources as
97
+ * produced by the review workbench or adapters.
98
+ * @param options - `now` is required for determinism; `windowDays` and
99
+ * `presentedCount` are optional.
100
+ */
101
+ export declare function deriveOversightMetrics(decisions: readonly ReviewDecision[], options: DeriveOversightMetricsOptions): OversightMetrics;
102
+ /**
103
+ * A single projected oversight-quality claim from the metrics layer.
104
+ * Uses claimType "oversight-quality" per the task specification.
105
+ */
106
+ export interface OversightQualityClaim {
107
+ readonly claim: Claim;
108
+ readonly evidence: Evidence;
109
+ readonly event: VerificationEvent;
110
+ }
111
+ export interface OversightMetricsClaimsSubject {
112
+ /** Surface subject type (e.g. "review-session", "reviewer-actor"). */
113
+ readonly subjectType: string;
114
+ /** Surface subject id (e.g. a session name or actor id). */
115
+ readonly subjectId: string;
116
+ /** Surface name (e.g. "review.oversight"). */
117
+ readonly surface: string;
118
+ /** Actor id to record on events. */
119
+ readonly actor: string;
120
+ /** ISO 8601 timestamp for claim created/updated times. */
121
+ readonly observedAt: string;
122
+ /** Producer identifier for collectedBy field. */
123
+ readonly collectedBy: string;
124
+ }
125
+ /**
126
+ * Projects oversight metrics as Surface-ready claims.
127
+ *
128
+ * Produces one claim per measurable metric in `aggregate` (per task spec: claimType
129
+ * "oversight-quality", fieldOrBehavior per metric, numeric value). Evidence excerpts
130
+ * summarise the computation inputs (count, window) so Annex-pack rules can apply value
131
+ * predicates.
132
+ *
133
+ * Callers can pass these claims into buildSurveyTrustBundle by constructing a
134
+ * minimal SurveyInput with manual-entry raw sources, or merge the returned Claim /
135
+ * Evidence / VerificationEvent objects directly into an existing TrustBundle.
136
+ *
137
+ * @param metrics - Output of `deriveOversightMetrics`.
138
+ * @param subject - Surface identity for the claims.
139
+ */
140
+ export declare function oversightMetricsToClaims(metrics: OversightMetrics, subject: OversightMetricsClaimsSubject): OversightQualityClaim[];
141
+ /**
142
+ * Merges oversight-quality claims into an existing TrustBundle.
143
+ * Convenience wrapper for callers who already have a bundle and want to append
144
+ * oversight metrics without rebuilding from SurveyInput.
145
+ */
146
+ export declare function mergeTrustBundleWithOversightMetrics(bundle: TrustBundle, claims: readonly OversightQualityClaim[]): TrustBundle;
@@ -0,0 +1,296 @@
1
+ /**
2
+ * Oversight-quality metrics for EU AI Act Art. 14 "effective oversight" / automation-bias evidence.
3
+ *
4
+ * These metrics are INDICATORS, not proof of reviewer cognition. See the Honest Limits section
5
+ * in docs/record-contracts.md. Pace statistics can be gamed; metrics complement (not replace)
6
+ * identity signing and authorizing provenance.
7
+ *
8
+ * All computations are deterministic with injected `now`.
9
+ */
10
+ // ---------------------------------------------------------------------------
11
+ // deriveOversightMetrics
12
+ // ---------------------------------------------------------------------------
13
+ /**
14
+ * Derives per-reviewer and aggregate oversight-quality metrics from a stream
15
+ * of ReviewDecision resources.
16
+ *
17
+ * @param decisions - Array of ReviewDecision Kubernetes-style resources as
18
+ * produced by the review workbench or adapters.
19
+ * @param options - `now` is required for determinism; `windowDays` and
20
+ * `presentedCount` are optional.
21
+ */
22
+ export function deriveOversightMetrics(decisions, options) {
23
+ const cutoff = options.windowDays !== undefined
24
+ ? new Date(options.now.getTime() - options.windowDays * 24 * 60 * 60 * 1000)
25
+ : undefined;
26
+ const windowed = cutoff
27
+ ? decisions.filter((d) => {
28
+ const t = d.spec.reviewedAt ? Date.parse(d.spec.reviewedAt) : NaN;
29
+ return !isNaN(t) && t >= cutoff.getTime();
30
+ })
31
+ : [...decisions];
32
+ const inputDecisionCount = windowed.length;
33
+ if (inputDecisionCount === 0) {
34
+ return {
35
+ byReviewer: [],
36
+ aggregate: {
37
+ decisionCount: 0,
38
+ decisionsPerHour: 0,
39
+ overrideRate: 0,
40
+ typedRationaleRate: 0,
41
+ medianInterDecisionSeconds: undefined,
42
+ samplingCoverage: options.presentedCount !== undefined ? 0 : undefined,
43
+ },
44
+ windowStart: undefined,
45
+ windowEnd: undefined,
46
+ windowDays: options.windowDays,
47
+ inputDecisionCount: 0,
48
+ };
49
+ }
50
+ // Group by actor
51
+ const byActor = new Map();
52
+ for (const d of windowed) {
53
+ const actor = d.spec.actor?.id ?? "unknown";
54
+ const existing = byActor.get(actor) ?? [];
55
+ byActor.set(actor, [...existing, d]);
56
+ }
57
+ // Compute timestamps for window bounds
58
+ const allTimestamps = windowed
59
+ .map((d) => d.spec.reviewedAt ? Date.parse(d.spec.reviewedAt) : NaN)
60
+ .filter((t) => !isNaN(t))
61
+ .sort((a, b) => a - b);
62
+ const windowStart = allTimestamps.length > 0 ? new Date(allTimestamps[0]).toISOString() : undefined;
63
+ const windowEnd = allTimestamps.length > 0 ? new Date(allTimestamps[allTimestamps.length - 1]).toISOString() : undefined;
64
+ // Per-reviewer rows
65
+ const byReviewer = [];
66
+ for (const [actorId, actorDecisions] of byActor) {
67
+ byReviewer.push(computeReviewerMetrics(actorId, actorDecisions, options.presentedCount));
68
+ }
69
+ // Aggregate
70
+ const aggregate = computeAggregateMetrics(windowed, allTimestamps, options.presentedCount);
71
+ return {
72
+ byReviewer,
73
+ aggregate,
74
+ windowStart,
75
+ windowEnd,
76
+ windowDays: options.windowDays,
77
+ inputDecisionCount,
78
+ };
79
+ }
80
+ // ---------------------------------------------------------------------------
81
+ // Internal helpers
82
+ // ---------------------------------------------------------------------------
83
+ function computeReviewerMetrics(actorId, decisions, presentedCount) {
84
+ const decisionCount = decisions.length;
85
+ const timestamps = decisions
86
+ .map((d) => d.spec.reviewedAt ? Date.parse(d.spec.reviewedAt) : NaN)
87
+ .filter((t) => !isNaN(t))
88
+ .sort((a, b) => a - b);
89
+ const decisionsPerHour = computeDecisionsPerHour(decisionCount, timestamps);
90
+ const overrideCount = decisions.filter(isOverrideDecision).length;
91
+ const overrideRate = decisionCount > 0 ? overrideCount / decisionCount : 0;
92
+ const typedCount = decisions.filter(isTypedRationale).length;
93
+ const typedRationaleRate = decisionCount > 0 ? typedCount / decisionCount : 0;
94
+ const medianInterDecisionSeconds = computeMedianInterDecisionSeconds(timestamps);
95
+ const samplingCoverage = presentedCount !== undefined
96
+ ? (presentedCount > 0 ? decisionCount / presentedCount : 0)
97
+ : undefined;
98
+ return {
99
+ actorId,
100
+ decisionCount,
101
+ decisionsPerHour,
102
+ overrideRate,
103
+ typedRationaleRate,
104
+ medianInterDecisionSeconds,
105
+ samplingCoverage,
106
+ };
107
+ }
108
+ function computeAggregateMetrics(decisions, sortedTimestamps, presentedCount) {
109
+ const decisionCount = decisions.length;
110
+ const decisionsPerHour = computeDecisionsPerHour(decisionCount, sortedTimestamps);
111
+ const overrideCount = decisions.filter(isOverrideDecision).length;
112
+ const overrideRate = decisionCount > 0 ? overrideCount / decisionCount : 0;
113
+ const typedCount = decisions.filter(isTypedRationale).length;
114
+ const typedRationaleRate = decisionCount > 0 ? typedCount / decisionCount : 0;
115
+ const medianInterDecisionSeconds = computeMedianInterDecisionSeconds(sortedTimestamps);
116
+ const samplingCoverage = presentedCount !== undefined
117
+ ? (presentedCount > 0 ? decisionCount / presentedCount : 0)
118
+ : undefined;
119
+ return {
120
+ decisionCount,
121
+ decisionsPerHour,
122
+ overrideRate,
123
+ typedRationaleRate,
124
+ medianInterDecisionSeconds,
125
+ samplingCoverage,
126
+ };
127
+ }
128
+ /**
129
+ * A decision is an "override" when the reviewer chose a candidate whose id
130
+ * differs from the item's pre-selected (proposed) candidate.
131
+ *
132
+ * The item's proposed candidate id is carried in
133
+ * `decision.spec.projection.candidateId` when set; if absent we fall back to
134
+ * checking whether the decision status is "rejected" (reviewer rejected the
135
+ * proposal entirely). When neither signal is available the decision is treated
136
+ * as non-override to avoid false positives.
137
+ */
138
+ function isOverrideDecision(decision) {
139
+ // "rejected" status always means the reviewer disagreed with the proposed value
140
+ if (decision.spec.status === "rejected") {
141
+ return true;
142
+ }
143
+ // If the decision carries a projection hint we can compare candidate ids.
144
+ // The proposed/pre-selected candidate is carried as the projection candidateId
145
+ // in ReviewDecision.spec.projection from the workbench accept-proposed path.
146
+ // When the reviewer keeps-current, the projection candidateId is the current
147
+ // candidate id, which differs from the proposed candidate — that counts as override.
148
+ // We detect this by checking whether the candidateId on the decision spec
149
+ // matches the projection candidateId and neither is undefined.
150
+ const decisionCandidateId = decision.spec.candidateId;
151
+ const projectionCandidateId = decision.spec.projection?.candidateId;
152
+ if (decisionCandidateId && projectionCandidateId && decisionCandidateId !== projectionCandidateId) {
153
+ return true;
154
+ }
155
+ return false;
156
+ }
157
+ /**
158
+ * A decision carries a "typed" rationale when the authorizing block is an
159
+ * `authorized-action` block with `action === "typed"`, indicating the reviewer
160
+ * explicitly wrote a note.
161
+ */
162
+ function isTypedRationale(decision) {
163
+ const auth = decision.spec.authorizing;
164
+ return auth?.kind === "authorized-action" && auth.action === "typed";
165
+ }
166
+ function computeDecisionsPerHour(decisionCount, sortedTimestamps) {
167
+ if (decisionCount < 2 || sortedTimestamps.length < 2) {
168
+ return 0;
169
+ }
170
+ const first = sortedTimestamps[0];
171
+ const last = sortedTimestamps[sortedTimestamps.length - 1];
172
+ const elapsedHours = (last - first) / (1000 * 60 * 60);
173
+ if (elapsedHours <= 0) {
174
+ return 0;
175
+ }
176
+ return decisionCount / elapsedHours;
177
+ }
178
+ function computeMedianInterDecisionSeconds(sortedTimestamps) {
179
+ if (sortedTimestamps.length < 2) {
180
+ return undefined;
181
+ }
182
+ const gaps = [];
183
+ for (let i = 1; i < sortedTimestamps.length; i++) {
184
+ gaps.push((sortedTimestamps[i] - sortedTimestamps[i - 1]) / 1000);
185
+ }
186
+ gaps.sort((a, b) => a - b);
187
+ const mid = Math.floor(gaps.length / 2);
188
+ return gaps.length % 2 === 0
189
+ ? ((gaps[mid - 1] + gaps[mid]) / 2)
190
+ : gaps[mid];
191
+ }
192
+ /**
193
+ * Projects oversight metrics as Surface-ready claims.
194
+ *
195
+ * Produces one claim per measurable metric in `aggregate` (per task spec: claimType
196
+ * "oversight-quality", fieldOrBehavior per metric, numeric value). Evidence excerpts
197
+ * summarise the computation inputs (count, window) so Annex-pack rules can apply value
198
+ * predicates.
199
+ *
200
+ * Callers can pass these claims into buildSurveyTrustBundle by constructing a
201
+ * minimal SurveyInput with manual-entry raw sources, or merge the returned Claim /
202
+ * Evidence / VerificationEvent objects directly into an existing TrustBundle.
203
+ *
204
+ * @param metrics - Output of `deriveOversightMetrics`.
205
+ * @param subject - Surface identity for the claims.
206
+ */
207
+ export function oversightMetricsToClaims(metrics, subject) {
208
+ const results = [];
209
+ const windowDesc = metrics.windowStart && metrics.windowEnd
210
+ ? `window ${metrics.windowStart} to ${metrics.windowEnd}`
211
+ : "all available decisions";
212
+ const inputSummary = `${metrics.inputDecisionCount} decisions, ${windowDesc}`;
213
+ const push = (fieldOrBehavior, value, excerptDetail) => {
214
+ const claimId = `oversight-quality.${subject.subjectId}.${fieldOrBehavior}`;
215
+ const evidenceId = `${claimId}.evidence`;
216
+ const claim = {
217
+ id: claimId,
218
+ subjectType: subject.subjectType,
219
+ subjectId: subject.subjectId,
220
+ surface: subject.surface,
221
+ claimType: "oversight-quality",
222
+ fieldOrBehavior,
223
+ value,
224
+ status: "proposed",
225
+ createdAt: subject.observedAt,
226
+ updatedAt: subject.observedAt,
227
+ impactLevel: "medium",
228
+ confidenceBasis: {
229
+ sourceQuality: "moderate",
230
+ reviewerAuthority: "none",
231
+ evidenceStrength: "weak",
232
+ impactLevel: "medium",
233
+ },
234
+ metadata: {
235
+ oversightMetrics: {
236
+ inputDecisionCount: metrics.inputDecisionCount,
237
+ windowStart: metrics.windowStart,
238
+ windowEnd: metrics.windowEnd,
239
+ windowDays: metrics.windowDays,
240
+ },
241
+ },
242
+ };
243
+ const evidence = {
244
+ id: evidenceId,
245
+ claimId,
246
+ evidenceType: "attestation",
247
+ method: "extraction",
248
+ sourceRef: "oversight-metrics://computed",
249
+ excerptOrSummary: `oversight-quality.${fieldOrBehavior}: ${excerptDetail}; computed from ${inputSummary}`,
250
+ observedAt: subject.observedAt,
251
+ collectedBy: subject.collectedBy,
252
+ };
253
+ const event = {
254
+ id: `${claimId}.event`,
255
+ claimId,
256
+ status: "proposed",
257
+ actor: subject.actor,
258
+ method: "candidate-proposal",
259
+ evidenceIds: [evidenceId],
260
+ createdAt: subject.observedAt,
261
+ };
262
+ results.push({ claim, evidence, event });
263
+ };
264
+ const agg = metrics.aggregate;
265
+ push("decisionCount", agg.decisionCount, `total decisions = ${agg.decisionCount}`);
266
+ push("decisionsPerHour", roundTo(agg.decisionsPerHour, 4), `decisions/hr = ${roundTo(agg.decisionsPerHour, 4)}`);
267
+ push("overrideRate", roundTo(agg.overrideRate, 4), `override rate = ${roundTo(agg.overrideRate, 4)} (${countFromRate(agg.overrideRate, agg.decisionCount)} of ${agg.decisionCount} decisions differed from proposed)`);
268
+ push("typedRationaleRate", roundTo(agg.typedRationaleRate, 4), `typed rationale rate = ${roundTo(agg.typedRationaleRate, 4)} (${countFromRate(agg.typedRationaleRate, agg.decisionCount)} of ${agg.decisionCount} decisions had typed note)`);
269
+ if (agg.medianInterDecisionSeconds !== undefined) {
270
+ push("medianInterDecisionSeconds", roundTo(agg.medianInterDecisionSeconds, 2), `median gap between decisions = ${roundTo(agg.medianInterDecisionSeconds, 2)} s`);
271
+ }
272
+ if (agg.samplingCoverage !== undefined) {
273
+ push("samplingCoverage", roundTo(agg.samplingCoverage, 4), `sampling coverage = ${roundTo(agg.samplingCoverage, 4)} (decided / presented)`);
274
+ }
275
+ return results;
276
+ }
277
+ /**
278
+ * Merges oversight-quality claims into an existing TrustBundle.
279
+ * Convenience wrapper for callers who already have a bundle and want to append
280
+ * oversight metrics without rebuilding from SurveyInput.
281
+ */
282
+ export function mergeTrustBundleWithOversightMetrics(bundle, claims) {
283
+ return {
284
+ ...bundle,
285
+ claims: [...bundle.claims, ...claims.map((c) => c.claim)],
286
+ evidence: [...bundle.evidence, ...claims.map((c) => c.evidence)],
287
+ events: [...bundle.events, ...claims.map((c) => c.event)],
288
+ };
289
+ }
290
+ function roundTo(value, decimals) {
291
+ const factor = Math.pow(10, decimals);
292
+ return Math.round(value * factor) / factor;
293
+ }
294
+ function countFromRate(rate, total) {
295
+ return Math.round(rate * total);
296
+ }
@@ -0,0 +1,2 @@
1
+ export declare const REVIEW_WORKBENCH_CSS: string;
2
+ export default REVIEW_WORKBENCH_CSS;