@ngockhoale/ukit 3.0.5 → 3.0.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/package.json +1 -1
- package/scripts/bench/data-foundation.mjs +562 -0
- package/src/core/observability/adapters/common.js +75 -0
- package/src/core/observability/adapters/contextAdapter.js +55 -0
- package/src/core/observability/adapters/decisionAdapter.js +61 -0
- package/src/core/observability/adapters/routeAdapter.js +135 -0
- package/src/core/observability/analytics/digest.js +186 -0
- package/src/core/observability/analytics/fingerprints.js +126 -0
- package/src/core/observability/analytics/opportunities.js +329 -0
- package/src/core/observability/analytics/rebuild.js +56 -0
- package/src/core/observability/analytics/summary.js +298 -0
- package/src/core/observability/emit/config.js +29 -0
- package/src/core/observability/emit/recorder.js +297 -0
- package/src/core/observability/evaluation/aiPacket.js +230 -0
- package/src/core/observability/evaluation/optimizationKnowledge.js +172 -0
- package/src/core/observability/evaluation/replay.js +143 -0
- package/src/core/observability/evaluation/scorecard.js +445 -0
- package/src/core/observability/privacy/allowlist.js +185 -0
- package/src/core/observability/privacy/redaction.js +113 -0
- package/src/core/observability/privacy/sanitizeForSupport.js +133 -0
- package/src/core/observability/privacy/sanitizeObserved.js +134 -0
- package/src/core/observability/rollout.js +155 -0
- package/src/core/observability/schema/constants.js +66 -0
- package/src/core/observability/schema/registry.js +223 -0
- package/src/core/observability/schema/validate.js +227 -0
- package/src/core/observability/segments/internal.js +241 -0
- package/src/core/observability/segments/readSegments.js +215 -0
- package/src/core/observability/segments/recovery.js +123 -0
- package/src/core/observability/segments/retention.js +381 -0
- package/src/core/observability/support/import.js +402 -0
- package/src/core/observability/support/manifest.js +135 -0
- package/src/core/observability/support/paths.js +94 -0
- package/src/core/observability/support/projector.js +483 -0
- package/src/core/observability/support/renderer.js +130 -0
- package/src/core/observability/support/retention.js +155 -0
- package/src/core/runtimeConfig.js +6 -3
- package/src/decision/client.js +11 -3
- package/template_project/.claude/ukit/index/unic-decision.mjs +6 -2
- package/template_project/.omp/RULES.md +6 -6
- package/template_project/.omp/config.yml +6 -0
- package/template_project/docs/UKIT_INTERNALS.md +9 -0
- package/template_project/instructions/overlays/omp-rules.md +6 -6
|
@@ -0,0 +1,445 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* scorecard.js (TASK-014, SPEC §5 DF-FR08/DF-FR09/DF-FR12, §8) — golden
|
|
3
|
+
* corpus evaluation.
|
|
4
|
+
*
|
|
5
|
+
* evaluateVariant(corpus, policy): Scorecard
|
|
6
|
+
* compareScorecards(a, b, { seed }): Comparison
|
|
7
|
+
*
|
|
8
|
+
* Quality is the guardrail: golden cases declare evidence / actions /
|
|
9
|
+
* forbidden mistakes / verification / acceptable bounds — never exact-prose
|
|
10
|
+
* equality. Correction semantics are honest: when a user correction follows
|
|
11
|
+
* the first answer, that first answer is NOT an accepted success and
|
|
12
|
+
* tokens_to_success / time_to_success_ms include the observed retry.
|
|
13
|
+
*
|
|
14
|
+
* Scorecard (SPEC §8):
|
|
15
|
+
* { metric_version, policy, quality: { met, unmet, violated, unknown,
|
|
16
|
+
* forbidden, bounds }, latency, tokens, cohorts, provenance, fidelity,
|
|
17
|
+
* sample_size, verdict: 'pass'|'fail'|'inconclusive', cases }
|
|
18
|
+
*
|
|
19
|
+
* Everything here is offline and deterministic — no model calls, no IO, no
|
|
20
|
+
* clock, no randomness outside the seeded bootstrap in compareScorecards.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
import { summarizeTrace, percentileOf } from '../analytics/summary.js';
|
|
24
|
+
import { replayCase } from './replay.js';
|
|
25
|
+
|
|
26
|
+
export const EVALUATION_METRIC_VERSION = 'df-m1';
|
|
27
|
+
export const DEFAULT_MIN_COHORT_N = 3;
|
|
28
|
+
const BOOTSTRAP_RESAMPLES = 2000;
|
|
29
|
+
|
|
30
|
+
function isPlainObject(value) {
|
|
31
|
+
return value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** Deterministic PRNG (mulberry32) — same construction as the bench corpus. */
|
|
35
|
+
function mulberry32(seed) {
|
|
36
|
+
let a = seed >>> 0;
|
|
37
|
+
return () => {
|
|
38
|
+
a |= 0;
|
|
39
|
+
a = (a + 0x6d2b79f5) | 0;
|
|
40
|
+
let t = Math.imul(a ^ (a >>> 15), 1 | a);
|
|
41
|
+
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
|
|
42
|
+
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
|
43
|
+
};
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
function resolvePolicy(corpus, policy) {
|
|
47
|
+
if (typeof policy === 'string') {
|
|
48
|
+
const registered = isPlainObject(corpus.pre_registered)
|
|
49
|
+
&& isPlainObject(corpus.pre_registered.policies)
|
|
50
|
+
? corpus.pre_registered.policies[policy]
|
|
51
|
+
: null;
|
|
52
|
+
return { id: policy, ...(isPlainObject(registered) ? registered : {}) };
|
|
53
|
+
}
|
|
54
|
+
if (isPlainObject(policy)) {
|
|
55
|
+
return { id: typeof policy.id === 'string' ? policy.id : 'custom', ...policy };
|
|
56
|
+
}
|
|
57
|
+
return { id: 'custom' };
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
function semanticNames(records) {
|
|
61
|
+
const names = new Set();
|
|
62
|
+
for (const r of records) {
|
|
63
|
+
if (isPlainObject(r) && typeof r.semantic_name === 'string') names.add(r.semantic_name);
|
|
64
|
+
}
|
|
65
|
+
return names;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/** Human-reviewable action labels: operation/tool names on span records. */
|
|
69
|
+
function actionLabels(records) {
|
|
70
|
+
const actions = new Set();
|
|
71
|
+
for (const r of records) {
|
|
72
|
+
if (!isPlainObject(r) || !isPlainObject(r.payload)) continue;
|
|
73
|
+
if (typeof r.payload.operation === 'string') actions.add(r.payload.operation);
|
|
74
|
+
if (typeof r.payload.tool === 'string') actions.add(r.payload.tool);
|
|
75
|
+
}
|
|
76
|
+
return actions;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
function executionOutcome(records) {
|
|
80
|
+
for (const r of records) {
|
|
81
|
+
if (!isPlainObject(r) || !isPlainObject(r.payload)) continue;
|
|
82
|
+
if ((r.semantic_name === 'execution.completed' || r.semantic_name === 'execution.failed'
|
|
83
|
+
|| r.semantic_name === 'execution.blocked') && typeof r.payload.outcome === 'string') {
|
|
84
|
+
return r.payload.outcome;
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
return null;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
function correctionObserved(records) {
|
|
91
|
+
return records.some((r) => isPlainObject(r) && isPlainObject(r.payload)
|
|
92
|
+
&& (r.payload.correction_observed === true || r.payload.kind === 'user_correction'));
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
const BOUND_METRICS = Object.freeze(['tokens_to_success', 'time_to_success_ms']);
|
|
96
|
+
|
|
97
|
+
function checkBounds(bounds, metrics) {
|
|
98
|
+
const result = { checked: 0, violated: 0, unknown: 0, detail: {} };
|
|
99
|
+
if (!isPlainObject(bounds)) return result;
|
|
100
|
+
for (const metric of BOUND_METRICS) {
|
|
101
|
+
const bound = bounds[metric];
|
|
102
|
+
if (!isPlainObject(bound)) continue;
|
|
103
|
+
const value = metrics[metric];
|
|
104
|
+
if (typeof value !== 'number' || !Number.isFinite(value)) {
|
|
105
|
+
result.unknown += 1;
|
|
106
|
+
result.detail[metric] = 'unknown';
|
|
107
|
+
continue;
|
|
108
|
+
}
|
|
109
|
+
result.checked += 1;
|
|
110
|
+
const overMax = typeof bound.max === 'number' && value > bound.max;
|
|
111
|
+
const underMin = typeof bound.min === 'number' && value < bound.min;
|
|
112
|
+
if (overMax || underMin) {
|
|
113
|
+
result.violated += 1;
|
|
114
|
+
result.detail[metric] = 'violated';
|
|
115
|
+
} else {
|
|
116
|
+
result.detail[metric] = 'met';
|
|
117
|
+
}
|
|
118
|
+
}
|
|
119
|
+
return result;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* Evaluate one replayed case against its rubric. Quality precedence:
|
|
124
|
+
* violated > unmet > unknown > met. SIMULATED replays can never claim a
|
|
125
|
+
* quality label — user acceptance is not observable.
|
|
126
|
+
*/
|
|
127
|
+
function evaluateCase(goldenCase, policyId) {
|
|
128
|
+
const replay = replayCase(goldenCase, { policyId });
|
|
129
|
+
const base = {
|
|
130
|
+
case_id: replay.case_id,
|
|
131
|
+
cohort: typeof goldenCase.cohort === 'string' ? goldenCase.cohort : 'default',
|
|
132
|
+
fidelity: { tag: replay.tag, limitations: replay.limitations },
|
|
133
|
+
};
|
|
134
|
+
if (!replay.ok) {
|
|
135
|
+
return { ...base, ok: false, error: replay.reason, quality: 'unknown', metrics: {} };
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
const records = replay.records;
|
|
139
|
+
const summary = summarizeTrace(records);
|
|
140
|
+
const names = semanticNames(records);
|
|
141
|
+
const actions = actionLabels(records);
|
|
142
|
+
const expect = isPlainObject(goldenCase.expect) ? goldenCase.expect : {};
|
|
143
|
+
|
|
144
|
+
const evidenceRequired = Array.isArray(expect.evidence) ? expect.evidence : [];
|
|
145
|
+
const evidenceMissing = evidenceRequired.filter((name) => !names.has(name));
|
|
146
|
+
const actionsRequired = Array.isArray(expect.actions) ? expect.actions : [];
|
|
147
|
+
const actionsMissing = actionsRequired.filter((name) => !actions.has(name));
|
|
148
|
+
const forbiddenList = Array.isArray(expect.forbidden_mistakes) ? expect.forbidden_mistakes : [];
|
|
149
|
+
const forbiddenHits = forbiddenList.filter((name) => names.has(name));
|
|
150
|
+
|
|
151
|
+
const outcome = executionOutcome(records);
|
|
152
|
+
const expectedOutcome = isPlainObject(expect.verification) && typeof expect.verification.outcome === 'string'
|
|
153
|
+
? expect.verification.outcome
|
|
154
|
+
: null;
|
|
155
|
+
const verification = expectedOutcome === null
|
|
156
|
+
? (outcome === null ? 'unknown' : 'met')
|
|
157
|
+
: (outcome === null ? 'unknown' : (outcome === expectedOutcome ? 'met' : 'violated'));
|
|
158
|
+
|
|
159
|
+
const succeeded = outcome === 'success'
|
|
160
|
+
|| (outcome === null && summary.coverage.open_spans === 0 && summary.retries.failed_spans === 0);
|
|
161
|
+
const corrected = correctionObserved(records);
|
|
162
|
+
const metrics = {
|
|
163
|
+
tokens_to_success: succeeded ? summary.resource.total_value : null,
|
|
164
|
+
time_to_success_ms: succeeded ? summary.critical_path_ms : null,
|
|
165
|
+
model_attempts: summary.retries.model_attempts,
|
|
166
|
+
correction_observed: corrected,
|
|
167
|
+
// one accepted success per successful episode; a correction means the
|
|
168
|
+
// first answer was NOT it.
|
|
169
|
+
accepted_success: succeeded ? 1 : 0,
|
|
170
|
+
first_answer_accepted: succeeded && !corrected && summary.retries.model_attempts <= 1,
|
|
171
|
+
succeeded,
|
|
172
|
+
};
|
|
173
|
+
const bounds = checkBounds(expect.bounds, metrics);
|
|
174
|
+
|
|
175
|
+
let quality;
|
|
176
|
+
if (replay.tag === 'SIMULATED') {
|
|
177
|
+
quality = 'unknown';
|
|
178
|
+
} else if (forbiddenHits.length > 0 || verification === 'violated' || bounds.violated > 0) {
|
|
179
|
+
quality = 'violated';
|
|
180
|
+
} else if (evidenceMissing.length > 0 || actionsMissing.length > 0) {
|
|
181
|
+
quality = 'unmet';
|
|
182
|
+
} else if (verification === 'unknown' || bounds.unknown > 0) {
|
|
183
|
+
quality = 'unknown';
|
|
184
|
+
} else {
|
|
185
|
+
quality = 'met';
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
return {
|
|
189
|
+
...base,
|
|
190
|
+
ok: true,
|
|
191
|
+
quality,
|
|
192
|
+
evidence: { required: evidenceRequired, missing: evidenceMissing },
|
|
193
|
+
actions: { required: actionsRequired, missing: actionsMissing },
|
|
194
|
+
forbidden_mistakes: { forbidden: forbiddenList, hits: forbiddenHits },
|
|
195
|
+
verification: { expected: expectedOutcome, observed: outcome, status: verification },
|
|
196
|
+
bounds,
|
|
197
|
+
metrics,
|
|
198
|
+
provenance: replay.provenance,
|
|
199
|
+
};
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
function meanOf(values) {
|
|
203
|
+
if (values.length === 0) return null;
|
|
204
|
+
return values.reduce((a, b) => a + b, 0) / values.length;
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
function distribution(values) {
|
|
208
|
+
const sorted = values.filter((v) => typeof v === 'number' && Number.isFinite(v)).sort((a, b) => a - b);
|
|
209
|
+
return {
|
|
210
|
+
n: sorted.length,
|
|
211
|
+
mean: meanOf(sorted),
|
|
212
|
+
p50: percentileOf(sorted, 50),
|
|
213
|
+
p95: percentileOf(sorted, 95),
|
|
214
|
+
};
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
function emptyQuality() {
|
|
218
|
+
return {
|
|
219
|
+
met: 0, unmet: 0, violated: 0, unknown: 0, forbidden: 0,
|
|
220
|
+
bounds: { checked: 0, violated: 0, unknown: 0 },
|
|
221
|
+
};
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
function accumulateQuality(bucket, result) {
|
|
225
|
+
if (result.quality === 'met') bucket.met += 1;
|
|
226
|
+
else if (result.quality === 'unmet') bucket.unmet += 1;
|
|
227
|
+
else if (result.quality === 'violated') bucket.violated += 1;
|
|
228
|
+
else bucket.unknown += 1;
|
|
229
|
+
if (result.forbidden_mistakes) bucket.forbidden += result.forbidden_mistakes.hits.length;
|
|
230
|
+
if (result.bounds) {
|
|
231
|
+
bucket.bounds.checked += result.bounds.checked;
|
|
232
|
+
bucket.bounds.violated += result.bounds.violated;
|
|
233
|
+
bucket.bounds.unknown += result.bounds.unknown;
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
function cohortVerdict(results, minN) {
|
|
238
|
+
if (results.length < minN) return 'inconclusive';
|
|
239
|
+
if (results.some((r) => r.quality === 'violated' || r.quality === 'unmet')) return 'fail';
|
|
240
|
+
if (results.every((r) => r.quality === 'unknown')) return 'inconclusive';
|
|
241
|
+
return 'pass';
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
/**
|
|
245
|
+
* @param {object} corpus — `{ cases: GoldenCase[], pre_registered?, metric_version? }`.
|
|
246
|
+
* @param {string|object} policy — policy id (resolved against
|
|
247
|
+
* `corpus.pre_registered.policies`) or an inline policy object.
|
|
248
|
+
* @returns {object} Scorecard
|
|
249
|
+
*/
|
|
250
|
+
export function evaluateVariant(corpus, policy) {
|
|
251
|
+
const source = isPlainObject(corpus) ? corpus : {};
|
|
252
|
+
const cases = Array.isArray(source.cases) ? source.cases : [];
|
|
253
|
+
const pre = isPlainObject(source.pre_registered) ? source.pre_registered : {};
|
|
254
|
+
const minN = Number.isInteger(pre.min_cohort_n) && pre.min_cohort_n > 0
|
|
255
|
+
? pre.min_cohort_n
|
|
256
|
+
: DEFAULT_MIN_COHORT_N;
|
|
257
|
+
const policyInfo = resolvePolicy(source, policy);
|
|
258
|
+
|
|
259
|
+
const results = cases.map((c) => evaluateCase(isPlainObject(c) ? c : {}, policyInfo.id));
|
|
260
|
+
|
|
261
|
+
const quality = emptyQuality();
|
|
262
|
+
const fidelityTags = {};
|
|
263
|
+
const fidelityLimitations = new Set();
|
|
264
|
+
const cohorts = new Map();
|
|
265
|
+
const latencyValues = [];
|
|
266
|
+
const tokenValues = [];
|
|
267
|
+
|
|
268
|
+
for (const result of results) {
|
|
269
|
+
accumulateQuality(quality, result);
|
|
270
|
+
fidelityTags[result.fidelity.tag] = (fidelityTags[result.fidelity.tag] || 0) + 1;
|
|
271
|
+
for (const lim of result.fidelity.limitations) fidelityLimitations.add(lim);
|
|
272
|
+
if (typeof result.metrics.time_to_success_ms === 'number') {
|
|
273
|
+
latencyValues.push(result.metrics.time_to_success_ms);
|
|
274
|
+
}
|
|
275
|
+
if (typeof result.metrics.tokens_to_success === 'number') {
|
|
276
|
+
tokenValues.push(result.metrics.tokens_to_success);
|
|
277
|
+
}
|
|
278
|
+
if (!cohorts.has(result.cohort)) cohorts.set(result.cohort, []);
|
|
279
|
+
cohorts.get(result.cohort).push(result);
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
const cohortReport = {};
|
|
283
|
+
for (const [key, members] of [...cohorts.entries()].sort(([a], [b]) => a.localeCompare(b))) {
|
|
284
|
+
const q = emptyQuality();
|
|
285
|
+
for (const m of members) accumulateQuality(q, m);
|
|
286
|
+
cohortReport[key] = {
|
|
287
|
+
n: members.length,
|
|
288
|
+
min_n: minN,
|
|
289
|
+
verdict: cohortVerdict(members, minN),
|
|
290
|
+
quality: q,
|
|
291
|
+
latency: distribution(members.map((m) => m.metrics.time_to_success_ms)),
|
|
292
|
+
tokens: distribution(members.map((m) => m.metrics.tokens_to_success)),
|
|
293
|
+
cases: members.map((m) => m.case_id),
|
|
294
|
+
};
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
const cohortVerdicts = Object.values(cohortReport).map((c) => c.verdict);
|
|
298
|
+
const verdict = cohortVerdicts.includes('fail')
|
|
299
|
+
? 'fail'
|
|
300
|
+
: (cohortVerdicts.length === 0 || cohortVerdicts.includes('inconclusive') ? 'inconclusive' : 'pass');
|
|
301
|
+
|
|
302
|
+
return {
|
|
303
|
+
metric_version: typeof source.metric_version === 'string'
|
|
304
|
+
? source.metric_version
|
|
305
|
+
: EVALUATION_METRIC_VERSION,
|
|
306
|
+
policy: policyInfo,
|
|
307
|
+
quality,
|
|
308
|
+
latency: distribution(latencyValues),
|
|
309
|
+
tokens: distribution(tokenValues),
|
|
310
|
+
cohorts: cohortReport,
|
|
311
|
+
provenance: {
|
|
312
|
+
policy: policyInfo,
|
|
313
|
+
versions: {
|
|
314
|
+
metric_version: typeof source.metric_version === 'string'
|
|
315
|
+
? source.metric_version
|
|
316
|
+
: EVALUATION_METRIC_VERSION,
|
|
317
|
+
...(isPlainObject(pre.versions) ? pre.versions : {}),
|
|
318
|
+
},
|
|
319
|
+
pre_registered: {
|
|
320
|
+
min_cohort_n: minN,
|
|
321
|
+
cohorts: Array.isArray(pre.cohorts) ? pre.cohorts : Object.keys(cohortReport),
|
|
322
|
+
policies: isPlainObject(pre.policies) ? Object.keys(pre.policies).sort() : [policyInfo.id],
|
|
323
|
+
},
|
|
324
|
+
corpus_cases: cases.length,
|
|
325
|
+
},
|
|
326
|
+
fidelity: {
|
|
327
|
+
tags: fidelityTags,
|
|
328
|
+
limitations: [...fidelityLimitations].sort(),
|
|
329
|
+
},
|
|
330
|
+
sample_size: results.length,
|
|
331
|
+
verdict,
|
|
332
|
+
cases: results,
|
|
333
|
+
};
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
/**
|
|
337
|
+
* Paired bootstrap CI for per-case deltas (fixed seed → deterministic).
|
|
338
|
+
* Returns null when the pair count cannot support an interval.
|
|
339
|
+
*/
|
|
340
|
+
function bootstrapCi95(deltas, seed) {
|
|
341
|
+
if (deltas.length < 2) return null;
|
|
342
|
+
const rand = mulberry32(seed);
|
|
343
|
+
const means = [];
|
|
344
|
+
for (let i = 0; i < BOOTSTRAP_RESAMPLES; i += 1) {
|
|
345
|
+
let sum = 0;
|
|
346
|
+
for (let j = 0; j < deltas.length; j += 1) {
|
|
347
|
+
sum += deltas[Math.floor(rand() * deltas.length)];
|
|
348
|
+
}
|
|
349
|
+
means.push(sum / deltas.length);
|
|
350
|
+
}
|
|
351
|
+
means.sort((a, b) => a - b);
|
|
352
|
+
return [percentileOf(means, 2.5), percentileOf(means, 97.5)];
|
|
353
|
+
}
|
|
354
|
+
|
|
355
|
+
function metricDelta(pairs, key, seed) {
|
|
356
|
+
const deltas = [];
|
|
357
|
+
let skipped = 0;
|
|
358
|
+
for (const [a, b] of pairs) {
|
|
359
|
+
const va = a.metrics[key];
|
|
360
|
+
const vb = b.metrics[key];
|
|
361
|
+
if (typeof va === 'number' && Number.isFinite(va)
|
|
362
|
+
&& typeof vb === 'number' && Number.isFinite(vb)) {
|
|
363
|
+
deltas.push(vb - va);
|
|
364
|
+
} else {
|
|
365
|
+
skipped += 1;
|
|
366
|
+
}
|
|
367
|
+
}
|
|
368
|
+
return {
|
|
369
|
+
n_paired: deltas.length,
|
|
370
|
+
skipped,
|
|
371
|
+
mean_delta: meanOf(deltas),
|
|
372
|
+
ci95: bootstrapCi95(deltas, seed),
|
|
373
|
+
};
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
function directionOf(metric) {
|
|
377
|
+
if (!metric.ci95) return 'inconclusive';
|
|
378
|
+
if (metric.ci95[1] < 0) return 'improvement';
|
|
379
|
+
if (metric.ci95[0] > 0) return 'regression';
|
|
380
|
+
return 'no_difference';
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
/**
|
|
384
|
+
* Compare two scorecards over the same golden case set. Paired by case_id
|
|
385
|
+
* within each cohort; differences carry a deterministic bootstrap
|
|
386
|
+
* confidence band. Token/latency deltas are proxy evidence — the report
|
|
387
|
+
* never upgrades a proxy-only delta into a user-visible improvement claim.
|
|
388
|
+
*/
|
|
389
|
+
export function compareScorecards(a, b, { seed = 42 } = {}) {
|
|
390
|
+
const aById = new Map((a.cases || []).map((c) => [c.case_id, c]));
|
|
391
|
+
const bById = new Map((b.cases || []).map((c) => [c.case_id, c]));
|
|
392
|
+
const cohortKeys = new Set([...Object.keys(a.cohorts || {}), ...Object.keys(b.cohorts || {})]);
|
|
393
|
+
|
|
394
|
+
const cohorts = {};
|
|
395
|
+
for (const key of [...cohortKeys].sort()) {
|
|
396
|
+
const pairs = [];
|
|
397
|
+
for (const [id, ca] of aById) {
|
|
398
|
+
const cb = bById.get(id);
|
|
399
|
+
if (cb && ca.cohort === key && cb.cohort === key && ca.ok && cb.ok) pairs.push([ca, cb]);
|
|
400
|
+
}
|
|
401
|
+
const tokens = metricDelta(pairs, 'tokens_to_success', seed);
|
|
402
|
+
const time = metricDelta(pairs, 'time_to_success_ms', seed + 1);
|
|
403
|
+
const directions = [directionOf(tokens), directionOf(time)];
|
|
404
|
+
const verdict = directions.includes('regression')
|
|
405
|
+
? 'regression'
|
|
406
|
+
: directions.includes('improvement')
|
|
407
|
+
? 'improvement'
|
|
408
|
+
: directions.every((d) => d === 'inconclusive') ? 'inconclusive' : 'no_difference';
|
|
409
|
+
cohorts[key] = {
|
|
410
|
+
n_paired: pairs.length,
|
|
411
|
+
tokens_to_success: tokens,
|
|
412
|
+
time_to_success_ms: time,
|
|
413
|
+
quality: {
|
|
414
|
+
met_delta: (b.cohorts?.[key]?.quality.met ?? 0) - (a.cohorts?.[key]?.quality.met ?? 0),
|
|
415
|
+
violated_delta: (b.cohorts?.[key]?.quality.violated ?? 0)
|
|
416
|
+
- (a.cohorts?.[key]?.quality.violated ?? 0),
|
|
417
|
+
},
|
|
418
|
+
verdict,
|
|
419
|
+
};
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
const verdicts = Object.values(cohorts).map((c) => c.verdict);
|
|
423
|
+
const verdict = verdicts.includes('regression')
|
|
424
|
+
? 'regression'
|
|
425
|
+
: verdicts.includes('improvement')
|
|
426
|
+
? 'improvement'
|
|
427
|
+
: verdicts.includes('inconclusive') ? 'inconclusive' : 'no_difference';
|
|
428
|
+
|
|
429
|
+
return {
|
|
430
|
+
metric_version: EVALUATION_METRIC_VERSION,
|
|
431
|
+
seed,
|
|
432
|
+
baseline: a.policy?.id ?? 'a',
|
|
433
|
+
variant: b.policy?.id ?? 'b',
|
|
434
|
+
cohorts,
|
|
435
|
+
verdict,
|
|
436
|
+
basis: 'proxy_metrics_plus_quality_labels',
|
|
437
|
+
notes: [
|
|
438
|
+
'tokens_to_success and time_to_success_ms are proxy metrics: a proxy-only '
|
|
439
|
+
+ 'delta does not claim user-visible improvement — human-reviewable '
|
|
440
|
+
+ 'quality labels (met/unmet/violated) are reported separately.',
|
|
441
|
+
'confidence bands are paired bootstrap intervals at a fixed seed; '
|
|
442
|
+
+ 'cohorts below the pre-registered minimum are inconclusive, never failed.',
|
|
443
|
+
],
|
|
444
|
+
};
|
|
445
|
+
}
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* allowlist.js (TASK-004, SPEC §5 DF-FR06 / §8)
|
|
3
|
+
*
|
|
4
|
+
* Canonical privacy-gate constants for the observability pipeline. The
|
|
5
|
+
* allowlist — not the secret scanner — is the persistence boundary: only
|
|
6
|
+
* fields named here may appear on a record that reaches disk, and only
|
|
7
|
+
* ALLOWED_SUPPORT_PAYLOAD_FIELDS may appear in a record projected to the
|
|
8
|
+
* user-facing support view.
|
|
9
|
+
*
|
|
10
|
+
* Bump REDACTION_VERSION whenever the allowlist, deny list, caps, or
|
|
11
|
+
* redaction rules change so downstream readers can detect stale records.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
export const REDACTION_VERSION = 'df-redact-3';
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* Envelope fields permitted on a persisted record (DF-FR01) plus the
|
|
18
|
+
* redaction stamp this module adds. Anything else is dropped by
|
|
19
|
+
* sanitizeObserved and rejected by sanitizeForSupport.
|
|
20
|
+
*/
|
|
21
|
+
export const ALLOWED_FIELDS = new Set([
|
|
22
|
+
'record_type',
|
|
23
|
+
'semantic_name',
|
|
24
|
+
'schema_version',
|
|
25
|
+
'record_id',
|
|
26
|
+
'trace_id',
|
|
27
|
+
'span_id',
|
|
28
|
+
'parent_span_id',
|
|
29
|
+
'execution_id',
|
|
30
|
+
'session_id',
|
|
31
|
+
'project_ref',
|
|
32
|
+
'boot_id',
|
|
33
|
+
'writer_id',
|
|
34
|
+
'sequence',
|
|
35
|
+
'wall_time_utc',
|
|
36
|
+
'monotonic_ns',
|
|
37
|
+
'importance',
|
|
38
|
+
'privacy_class',
|
|
39
|
+
'payload',
|
|
40
|
+
'origin',
|
|
41
|
+
'redaction_version',
|
|
42
|
+
]);
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Payload keys whose values are free-form content by definition and are
|
|
46
|
+
* never persisted in any form — user-entered feedback is excluded by
|
|
47
|
+
* default per SPEC §14/planner note, and raw prompts/commands/output are
|
|
48
|
+
* content, not metadata. Dropped silently by sanitizeObserved.
|
|
49
|
+
*/
|
|
50
|
+
export const DENIED_PAYLOAD_FIELDS = new Set([
|
|
51
|
+
'error_detail',
|
|
52
|
+
'feedback',
|
|
53
|
+
'comment',
|
|
54
|
+
'note_text',
|
|
55
|
+
'prompt',
|
|
56
|
+
'prompt_text',
|
|
57
|
+
'system_prompt',
|
|
58
|
+
'messages',
|
|
59
|
+
'command',
|
|
60
|
+
'cmd',
|
|
61
|
+
'argv',
|
|
62
|
+
'stdin',
|
|
63
|
+
'stdout',
|
|
64
|
+
'stderr',
|
|
65
|
+
'output',
|
|
66
|
+
'raw_output',
|
|
67
|
+
'body',
|
|
68
|
+
'request_body',
|
|
69
|
+
'response_body',
|
|
70
|
+
'content',
|
|
71
|
+
'text',
|
|
72
|
+
'diff',
|
|
73
|
+
'patch',
|
|
74
|
+
'snippet',
|
|
75
|
+
'env',
|
|
76
|
+
'dotenv',
|
|
77
|
+
'secret',
|
|
78
|
+
'token',
|
|
79
|
+
'api_key',
|
|
80
|
+
'apikey',
|
|
81
|
+
'password',
|
|
82
|
+
'passwd',
|
|
83
|
+
'credential',
|
|
84
|
+
'credentials',
|
|
85
|
+
'auth',
|
|
86
|
+
'authorization',
|
|
87
|
+
'cookie',
|
|
88
|
+
'cookies',
|
|
89
|
+
'session_token',
|
|
90
|
+
'private_key',
|
|
91
|
+
'email',
|
|
92
|
+
'username',
|
|
93
|
+
'hostname',
|
|
94
|
+
'host',
|
|
95
|
+
'ip',
|
|
96
|
+
'ip_address',
|
|
97
|
+
'url',
|
|
98
|
+
'uri',
|
|
99
|
+
'path',
|
|
100
|
+
'file_path',
|
|
101
|
+
'filepath',
|
|
102
|
+
'abs_path',
|
|
103
|
+
'cwd',
|
|
104
|
+
'home',
|
|
105
|
+
'home_dir',
|
|
106
|
+
'stack',
|
|
107
|
+
'stack_trace',
|
|
108
|
+
'traceback',
|
|
109
|
+
]);
|
|
110
|
+
|
|
111
|
+
/**
|
|
112
|
+
* Payload keys permitted in the support projection — the second,
|
|
113
|
+
* independent, default-deny boundary. Deliberately narrower than "anything
|
|
114
|
+
* the observed gate kept": codes, counts, timings, statuses, and typed
|
|
115
|
+
* resource usage only. Applied recursively to every nested object.
|
|
116
|
+
*/
|
|
117
|
+
export const ALLOWED_SUPPORT_PAYLOAD_FIELDS = new Set([
|
|
118
|
+
'operation',
|
|
119
|
+
'status',
|
|
120
|
+
'outcome',
|
|
121
|
+
'phase',
|
|
122
|
+
'stage',
|
|
123
|
+
'kind',
|
|
124
|
+
'type',
|
|
125
|
+
'name',
|
|
126
|
+
'tool_name',
|
|
127
|
+
'error_code',
|
|
128
|
+
'reason_code',
|
|
129
|
+
'wait_code',
|
|
130
|
+
'exit_code',
|
|
131
|
+
'signal',
|
|
132
|
+
'attempt',
|
|
133
|
+
'attempts',
|
|
134
|
+
'retry_count',
|
|
135
|
+
'duration_ms',
|
|
136
|
+
'elapsed_ms',
|
|
137
|
+
'latency_ms',
|
|
138
|
+
'queue_ms',
|
|
139
|
+
'sequence',
|
|
140
|
+
'count',
|
|
141
|
+
'total',
|
|
142
|
+
'dropped',
|
|
143
|
+
'sampled',
|
|
144
|
+
'truncated',
|
|
145
|
+
'cache',
|
|
146
|
+
'hit',
|
|
147
|
+
'miss',
|
|
148
|
+
'resource',
|
|
149
|
+
'value',
|
|
150
|
+
'unit',
|
|
151
|
+
'source',
|
|
152
|
+
'confidence',
|
|
153
|
+
'schema_version',
|
|
154
|
+
'metric_version',
|
|
155
|
+
'fingerprint',
|
|
156
|
+
'p50',
|
|
157
|
+
'p95',
|
|
158
|
+
'p99',
|
|
159
|
+
'min',
|
|
160
|
+
'max',
|
|
161
|
+
'mean',
|
|
162
|
+
]);
|
|
163
|
+
|
|
164
|
+
// --- size caps (prototype defaults per SPEC §14; freeze after measurement) ---
|
|
165
|
+
|
|
166
|
+
/** Max serialized bytes for a sanitized record; larger records are rejected. */
|
|
167
|
+
export const MAX_RECORD_BYTES = 64 * 1024;
|
|
168
|
+
|
|
169
|
+
/** Max characters kept per string value; longer strings are truncated. */
|
|
170
|
+
export const MAX_STRING_CHARS = 8 * 1024;
|
|
171
|
+
|
|
172
|
+
/** Max serialized bytes for a record admitted to the support view. */
|
|
173
|
+
export const MAX_SUPPORT_RECORD_BYTES = 16 * 1024;
|
|
174
|
+
|
|
175
|
+
/** Max characters per string value in the support view. */
|
|
176
|
+
export const MAX_SUPPORT_STRING_CHARS = 1024;
|
|
177
|
+
|
|
178
|
+
/** Max keys kept per object at any depth. */
|
|
179
|
+
export const MAX_OBJECT_KEYS = 64;
|
|
180
|
+
|
|
181
|
+
/** Max items kept per array. */
|
|
182
|
+
export const MAX_ARRAY_ITEMS = 128;
|
|
183
|
+
|
|
184
|
+
/** Max nesting depth; deeper values are dropped. */
|
|
185
|
+
export const MAX_DEPTH = 8;
|