@ngockhoale/ukit 3.0.6 → 3.0.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/package.json +1 -1
- package/scripts/bench/data-foundation.mjs +562 -0
- package/src/core/observability/adapters/common.js +75 -0
- package/src/core/observability/adapters/contextAdapter.js +55 -0
- package/src/core/observability/adapters/decisionAdapter.js +61 -0
- package/src/core/observability/adapters/routeAdapter.js +135 -0
- package/src/core/observability/analytics/digest.js +186 -0
- package/src/core/observability/analytics/fingerprints.js +126 -0
- package/src/core/observability/analytics/opportunities.js +329 -0
- package/src/core/observability/analytics/rebuild.js +56 -0
- package/src/core/observability/analytics/summary.js +298 -0
- package/src/core/observability/emit/config.js +29 -0
- package/src/core/observability/emit/recorder.js +297 -0
- package/src/core/observability/evaluation/aiPacket.js +230 -0
- package/src/core/observability/evaluation/optimizationKnowledge.js +172 -0
- package/src/core/observability/evaluation/replay.js +143 -0
- package/src/core/observability/evaluation/scorecard.js +445 -0
- package/src/core/observability/privacy/allowlist.js +185 -0
- package/src/core/observability/privacy/redaction.js +113 -0
- package/src/core/observability/privacy/sanitizeForSupport.js +133 -0
- package/src/core/observability/privacy/sanitizeObserved.js +134 -0
- package/src/core/observability/rollout.js +155 -0
- package/src/core/observability/schema/constants.js +66 -0
- package/src/core/observability/schema/registry.js +223 -0
- package/src/core/observability/schema/validate.js +227 -0
- package/src/core/observability/segments/internal.js +241 -0
- package/src/core/observability/segments/readSegments.js +215 -0
- package/src/core/observability/segments/recovery.js +123 -0
- package/src/core/observability/segments/retention.js +381 -0
- package/src/core/observability/support/import.js +402 -0
- package/src/core/observability/support/manifest.js +135 -0
- package/src/core/observability/support/paths.js +94 -0
- package/src/core/observability/support/projector.js +483 -0
- package/src/core/observability/support/renderer.js +130 -0
- package/src/core/observability/support/retention.js +155 -0
- package/template_project/.omp/RULES.md +6 -6
- package/template_project/.omp/config.yml +6 -0
- package/template_project/instructions/overlays/omp-rules.md +6 -6
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,18 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to UKit are documented here.
|
|
4
4
|
|
|
5
|
+
## 3.0.7 - 2026-09-24
|
|
6
|
+
|
|
7
|
+
- **Data foundation & flight recorder (cycle C54)** — new `src/core/observability/` pipeline, all behind `observability.stage` (default `off`, staged `off→shadow→canary→default` per seam with kill switch):
|
|
8
|
+
- **Schema + privacy**: versioned record envelope (`schema_version: 1`), semantic/reason-code registry, and a two-gate privacy boundary — `sanitizeObserved` before any persistent write (allowlist + deny-list + redaction), `sanitizeForSupport` as an independent default-deny second gate.
|
|
9
|
+
- **Recorder + segments**: bounded non-blocking `emit()` with live stage gate, span tracking, deadline-bounded flush, and append-only JSONL segments with rotation/retention and partial-tail recovery.
|
|
10
|
+
- **Analytics**: deterministic trace summaries (p50/p95/p99, causal critical path, retry/drop/cache metrics), honest `telemetry_complete=false` on drops/gaps/UNKNOWN host usage, and rebuild-from-segments recovery.
|
|
11
|
+
- **Support view**: `UKit Support` materialized projection under the OS-native Documents folder — bundle-local pseudonyms, atomic writes, byte caps, failure-promoted digests, and a validated import path (manifest/checksums/traversal/decompression bounds).
|
|
12
|
+
- **Evaluation**: offline evaluator packet (summary → anomalies → digest → evidence) and an optimization knowledge base recording hypothesis → experiment → guardrails → decision → rollback.
|
|
13
|
+
- **Rollout**: `resolveSeamStages`/`resolveSeamConfig`/`killSwitch` in `src/core/observability/rollout.js` — recorder activates at `shadow`, support projection at `canary`, evaluation at `default`; per-seam overrides are restrictive-only so a global `off` is an absolute kill.
|
|
14
|
+
- **Privacy fix**: `error_detail` added to the deny-list (free-form text, not a code) and `REDACTION_VERSION` bumped to `df-redact-3` — fixture secrets can no longer persist in canonical segments.
|
|
15
|
+
- **unic-decision docs**: `docs/pstack/UNIC_DECISION_GUIDE.md` documents the local Lava/JEV decision model (not an LLM), checkpoint IDs, typed `tool_calls` contract, and `modelRoles.decision` binding for omp.
|
|
16
|
+
|
|
5
17
|
## 3.0.6 - 2026-09-24
|
|
6
18
|
|
|
7
19
|
- Clarified the project owner's `unic-decision` contract in owner instructions, project memory, the technical spec, and shipped internals: a local Lava/JEV model in UNIC Provider, not an LLM; the OpenAI-compatible API is transport for convenient integration, not evidence of remote hosting or generative-model pricing.
|
package/package.json
CHANGED
|
@@ -0,0 +1,562 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// data-foundation bench harness (C54 / DF-00)
|
|
3
|
+
//
|
|
4
|
+
// --fixture Regenerate tests/fixtures/observability/corpus.json deterministically (seed=42).
|
|
5
|
+
// --recorder Emit the corpus through createRecorder into a temp segment store.
|
|
6
|
+
// --analytics Rebuild the derived index from retained segments and print summaries.
|
|
7
|
+
// --evaluate Replay the golden corpus (tests/fixtures/observability/golden/cases.json)
|
|
8
|
+
// under both pre-registered policies and print paired scorecards +
|
|
9
|
+
// a fixed-seed variance report (TASK-014).
|
|
10
|
+
//
|
|
11
|
+
// The corpus is a synthetic, seeded fixture: no real user content. Records are
|
|
12
|
+
// envelope-shaped per SPEC §7 (DF-FR01) but intentionally NOT schema-validated —
|
|
13
|
+
// TASK-003 owns validation. Unknown host measurements are marked UNKNOWN, never 0.
|
|
14
|
+
|
|
15
|
+
import fs from 'node:fs';
|
|
16
|
+
import os from 'node:os';
|
|
17
|
+
import path from 'node:path';
|
|
18
|
+
import { fileURLToPath } from 'node:url';
|
|
19
|
+
import { SEMANTIC_REGISTRY } from '../../src/core/observability/schema/registry.js';
|
|
20
|
+
import { PRIVACY_CLASSES } from '../../src/core/observability/schema/constants.js';
|
|
21
|
+
|
|
22
|
+
const SEED = 42;
|
|
23
|
+
const METRIC_VERSION = 'df-m1';
|
|
24
|
+
const SCHEMA_VERSION = 1;
|
|
25
|
+
const BOOT_ID = 'boot-fixture-0001';
|
|
26
|
+
const WRITER_ID = 'writer-fixture-0001';
|
|
27
|
+
const SESSION_ID = 'session-fixture-0001';
|
|
28
|
+
const PROJECT_REF = 'project-fixture-0001';
|
|
29
|
+
const WALL_EPOCH_MS = Date.UTC(2026, 0, 1, 0, 0, 0); // fixed epoch → deterministic wall_time_utc
|
|
30
|
+
|
|
31
|
+
const repoRoot = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..');
|
|
32
|
+
const corpusPath = path.join(repoRoot, 'tests/fixtures/observability/corpus.json');
|
|
33
|
+
|
|
34
|
+
// Records may raise privacy above the registry floor but never lower it —
|
|
35
|
+
// validateSemanticRecord (TASK-003) rejects below-floor classes, so the
|
|
36
|
+
// fixture clamps each record's declared class up to the registry floor.
|
|
37
|
+
function clampToRegistryFloor(semanticName, privacyClass) {
|
|
38
|
+
const floor = SEMANTIC_REGISTRY[semanticName]?.privacy_class;
|
|
39
|
+
if (!floor) return privacyClass;
|
|
40
|
+
return PRIVACY_CLASSES.indexOf(privacyClass) < PRIVACY_CLASSES.indexOf(floor) ? floor : privacyClass;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
// Deterministic PRNG (mulberry32) — used only for token counts; IDs are counters.
|
|
44
|
+
function mulberry32(seed) {
|
|
45
|
+
let a = seed >>> 0;
|
|
46
|
+
return function next() {
|
|
47
|
+
a |= 0;
|
|
48
|
+
a = (a + 0x6d2b79f5) | 0;
|
|
49
|
+
let t = Math.imul(a ^ (a >>> 15), 1 | a);
|
|
50
|
+
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
|
|
51
|
+
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
|
52
|
+
};
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
function makeCtx() {
|
|
56
|
+
const rand = mulberry32(SEED);
|
|
57
|
+
let seq = 0;
|
|
58
|
+
let rec = 0;
|
|
59
|
+
let span = 0;
|
|
60
|
+
let trace = 0;
|
|
61
|
+
let clockMs = 0;
|
|
62
|
+
return {
|
|
63
|
+
nextSeq: () => ++seq,
|
|
64
|
+
recordId: () => `rec-${String(++rec).padStart(4, '0')}`,
|
|
65
|
+
spanId: () => `span-${String(++span).padStart(4, '0')}`,
|
|
66
|
+
traceId: () => `trace-${String(++trace).padStart(4, '0')}`,
|
|
67
|
+
// advance the deterministic clock by `ms` and return { wall_time_utc, monotonic_ns }
|
|
68
|
+
tick: (ms) => {
|
|
69
|
+
clockMs += ms;
|
|
70
|
+
return {
|
|
71
|
+
wall_time_utc: new Date(WALL_EPOCH_MS + clockMs).toISOString(),
|
|
72
|
+
monotonic_ns: clockMs * 1e6,
|
|
73
|
+
};
|
|
74
|
+
},
|
|
75
|
+
tokens: (lo, hi) => lo + Math.floor(rand() * (hi - lo + 1)),
|
|
76
|
+
};
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
function record(ctx, { semanticName, traceId, spanId, parentSpanId, executionId, importance, privacyClass, payload, dtMs }) {
|
|
80
|
+
const t = ctx.tick(dtMs);
|
|
81
|
+
return {
|
|
82
|
+
record_type: 'fact',
|
|
83
|
+
semantic_name: semanticName,
|
|
84
|
+
schema_version: SCHEMA_VERSION,
|
|
85
|
+
record_id: ctx.recordId(),
|
|
86
|
+
trace_id: traceId,
|
|
87
|
+
span_id: spanId,
|
|
88
|
+
parent_span_id: parentSpanId,
|
|
89
|
+
execution_id: executionId,
|
|
90
|
+
session_id: SESSION_ID,
|
|
91
|
+
project_ref: PROJECT_REF,
|
|
92
|
+
boot_id: BOOT_ID,
|
|
93
|
+
writer_id: WRITER_ID,
|
|
94
|
+
sequence: ctx.nextSeq(),
|
|
95
|
+
wall_time_utc: t.wall_time_utc,
|
|
96
|
+
monotonic_ns: t.monotonic_ns,
|
|
97
|
+
importance,
|
|
98
|
+
privacy_class: clampToRegistryFloor(semanticName, privacyClass),
|
|
99
|
+
payload,
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
function execSpan(ctx, traceId, executionId, privacyClass) {
|
|
104
|
+
const spanId = ctx.spanId();
|
|
105
|
+
return {
|
|
106
|
+
spanId,
|
|
107
|
+
started: record(ctx, {
|
|
108
|
+
semanticName: 'execution.started',
|
|
109
|
+
traceId,
|
|
110
|
+
spanId,
|
|
111
|
+
parentSpanId: null,
|
|
112
|
+
executionId,
|
|
113
|
+
importance: 'normal',
|
|
114
|
+
privacyClass,
|
|
115
|
+
payload: { operation: 'task', duration_ms: null },
|
|
116
|
+
dtMs: 10,
|
|
117
|
+
}),
|
|
118
|
+
};
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function modelAttempt(ctx, { traceId, parentSpanId, executionId, attemptIndex, status, resource, privacyClass, extra = {} }) {
|
|
122
|
+
const spanId = ctx.spanId();
|
|
123
|
+
const started = record(ctx, {
|
|
124
|
+
semanticName: 'model.started',
|
|
125
|
+
traceId,
|
|
126
|
+
spanId,
|
|
127
|
+
parentSpanId,
|
|
128
|
+
executionId,
|
|
129
|
+
importance: 'normal',
|
|
130
|
+
privacyClass,
|
|
131
|
+
payload: { attempt_index: attemptIndex, ...extra },
|
|
132
|
+
dtMs: 5,
|
|
133
|
+
});
|
|
134
|
+
const completed = record(ctx, {
|
|
135
|
+
semanticName: status === 'ok' ? 'model.completed' : 'model.failed',
|
|
136
|
+
traceId,
|
|
137
|
+
spanId,
|
|
138
|
+
parentSpanId,
|
|
139
|
+
executionId,
|
|
140
|
+
importance: 'normal',
|
|
141
|
+
privacyClass,
|
|
142
|
+
payload: {
|
|
143
|
+
attempt_index: attemptIndex,
|
|
144
|
+
duration_ms: 120 + attemptIndex * 30,
|
|
145
|
+
resource,
|
|
146
|
+
...(status === 'ok' ? {} : { error_code: extra.error_code || 'MODEL_ERROR' }),
|
|
147
|
+
...extra,
|
|
148
|
+
},
|
|
149
|
+
dtMs: 120 + attemptIndex * 30,
|
|
150
|
+
});
|
|
151
|
+
return [started, completed];
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
function execEnd(ctx, { traceId, spanId, executionId, status, privacyClass, extra = {} }) {
|
|
155
|
+
return record(ctx, {
|
|
156
|
+
semanticName: `execution.${status}`,
|
|
157
|
+
traceId,
|
|
158
|
+
spanId,
|
|
159
|
+
parentSpanId: null,
|
|
160
|
+
executionId,
|
|
161
|
+
importance: 'normal',
|
|
162
|
+
privacyClass,
|
|
163
|
+
payload: { duration_ms: 0, ...extra },
|
|
164
|
+
dtMs: 10,
|
|
165
|
+
});
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
function sumUsage(attempts) {
|
|
169
|
+
const values = attempts
|
|
170
|
+
.map((r) => r.payload.resource)
|
|
171
|
+
.filter((res) => res && typeof res.value === 'number')
|
|
172
|
+
.map((res) => res.value);
|
|
173
|
+
return values.length ? values.reduce((a, b) => a + b, 0) : null;
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
function buildTrace(ctx, { scenario, privacyClass, build }) {
|
|
177
|
+
const traceId = ctx.traceId();
|
|
178
|
+
const executionId = `exec-${traceId}`;
|
|
179
|
+
const exec = execSpan(ctx, traceId, executionId, privacyClass);
|
|
180
|
+
const { records, attempts, succeeded } = build({ traceId, executionId, execSpanId: exec.spanId });
|
|
181
|
+
const all = [exec.started, ...records];
|
|
182
|
+
return {
|
|
183
|
+
trace_id: traceId,
|
|
184
|
+
scenario,
|
|
185
|
+
privacy_class: privacyClass,
|
|
186
|
+
records: all,
|
|
187
|
+
metrics: {
|
|
188
|
+
metric_version: METRIC_VERSION,
|
|
189
|
+
tokens_to_success: succeeded ? sumUsage(attempts) : null,
|
|
190
|
+
correction_observed: scenario === 'retry',
|
|
191
|
+
},
|
|
192
|
+
};
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
function buildCorpus() {
|
|
196
|
+
const ctx = makeCtx();
|
|
197
|
+
const traces = [];
|
|
198
|
+
|
|
199
|
+
// success: single attempt, provider-reported usage.
|
|
200
|
+
traces.push(
|
|
201
|
+
buildTrace(ctx, {
|
|
202
|
+
scenario: 'success',
|
|
203
|
+
privacyClass: 'public',
|
|
204
|
+
build: ({ traceId, executionId, execSpanId }) => {
|
|
205
|
+
const resource = { value: ctx.tokens(800, 1200), source: 'PROVIDER' };
|
|
206
|
+
const attempts = modelAttempt(ctx, {
|
|
207
|
+
traceId, parentSpanId: execSpanId, executionId,
|
|
208
|
+
attemptIndex: 1, status: 'ok', resource, privacyClass: 'public',
|
|
209
|
+
});
|
|
210
|
+
const end = execEnd(ctx, {
|
|
211
|
+
traceId, spanId: execSpanId, executionId, status: 'completed',
|
|
212
|
+
privacyClass: 'public', extra: { outcome: 'success' },
|
|
213
|
+
});
|
|
214
|
+
return { records: [...attempts, end], attempts, succeeded: true };
|
|
215
|
+
},
|
|
216
|
+
}),
|
|
217
|
+
);
|
|
218
|
+
|
|
219
|
+
// retry: user correction observed → second attempt; tokens_to_success sums both.
|
|
220
|
+
traces.push(
|
|
221
|
+
buildTrace(ctx, {
|
|
222
|
+
scenario: 'retry',
|
|
223
|
+
privacyClass: 'public',
|
|
224
|
+
build: ({ traceId, executionId, execSpanId }) => {
|
|
225
|
+
const res1 = { value: ctx.tokens(400, 700), source: 'PROVIDER' };
|
|
226
|
+
const res2 = { value: ctx.tokens(500, 900), source: 'PROVIDER' };
|
|
227
|
+
const a1 = modelAttempt(ctx, {
|
|
228
|
+
traceId, parentSpanId: execSpanId, executionId,
|
|
229
|
+
attemptIndex: 1, status: 'ok', resource: res1, privacyClass: 'public',
|
|
230
|
+
});
|
|
231
|
+
const correction = record(ctx, {
|
|
232
|
+
semanticName: 'context.item.injected',
|
|
233
|
+
traceId, spanId: ctx.spanId(), parentSpanId: execSpanId, executionId,
|
|
234
|
+
importance: 'high', privacyClass: 'public',
|
|
235
|
+
payload: { kind: 'user_correction', correction_observed: true },
|
|
236
|
+
dtMs: 15,
|
|
237
|
+
});
|
|
238
|
+
const a2 = modelAttempt(ctx, {
|
|
239
|
+
traceId, parentSpanId: execSpanId, executionId,
|
|
240
|
+
attemptIndex: 2, status: 'ok', resource: res2, privacyClass: 'public',
|
|
241
|
+
});
|
|
242
|
+
const end = execEnd(ctx, {
|
|
243
|
+
traceId, spanId: execSpanId, executionId, status: 'completed',
|
|
244
|
+
privacyClass: 'public', extra: { outcome: 'success', attempts: 2 },
|
|
245
|
+
});
|
|
246
|
+
return { records: [...a1, correction, ...a2, end], attempts: [...a1, ...a2], succeeded: true };
|
|
247
|
+
},
|
|
248
|
+
}),
|
|
249
|
+
);
|
|
250
|
+
|
|
251
|
+
// failure: host-blind — usage UNKNOWN (never 0), execution failed.
|
|
252
|
+
traces.push(
|
|
253
|
+
buildTrace(ctx, {
|
|
254
|
+
scenario: 'failure',
|
|
255
|
+
privacyClass: 'public',
|
|
256
|
+
build: ({ traceId, executionId, execSpanId }) => {
|
|
257
|
+
const resource = { value: null, source: 'UNKNOWN' };
|
|
258
|
+
const attempts = modelAttempt(ctx, {
|
|
259
|
+
traceId, parentSpanId: execSpanId, executionId,
|
|
260
|
+
attemptIndex: 1, status: 'error', resource, privacyClass: 'public',
|
|
261
|
+
extra: { error_code: 'TOOL_TIMEOUT' },
|
|
262
|
+
});
|
|
263
|
+
const end = execEnd(ctx, {
|
|
264
|
+
traceId, spanId: execSpanId, executionId, status: 'failed',
|
|
265
|
+
privacyClass: 'public', extra: { outcome: 'failure', error_code: 'TOOL_TIMEOUT' },
|
|
266
|
+
});
|
|
267
|
+
return { records: [...attempts, end], attempts, succeeded: false };
|
|
268
|
+
},
|
|
269
|
+
}),
|
|
270
|
+
);
|
|
271
|
+
|
|
272
|
+
// dropped: queue overflow — telemetry.dropped counted, telemetry_complete=false.
|
|
273
|
+
traces.push(
|
|
274
|
+
buildTrace(ctx, {
|
|
275
|
+
scenario: 'dropped',
|
|
276
|
+
privacyClass: 'public',
|
|
277
|
+
build: ({ traceId, executionId, execSpanId }) => {
|
|
278
|
+
const resource = { value: ctx.tokens(300, 600), source: 'PROVIDER' };
|
|
279
|
+
const attempts = modelAttempt(ctx, {
|
|
280
|
+
traceId, parentSpanId: execSpanId, executionId,
|
|
281
|
+
attemptIndex: 1, status: 'ok', resource, privacyClass: 'public',
|
|
282
|
+
});
|
|
283
|
+
const dropped = record(ctx, {
|
|
284
|
+
semanticName: 'telemetry.dropped',
|
|
285
|
+
traceId, spanId: ctx.spanId(), parentSpanId: execSpanId, executionId,
|
|
286
|
+
importance: 'high', privacyClass: 'public',
|
|
287
|
+
payload: { reason_code: 'QUEUE_OVERFLOW', dropped_count: 3, telemetry_complete: false },
|
|
288
|
+
dtMs: 5,
|
|
289
|
+
});
|
|
290
|
+
const end = execEnd(ctx, {
|
|
291
|
+
traceId, spanId: execSpanId, executionId, status: 'completed',
|
|
292
|
+
privacyClass: 'public', extra: { outcome: 'success', telemetry_complete: false },
|
|
293
|
+
});
|
|
294
|
+
return { records: [...attempts, dropped, end], attempts, succeeded: true };
|
|
295
|
+
},
|
|
296
|
+
}),
|
|
297
|
+
);
|
|
298
|
+
|
|
299
|
+
// secret: payload carries a synthetic fixture secret (never a real credential);
|
|
300
|
+
// host-blind usage → UNKNOWN. TASK-004 leak tests scan for the marker string.
|
|
301
|
+
traces.push(
|
|
302
|
+
buildTrace(ctx, {
|
|
303
|
+
scenario: 'secret',
|
|
304
|
+
privacyClass: 'sensitive',
|
|
305
|
+
build: ({ traceId, executionId, execSpanId }) => {
|
|
306
|
+
const resource = { value: null, source: 'UNKNOWN' };
|
|
307
|
+
const attempts = modelAttempt(ctx, {
|
|
308
|
+
traceId, parentSpanId: execSpanId, executionId,
|
|
309
|
+
attemptIndex: 1, status: 'error', resource, privacyClass: 'sensitive',
|
|
310
|
+
extra: {
|
|
311
|
+
error_code: 'AUTH_FAILED',
|
|
312
|
+
error_detail: 'credential rejected: FIXTURE_SECRET_DO_NOT_LEAK_7f3a9c',
|
|
313
|
+
},
|
|
314
|
+
});
|
|
315
|
+
const end = execEnd(ctx, {
|
|
316
|
+
traceId, spanId: execSpanId, executionId, status: 'failed',
|
|
317
|
+
privacyClass: 'sensitive', extra: { outcome: 'failure', error_code: 'AUTH_FAILED' },
|
|
318
|
+
});
|
|
319
|
+
return { records: [...attempts, end], attempts, succeeded: false };
|
|
320
|
+
},
|
|
321
|
+
}),
|
|
322
|
+
);
|
|
323
|
+
|
|
324
|
+
return {
|
|
325
|
+
metric_version: METRIC_VERSION,
|
|
326
|
+
seed: SEED,
|
|
327
|
+
generated_by: 'scripts/bench/data-foundation.mjs --fixture',
|
|
328
|
+
metric_definitions: {
|
|
329
|
+
tokens_to_success: {
|
|
330
|
+
metric_version: METRIC_VERSION,
|
|
331
|
+
unit: 'tokens',
|
|
332
|
+
definition:
|
|
333
|
+
'Sum of numeric resource.value across all model attempts in the episode; null when no attempt reported a numeric value or the episode did not succeed.',
|
|
334
|
+
},
|
|
335
|
+
correction_observed: {
|
|
336
|
+
metric_version: METRIC_VERSION,
|
|
337
|
+
unit: 'boolean',
|
|
338
|
+
definition:
|
|
339
|
+
'True when a user correction was injected mid-episode and the episode required ≥2 model attempts.',
|
|
340
|
+
},
|
|
341
|
+
},
|
|
342
|
+
traces,
|
|
343
|
+
};
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
function writeFixture() {
|
|
347
|
+
const corpus = buildCorpus();
|
|
348
|
+
fs.mkdirSync(path.dirname(corpusPath), { recursive: true });
|
|
349
|
+
fs.writeFileSync(corpusPath, `${JSON.stringify(corpus, null, 2)}\n`);
|
|
350
|
+
process.stdout.write(`wrote ${path.relative(repoRoot, corpusPath)} (seed=${SEED}, traces=${corpus.traces.length})\n`);
|
|
351
|
+
return 0;
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
function percentile(sorted, p) {
|
|
355
|
+
if (sorted.length === 0) return 0;
|
|
356
|
+
const idx = Math.min(sorted.length - 1, Math.ceil((p / 100) * sorted.length) - 1);
|
|
357
|
+
return sorted[Math.max(0, idx)];
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
// --recorder: emit the whole corpus through createRecorder into a temp
|
|
361
|
+
// segment root, measure per-emit latency distribution + flush outcome, then
|
|
362
|
+
// read the store back to prove the round-trip. Exit 0 on success, 1 when the
|
|
363
|
+
// pipeline loses or corrupts records.
|
|
364
|
+
async function runRecorderBench() {
|
|
365
|
+
const { createRecorder } = await import('../../src/core/observability/emit/recorder.js');
|
|
366
|
+
const { readSegments } = await import('../../src/core/observability/segments/readSegments.js');
|
|
367
|
+
|
|
368
|
+
const corpus = JSON.parse(fs.readFileSync(corpusPath, 'utf8'));
|
|
369
|
+
const inputs = corpus.traces.flatMap((t) => t.records);
|
|
370
|
+
|
|
371
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ukit-bench-rec-'));
|
|
372
|
+
try {
|
|
373
|
+
const recorder = createRecorder({
|
|
374
|
+
root,
|
|
375
|
+
config: { observability: { stage: 'default' } },
|
|
376
|
+
});
|
|
377
|
+
|
|
378
|
+
const latenciesMs = [];
|
|
379
|
+
let accepted = 0;
|
|
380
|
+
let dropped = 0;
|
|
381
|
+
for (const record of inputs) {
|
|
382
|
+
const t0 = process.hrtime.bigint();
|
|
383
|
+
const r = recorder.emit(record);
|
|
384
|
+
latenciesMs.push(Number(process.hrtime.bigint() - t0) / 1e6);
|
|
385
|
+
if (r.status === 'accepted') accepted += 1;
|
|
386
|
+
else dropped += 1;
|
|
387
|
+
}
|
|
388
|
+
|
|
389
|
+
const flushStart = process.hrtime.bigint();
|
|
390
|
+
const flush = await recorder.flush({ deadlineMs: 60_000 });
|
|
391
|
+
const flushMs = Number(process.hrtime.bigint() - flushStart) / 1e6;
|
|
392
|
+
|
|
393
|
+
let readBack = 0;
|
|
394
|
+
const it = readSegments(root);
|
|
395
|
+
for await (const _record of it) readBack += 1;
|
|
396
|
+
|
|
397
|
+
latenciesMs.sort((a, b) => a - b);
|
|
398
|
+
const summary = {
|
|
399
|
+
metric_version: METRIC_VERSION,
|
|
400
|
+
bench: 'recorder',
|
|
401
|
+
records: inputs.length,
|
|
402
|
+
accepted,
|
|
403
|
+
dropped,
|
|
404
|
+
written: flush.written,
|
|
405
|
+
read_back: readBack,
|
|
406
|
+
flush_status: flush.status,
|
|
407
|
+
flush_ms: Number(flushMs.toFixed(3)),
|
|
408
|
+
emit_latency_ms: {
|
|
409
|
+
p50: Number(percentile(latenciesMs, 50).toFixed(4)),
|
|
410
|
+
p95: Number(percentile(latenciesMs, 95).toFixed(4)),
|
|
411
|
+
p99: Number(percentile(latenciesMs, 99).toFixed(4)),
|
|
412
|
+
max: Number((latenciesMs[latenciesMs.length - 1] || 0).toFixed(4)),
|
|
413
|
+
},
|
|
414
|
+
health: recorder.health(),
|
|
415
|
+
};
|
|
416
|
+
process.stdout.write(`${JSON.stringify(summary, null, 2)}\n`);
|
|
417
|
+
const okRun = flush.status === 'ok' && readBack === accepted && dropped === 0;
|
|
418
|
+
return okRun ? 0 : 1;
|
|
419
|
+
} finally {
|
|
420
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
421
|
+
}
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
// --analytics: emit the corpus through createRecorder into a temp segment
|
|
425
|
+
// root, rebuild the derived index from retained facts, and print the
|
|
426
|
+
// per-trace summaries + coverage denominators. Exit 0 when the rebuild is
|
|
427
|
+
// complete and equivalent to direct summarization of the corpus records.
|
|
428
|
+
async function runAnalyticsBench() {
|
|
429
|
+
const { createRecorder } = await import('../../src/core/observability/emit/recorder.js');
|
|
430
|
+
const { rebuildIndex } = await import('../../src/core/observability/analytics/rebuild.js');
|
|
431
|
+
const { summarizeTrace } = await import('../../src/core/observability/analytics/summary.js');
|
|
432
|
+
|
|
433
|
+
const corpus = JSON.parse(fs.readFileSync(corpusPath, 'utf8'));
|
|
434
|
+
const inputs = corpus.traces.flatMap((t) => t.records);
|
|
435
|
+
|
|
436
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ukit-bench-analytics-'));
|
|
437
|
+
try {
|
|
438
|
+
const recorder = createRecorder({
|
|
439
|
+
root,
|
|
440
|
+
config: { observability: { stage: 'default' } },
|
|
441
|
+
});
|
|
442
|
+
for (const record of inputs) recorder.emit(record);
|
|
443
|
+
const flush = await recorder.flush({ deadlineMs: 60_000 });
|
|
444
|
+
|
|
445
|
+
const started = process.hrtime.bigint();
|
|
446
|
+
const index = await rebuildIndex(root);
|
|
447
|
+
const rebuildMs = Number(process.hrtime.bigint() - started) / 1e6;
|
|
448
|
+
|
|
449
|
+
// Equivalence: rebuild-from-retained-facts must reproduce the same
|
|
450
|
+
// summaries as direct analysis of the corpus records (sanitization only
|
|
451
|
+
// rewrites sensitive free-text, never metric fields).
|
|
452
|
+
const direct = corpus.traces
|
|
453
|
+
.map((t) => summarizeTrace(t.records))
|
|
454
|
+
.sort((a, b) => a.trace_id.localeCompare(b.trace_id));
|
|
455
|
+
const equivalent = JSON.stringify(index.summaries) === JSON.stringify(direct);
|
|
456
|
+
|
|
457
|
+
const report = {
|
|
458
|
+
metric_version: METRIC_VERSION,
|
|
459
|
+
bench: 'analytics',
|
|
460
|
+
records: inputs.length,
|
|
461
|
+
flush_status: flush.status,
|
|
462
|
+
rebuild_ms: Number(rebuildMs.toFixed(3)),
|
|
463
|
+
equivalent_to_direct: equivalent,
|
|
464
|
+
coverage: index.coverage,
|
|
465
|
+
summaries: index.summaries,
|
|
466
|
+
};
|
|
467
|
+
process.stdout.write(`${JSON.stringify(report, null, 2)}\n`);
|
|
468
|
+
const okRun = flush.status === 'ok' && index.ok && equivalent
|
|
469
|
+
&& index.summaries.length === corpus.traces.length;
|
|
470
|
+
return okRun ? 0 : 1;
|
|
471
|
+
} finally {
|
|
472
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
473
|
+
}
|
|
474
|
+
}
|
|
475
|
+
|
|
476
|
+
// --evaluate: replay the golden case set under every pre-registered policy,
|
|
477
|
+
// print each Scorecard plus the paired comparison (fixed-seed bootstrap
|
|
478
|
+
// variance). Exit 0 when no cohort verdict is 'fail'; 1 on fail verdict or
|
|
479
|
+
// an unloadable corpus. Evaluation is offline — no model calls, no writes.
|
|
480
|
+
async function runEvaluateBench() {
|
|
481
|
+
const { evaluateVariant, compareScorecards } = await import(
|
|
482
|
+
'../../src/core/observability/evaluation/scorecard.js'
|
|
483
|
+
);
|
|
484
|
+
|
|
485
|
+
const goldenPath = path.join(repoRoot, 'tests/fixtures/observability/golden/cases.json');
|
|
486
|
+
let corpus;
|
|
487
|
+
try {
|
|
488
|
+
corpus = JSON.parse(fs.readFileSync(goldenPath, 'utf8'));
|
|
489
|
+
} catch (err) {
|
|
490
|
+
process.stderr.write(`--evaluate: cannot load ${goldenPath}: ${err && err.message}\n`);
|
|
491
|
+
return 1;
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
const policies = Object.keys(
|
|
495
|
+
(corpus.pre_registered && corpus.pre_registered.policies) || {},
|
|
496
|
+
).sort();
|
|
497
|
+
if (policies.length === 0) {
|
|
498
|
+
process.stderr.write('--evaluate: corpus declares no pre_registered.policies\n');
|
|
499
|
+
return 1;
|
|
500
|
+
}
|
|
501
|
+
|
|
502
|
+
const scorecards = {};
|
|
503
|
+
for (const policyId of policies) {
|
|
504
|
+
scorecards[policyId] = evaluateVariant(corpus, policyId);
|
|
505
|
+
}
|
|
506
|
+
const comparison = policies.length >= 2
|
|
507
|
+
? compareScorecards(scorecards[policies[0]], scorecards[policies[1]], { seed: SEED })
|
|
508
|
+
: null;
|
|
509
|
+
|
|
510
|
+
const report = {
|
|
511
|
+
metric_version: METRIC_VERSION,
|
|
512
|
+
bench: 'evaluate',
|
|
513
|
+
seed: SEED,
|
|
514
|
+
corpus: path.relative(repoRoot, goldenPath),
|
|
515
|
+
policies,
|
|
516
|
+
scorecards,
|
|
517
|
+
comparison,
|
|
518
|
+
};
|
|
519
|
+
process.stdout.write(`${JSON.stringify(report, null, 2)}\n`);
|
|
520
|
+
const failed = Object.values(scorecards).some((sc) => sc.verdict === 'fail');
|
|
521
|
+
return failed ? 1 : 0;
|
|
522
|
+
}
|
|
523
|
+
|
|
524
|
+
function main(argv) {
|
|
525
|
+
const flag = argv[2];
|
|
526
|
+
switch (flag) {
|
|
527
|
+
case '--fixture':
|
|
528
|
+
return writeFixture();
|
|
529
|
+
case '--recorder':
|
|
530
|
+
return runRecorderBench().then(
|
|
531
|
+
(code) => code,
|
|
532
|
+
(err) => {
|
|
533
|
+
process.stderr.write(`--recorder failed: ${err && err.message}\n`);
|
|
534
|
+
return 1;
|
|
535
|
+
},
|
|
536
|
+
);
|
|
537
|
+
case '--analytics':
|
|
538
|
+
return runAnalyticsBench().then(
|
|
539
|
+
(code) => code,
|
|
540
|
+
(err) => {
|
|
541
|
+
process.stderr.write(`--analytics failed: ${err && err.message}\n`);
|
|
542
|
+
return 1;
|
|
543
|
+
},
|
|
544
|
+
);
|
|
545
|
+
case '--evaluate':
|
|
546
|
+
return runEvaluateBench().then(
|
|
547
|
+
(code) => code,
|
|
548
|
+
(err) => {
|
|
549
|
+
process.stderr.write(`--evaluate failed: ${err && err.message}\n`);
|
|
550
|
+
return 1;
|
|
551
|
+
},
|
|
552
|
+
);
|
|
553
|
+
default:
|
|
554
|
+
process.stderr.write(
|
|
555
|
+
`usage: node scripts/bench/data-foundation.mjs --fixture|--recorder|--analytics|--evaluate\n`,
|
|
556
|
+
);
|
|
557
|
+
return 2;
|
|
558
|
+
}
|
|
559
|
+
}
|
|
560
|
+
|
|
561
|
+
Promise.resolve(main(process.argv)).then((code) => process.exit(code));
|
|
562
|
+
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
// common.js (TASK-010, SPEC §5 DF-FR07/DF-FR08) — shared internals for the
|
|
2
|
+
// provenance adapters. Adapters are read-only importers: they join safe
|
|
3
|
+
// metadata from existing owners (route audit, exec-ledger, decision
|
|
4
|
+
// receipts, context items) into semantic records. Every emitted record
|
|
5
|
+
// passes validateSemanticRecord + sanitizeObserved before returning — a
|
|
6
|
+
// record that fails either gate is dropped, never emitted dirty.
|
|
7
|
+
|
|
8
|
+
import crypto from 'node:crypto';
|
|
9
|
+
|
|
10
|
+
import { SCHEMA_VERSION } from '../schema/constants.js';
|
|
11
|
+
import { SEMANTIC_REGISTRY } from '../schema/registry.js';
|
|
12
|
+
import { validateSemanticRecord } from '../schema/validate.js';
|
|
13
|
+
import { sanitizeObserved } from '../privacy/sanitizeObserved.js';
|
|
14
|
+
|
|
15
|
+
export function isPlainObject(value) {
|
|
16
|
+
return value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
export function asString(value) {
|
|
20
|
+
return typeof value === 'string' && value.length > 0 ? value : null;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export function asStringArray(value) {
|
|
24
|
+
if (!Array.isArray(value)) return [];
|
|
25
|
+
return value.filter((v) => typeof v === 'string' && v.length > 0).slice(0, 32);
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
export function asIsoTime(value) {
|
|
29
|
+
if (typeof value === 'number' && Number.isFinite(value)) {
|
|
30
|
+
const d = new Date(value);
|
|
31
|
+
return Number.isNaN(d.getTime()) ? null : d.toISOString();
|
|
32
|
+
}
|
|
33
|
+
if (typeof value === 'string' && !Number.isNaN(Date.parse(value))) return value;
|
|
34
|
+
return null;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Per-adapt-call identity state. boot_id/writer_id/sequence give emitted
|
|
39
|
+
* records a stable, total-ordered provenance within one adaptation run;
|
|
40
|
+
* callers may inject fixed ids for deterministic tests/corpora.
|
|
41
|
+
*/
|
|
42
|
+
export function createAdapterContext({ writerId, bootId, clock } = {}) {
|
|
43
|
+
return {
|
|
44
|
+
boot_id: asString(bootId) ?? `boot-${crypto.randomUUID()}`,
|
|
45
|
+
writer_id: asString(writerId) ?? `adapter-${crypto.randomUUID()}`,
|
|
46
|
+
sequence: 0,
|
|
47
|
+
now: typeof clock === 'function' ? clock : () => new Date().toISOString(),
|
|
48
|
+
};
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Fill the envelope, validate, sanitize. Returns the sanitized record or
|
|
53
|
+
* null — adapters emit only records that survive the canonical gate.
|
|
54
|
+
*/
|
|
55
|
+
export function emitRecord(ctx, record) {
|
|
56
|
+
const out = { ...record };
|
|
57
|
+
if (out.record_type === undefined || out.record_type === null) out.record_type = 'fact';
|
|
58
|
+
if (out.schema_version === undefined || out.schema_version === null) {
|
|
59
|
+
out.schema_version = SCHEMA_VERSION;
|
|
60
|
+
}
|
|
61
|
+
if (!asString(out.record_id)) out.record_id = `rec-${crypto.randomUUID()}`;
|
|
62
|
+
out.boot_id = ctx.boot_id;
|
|
63
|
+
out.writer_id = ctx.writer_id;
|
|
64
|
+
out.sequence = ++ctx.sequence;
|
|
65
|
+
if (!asIsoTime(out.wall_time_utc)) out.wall_time_utc = ctx.now();
|
|
66
|
+
if (typeof out.importance !== 'string') out.importance = 'normal';
|
|
67
|
+
if (typeof out.privacy_class !== 'string') {
|
|
68
|
+
const entry = SEMANTIC_REGISTRY[out.semantic_name];
|
|
69
|
+
out.privacy_class = entry ? entry.privacy_class : 'internal';
|
|
70
|
+
}
|
|
71
|
+
const validation = validateSemanticRecord(out);
|
|
72
|
+
if (!validation.ok) return null;
|
|
73
|
+
const clean = sanitizeObserved(out);
|
|
74
|
+
return clean.ok ? clean.record : null;
|
|
75
|
+
}
|