@ngockhoale/ukit 3.0.12 → 3.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/README.md +1 -0
- package/manifests/documentation.yaml +12 -0
- package/package.json +1 -1
- package/scripts/bench/data-foundation.mjs +368 -50
- package/src/cli/commands/doctor.js +232 -3
- package/src/cli/commands/feedback.js +64 -1
- package/src/cli/commands/install.js +18 -0
- package/src/cli/commands/memory.js +42 -37
- package/src/cli/commands/telemetry.js +460 -0
- package/src/cli/index.js +7 -0
- package/src/core/agentRuntime/adapters.js +83 -2
- package/src/core/agentRuntime/diagnostics.js +104 -0
- package/src/core/agentRuntime/supervisor.js +137 -0
- package/src/core/agentRuntime/telemetry.js +204 -0
- package/src/core/memory/memoryEmit.js +131 -0
- package/src/core/memory/memoryHit.js +1 -1
- package/src/core/memory/migrate.js +18 -11
- package/src/core/memory/migrateMapping.js +15 -7
- package/src/core/memory/mutateMemory.js +22 -4
- package/src/core/memory/recordIndex.js +10 -3
- package/src/core/memory/recordStore.js +28 -3
- package/src/core/memory/retrieval.js +79 -38
- package/src/core/memory/store.js +37 -37
- package/src/core/memory/storeV2.js +28 -25
- package/src/core/memory/storeV2Loader.js +2 -2
- package/src/core/observability/adapters/ingest.js +576 -0
- package/src/core/observability/analytics/anomalies.js +415 -0
- package/src/core/observability/analytics/summary.js +16 -1
- package/src/core/observability/emit/config.js +69 -1
- package/src/core/observability/emit/crash.js +434 -0
- package/src/core/observability/emit/lifecycle.js +349 -0
- package/src/core/observability/emit/recorder.js +135 -9
- package/src/core/observability/evaluation/aiPacket.js +52 -10
- package/src/core/observability/evaluation/outcomes.js +95 -0
- package/src/core/observability/evaluation/runner.js +225 -0
- package/src/core/observability/privacy/allowlist.js +23 -3
- package/src/core/observability/schema/compatibility.js +48 -3
- package/src/core/observability/schema/constants.js +5 -0
- package/src/core/observability/schema/registry.js +57 -0
- package/src/core/observability/schema/validate.js +68 -6
- package/src/core/observability/segments/internal.js +42 -8
- package/src/core/observability/segments/readSegments.js +35 -1
- package/src/core/observability/segments/recovery.js +3 -2
- package/src/core/observability/segments/retention.js +137 -33
- package/src/core/observability/support/projector.js +88 -18
- package/src/core/observability/support/provision.js +160 -0
- package/src/core/observability/support/renderer.js +2 -2
- package/src/core/observability/support/schedule.js +174 -0
- package/template_project/.claude/hooks/auto-allow-bash.sh +7 -1
- package/template_project/.claude/hooks/auto-prune-bash.sh +16 -7
- package/template_project/.claude/hooks/verification-guard.sh +13 -4
- package/template_project/.claude/ukit/runtime/async-lock.mjs +26 -0
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,18 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to UKit are documented here.
|
|
4
4
|
|
|
5
|
+
## 3.1.0 - 2026-09-26
|
|
6
|
+
|
|
7
|
+
- **Flight recorder — Data Foundation wave (cycle C81)** — the observability pipeline grows from record/segment plumbing into a full local diagnostics surface; still opt-in behind `observability.stage` (default `off`, `off→shadow→canary→default` per seam, absolute kill switch):
|
|
8
|
+
- **Envelope v1.1 + sampling**: additive `agent_id`/`project_instance_id`/`sampling` stamp fields and new semantic names (`memory.write`, `memory.retrieved`, `retrieval.query`, `outcome.observed`, `anomaly.detected`, `telemetry.crashed`); `observability.sampling` seeded-bucket keep-rates per importance level with `SAMPLED_OUT`/`CONFIG_INVALID` reason codes — `critical` is never sampled.
|
|
9
|
+
- **Segments**: sealed segments compress to `.jsonl.zst` on Node ≥ 22.15 (plain `.jsonl` otherwise); readers, recovery, and quarantine handle both transparently.
|
|
10
|
+
- **Crash capture + lifecycle**: bounded `crashes/crash-<boot>-<seq>.json` capture on uncaughtException/unhandledRejection/SIGINT/SIGTERM (≤10 files, FIFO, never throws); `getRecorder` memoized per (root, boot) with interval flush and deadline-bounded exit flush.
|
|
11
|
+
- **Ingestion + instrumentation**: `ingestStoredTelemetry` pulls stored hook telemetry, route/decision ledgers, and memory/context items into segments; agentRuntime and memory/retrieval paths emit real trace/span ids with honest `UNKNOWN` resource provenance.
|
|
12
|
+
- **CLI**: `ukit telemetry` — `status`, `collect`, `digest`, `export-support`, `import <path>`, `evaluate` (`--json` on status/digest/evaluate; exit 0/2, usage 1). `ukit doctor` gains a flight-recorder section (stage, segments, support lag, crashes — `unknown` when unavailable). `ukit feedback` emits `outcome.observed`.
|
|
13
|
+
- **Support view**: `Documents/UKit Support` provisioned at install (never deleted by uninstall); support refresh is scheduled after collect and on the lifecycle interval, with crash list, anomaly section, and freshness lag stamped in `SUMMARY.md`; `import` validates bundles (manifest/checksums/traversal/decompression bounds).
|
|
14
|
+
- **Anomaly detection + evaluation**: deterministic detectors (duration outliers, retry storms, cache-miss/drop-rate spikes, verification-failure clusters) produce `anomaly.detected` records with evidence refs; `runEvaluation` dispatches to a configured `observability.evaluator` provider — `skipped/no-provider` is the honest default — and appends validated findings to the optimization KB; expanded golden corpus with `--anomalies` bench flag.
|
|
15
|
+
- **Docs**: `docs/OBSERVABILITY.md` — stages, privacy contract (what is never recorded), the support-folder zip-send flow, CLI and config reference.
|
|
16
|
+
|
|
5
17
|
## 3.0.12 - 2026-09-26
|
|
6
18
|
|
|
7
19
|
- Re-publish of the 3.0.11 content: npm staged-publish left 3.0.11 in limbo (E409 "previously staged version"); re-published as 3.0.12 — same tree, version bump only.
|
package/README.md
CHANGED
|
@@ -95,6 +95,7 @@ Maintainer/debug commands (advanced workflows, not team onboarding):
|
|
|
95
95
|
- `ukit memory ...` — inspect/export/forget shared memory items
|
|
96
96
|
- `ukit update` — upgrade the global UKit CLI to the latest published version
|
|
97
97
|
- `ukit index ...` — run repo indexing/query/triage tools directly
|
|
98
|
+
- `ukit telemetry ...` — inspect the opt-in local flight recorder (segments, digests, `Documents/UKit Support` export) — see [docs/OBSERVABILITY.md](docs/OBSERVABILITY.md)
|
|
98
99
|
- `ukit build index` — alias for `ukit index build`
|
|
99
100
|
|
|
100
101
|
## Development
|
|
@@ -296,6 +296,18 @@ entries:
|
|
|
296
296
|
archive_policy: date-rotate
|
|
297
297
|
budget: { max_lines: 150, enforcement: error }
|
|
298
298
|
notes: enforcement flipped to error by TASK-005 (2026-09-19); TASK-217 audit — actual 83 lines (~55% headroom), value kept
|
|
299
|
+
- id: docs-observability
|
|
300
|
+
path: docs/OBSERVABILITY.md
|
|
301
|
+
class: canonical
|
|
302
|
+
audience: [agent, maintainer]
|
|
303
|
+
owner: product
|
|
304
|
+
source_of_truth: docs/OBSERVABILITY.md
|
|
305
|
+
merge_strategy: none
|
|
306
|
+
load_policy: on-demand
|
|
307
|
+
validation: [manual]
|
|
308
|
+
archive_policy: never
|
|
309
|
+
notes: 'registered by C81 fix round 1 — Data Foundation/Flight Recorder surface doc'
|
|
310
|
+
|
|
299
311
|
- id: docs-worklog
|
|
300
312
|
path: docs/WORKLOG.md
|
|
301
313
|
class: runtime
|
package/package.json
CHANGED
|
@@ -9,6 +9,15 @@
|
|
|
9
9
|
// a fixed-seed variance report (TASK-014).
|
|
10
10
|
// --perturb Inject seeded synthetic mutations into corpus summaries and measure
|
|
11
11
|
// detection through computeFingerprint/detectOpportunities (TASK-002).
|
|
12
|
+
// --anomalies Replay the golden cases, summarize each run, and report anomaly
|
|
13
|
+
// detection counts per case (TASK-015). Degrades honestly to
|
|
14
|
+
// `detector: 'unavailable'` while analytics/anomalies.js
|
|
15
|
+
// (TASK-011) does not exist — never a stub detector.
|
|
16
|
+
//
|
|
17
|
+
// Flags run in argument order; the exit code is the first non-zero code.
|
|
18
|
+
// --evaluate exits 1 by design while the corpus carries expected-negative
|
|
19
|
+
// strata (a fail verdict is the honest scorecard), so a combined invocation
|
|
20
|
+
// ending in --evaluate exits 1.
|
|
12
21
|
//
|
|
13
22
|
// The corpus is a synthetic, seeded fixture: no real user content. Records are
|
|
14
23
|
// envelope-shaped per SPEC §7 (DF-FR01) but intentionally NOT schema-validated —
|
|
@@ -77,10 +86,9 @@ function makeCtx() {
|
|
|
77
86
|
tokens: (lo, hi) => lo + Math.floor(rand() * (hi - lo + 1)),
|
|
78
87
|
};
|
|
79
88
|
}
|
|
80
|
-
|
|
81
|
-
function record(ctx, { semanticName, traceId, spanId, parentSpanId, executionId, importance, privacyClass, payload, dtMs }) {
|
|
89
|
+
function record(ctx, { semanticName, traceId, spanId, parentSpanId, executionId, agentId, importance, privacyClass, payload, dtMs }) {
|
|
82
90
|
const t = ctx.tick(dtMs);
|
|
83
|
-
|
|
91
|
+
const rec = {
|
|
84
92
|
record_type: 'fact',
|
|
85
93
|
semantic_name: semanticName,
|
|
86
94
|
schema_version: SCHEMA_VERSION,
|
|
@@ -100,9 +108,13 @@ function record(ctx, { semanticName, traceId, spanId, parentSpanId, executionId,
|
|
|
100
108
|
privacy_class: clampToRegistryFloor(semanticName, privacyClass),
|
|
101
109
|
payload,
|
|
102
110
|
};
|
|
111
|
+
// DF2-FR01 v1.1: agent_id is the optional multi-agent correlation field —
|
|
112
|
+
// emitted only when the scenario declares one.
|
|
113
|
+
if (typeof agentId === 'string' && agentId.length > 0) rec.agent_id = agentId;
|
|
114
|
+
return rec;
|
|
103
115
|
}
|
|
104
116
|
|
|
105
|
-
function execSpan(ctx, traceId, executionId, privacyClass) {
|
|
117
|
+
function execSpan(ctx, traceId, executionId, privacyClass, agentId) {
|
|
106
118
|
const spanId = ctx.spanId();
|
|
107
119
|
return {
|
|
108
120
|
spanId,
|
|
@@ -112,6 +124,7 @@ function execSpan(ctx, traceId, executionId, privacyClass) {
|
|
|
112
124
|
spanId,
|
|
113
125
|
parentSpanId: null,
|
|
114
126
|
executionId,
|
|
127
|
+
agentId,
|
|
115
128
|
importance: 'normal',
|
|
116
129
|
privacyClass,
|
|
117
130
|
payload: { operation: 'task', duration_ms: null },
|
|
@@ -120,7 +133,7 @@ function execSpan(ctx, traceId, executionId, privacyClass) {
|
|
|
120
133
|
};
|
|
121
134
|
}
|
|
122
135
|
|
|
123
|
-
function modelAttempt(ctx, { traceId, parentSpanId, executionId, attemptIndex, status, resource, privacyClass, extra = {} }) {
|
|
136
|
+
function modelAttempt(ctx, { traceId, parentSpanId, executionId, agentId, attemptIndex, status, resource, privacyClass, extra = {} }) {
|
|
124
137
|
const spanId = ctx.spanId();
|
|
125
138
|
const started = record(ctx, {
|
|
126
139
|
semanticName: 'model.started',
|
|
@@ -128,6 +141,7 @@ function modelAttempt(ctx, { traceId, parentSpanId, executionId, attemptIndex, s
|
|
|
128
141
|
spanId,
|
|
129
142
|
parentSpanId,
|
|
130
143
|
executionId,
|
|
144
|
+
agentId,
|
|
131
145
|
importance: 'normal',
|
|
132
146
|
privacyClass,
|
|
133
147
|
payload: { attempt_index: attemptIndex, ...extra },
|
|
@@ -139,6 +153,7 @@ function modelAttempt(ctx, { traceId, parentSpanId, executionId, attemptIndex, s
|
|
|
139
153
|
spanId,
|
|
140
154
|
parentSpanId,
|
|
141
155
|
executionId,
|
|
156
|
+
agentId,
|
|
142
157
|
importance: 'normal',
|
|
143
158
|
privacyClass,
|
|
144
159
|
payload: {
|
|
@@ -152,14 +167,14 @@ function modelAttempt(ctx, { traceId, parentSpanId, executionId, attemptIndex, s
|
|
|
152
167
|
});
|
|
153
168
|
return [started, completed];
|
|
154
169
|
}
|
|
155
|
-
|
|
156
|
-
function execEnd(ctx, { traceId, spanId, executionId, status, privacyClass, extra = {} }) {
|
|
170
|
+
function execEnd(ctx, { traceId, spanId, executionId, agentId, status, privacyClass, extra = {} }) {
|
|
157
171
|
return record(ctx, {
|
|
158
172
|
semanticName: `execution.${status}`,
|
|
159
173
|
traceId,
|
|
160
174
|
spanId,
|
|
161
175
|
parentSpanId: null,
|
|
162
176
|
executionId,
|
|
177
|
+
agentId,
|
|
163
178
|
importance: 'normal',
|
|
164
179
|
privacyClass,
|
|
165
180
|
payload: { duration_ms: 0, ...extra },
|
|
@@ -174,11 +189,10 @@ function sumUsage(attempts) {
|
|
|
174
189
|
.map((res) => res.value);
|
|
175
190
|
return values.length ? values.reduce((a, b) => a + b, 0) : null;
|
|
176
191
|
}
|
|
177
|
-
|
|
178
|
-
function buildTrace(ctx, { scenario, privacyClass, build }) {
|
|
192
|
+
function buildTrace(ctx, { scenario, privacyClass, agentId, build }) {
|
|
179
193
|
const traceId = ctx.traceId();
|
|
180
194
|
const executionId = `exec-${traceId}`;
|
|
181
|
-
const exec = execSpan(ctx, traceId, executionId, privacyClass);
|
|
195
|
+
const exec = execSpan(ctx, traceId, executionId, privacyClass, agentId);
|
|
182
196
|
const { records, attempts, succeeded } = build({ traceId, executionId, execSpanId: exec.spanId });
|
|
183
197
|
const all = [exec.started, ...records];
|
|
184
198
|
return {
|
|
@@ -323,6 +337,109 @@ function buildCorpus() {
|
|
|
323
337
|
}),
|
|
324
338
|
);
|
|
325
339
|
|
|
340
|
+
// memory_outcome: retrieval.query → memory.retrieved hits → model →
|
|
341
|
+
// memory.write → execution.completed → outcome.observed — metadata-only
|
|
342
|
+
// payloads (counts, kinds, opaque refs), never content.
|
|
343
|
+
traces.push(
|
|
344
|
+
buildTrace(ctx, {
|
|
345
|
+
scenario: 'memory_outcome',
|
|
346
|
+
privacyClass: 'sensitive',
|
|
347
|
+
build: ({ traceId, executionId, execSpanId }) => {
|
|
348
|
+
const query = record(ctx, {
|
|
349
|
+
semanticName: 'retrieval.query',
|
|
350
|
+
traceId, spanId: ctx.spanId(), parentSpanId: execSpanId, executionId,
|
|
351
|
+
importance: 'normal', privacyClass: 'internal',
|
|
352
|
+
payload: {
|
|
353
|
+
lane: 'v2', scope: 'all', indexed: true,
|
|
354
|
+
candidate_count: ctx.tokens(10, 40), returned_count: 3,
|
|
355
|
+
duration_ms: ctx.tokens(5, 20), freshness: {},
|
|
356
|
+
},
|
|
357
|
+
dtMs: 8,
|
|
358
|
+
});
|
|
359
|
+
const retrieved = [
|
|
360
|
+
{ kind: 'fact' }, { kind: 'decision' }, { kind: 'decision' },
|
|
361
|
+
].map((hit) => record(ctx, {
|
|
362
|
+
semanticName: 'memory.retrieved',
|
|
363
|
+
traceId, spanId: ctx.spanId(), parentSpanId: execSpanId, executionId,
|
|
364
|
+
importance: 'normal', privacyClass: 'sensitive',
|
|
365
|
+
payload: { kind: hit.kind, record_ref: `memref-${ctx.tokens(1000, 9999)}` },
|
|
366
|
+
dtMs: 4,
|
|
367
|
+
}));
|
|
368
|
+
const resource = { value: ctx.tokens(800, 1200), source: 'PROVIDER' };
|
|
369
|
+
const attempts = modelAttempt(ctx, {
|
|
370
|
+
traceId, parentSpanId: execSpanId, executionId,
|
|
371
|
+
attemptIndex: 1, status: 'ok', resource, privacyClass: 'public',
|
|
372
|
+
});
|
|
373
|
+
const write = record(ctx, {
|
|
374
|
+
semanticName: 'memory.write',
|
|
375
|
+
traceId, spanId: ctx.spanId(), parentSpanId: execSpanId, executionId,
|
|
376
|
+
importance: 'normal', privacyClass: 'sensitive',
|
|
377
|
+
payload: {
|
|
378
|
+
op: 'write', record_count: ctx.tokens(1, 5), changed_count: 1,
|
|
379
|
+
kinds: { fact: 1 }, record_refs: [`memref-${ctx.tokens(1000, 9999)}`],
|
|
380
|
+
tombstone_count: 0,
|
|
381
|
+
},
|
|
382
|
+
dtMs: 6,
|
|
383
|
+
});
|
|
384
|
+
const end = execEnd(ctx, {
|
|
385
|
+
traceId, spanId: execSpanId, executionId, status: 'completed',
|
|
386
|
+
privacyClass: 'public', extra: { outcome: 'success' },
|
|
387
|
+
});
|
|
388
|
+
const outcome = record(ctx, {
|
|
389
|
+
semanticName: 'outcome.observed',
|
|
390
|
+
traceId, spanId: ctx.spanId(), parentSpanId: execSpanId, executionId,
|
|
391
|
+
importance: 'high', privacyClass: 'internal',
|
|
392
|
+
payload: {
|
|
393
|
+
kind: 'feedback', verdict: 'accepted', source: 'user',
|
|
394
|
+
evidence_refs: [end.record_id],
|
|
395
|
+
},
|
|
396
|
+
dtMs: 30,
|
|
397
|
+
});
|
|
398
|
+
return {
|
|
399
|
+
records: [query, ...retrieved, ...attempts, write, end, outcome],
|
|
400
|
+
attempts,
|
|
401
|
+
succeeded: true,
|
|
402
|
+
};
|
|
403
|
+
},
|
|
404
|
+
}),
|
|
405
|
+
);
|
|
406
|
+
|
|
407
|
+
// multi_agent: one trace spanning two agents — orchestrator runs the
|
|
408
|
+
// retrieval query and owns the execution span, worker runs the model
|
|
409
|
+
// attempt. agent_id is the v1.1 correlation field; trace_id stays single.
|
|
410
|
+
traces.push(
|
|
411
|
+
buildTrace(ctx, {
|
|
412
|
+
scenario: 'multi_agent',
|
|
413
|
+
privacyClass: 'public',
|
|
414
|
+
agentId: 'agent-orchestrator',
|
|
415
|
+
build: ({ traceId, executionId, execSpanId }) => {
|
|
416
|
+
const query = record(ctx, {
|
|
417
|
+
semanticName: 'retrieval.query',
|
|
418
|
+
traceId, spanId: ctx.spanId(), parentSpanId: execSpanId, executionId,
|
|
419
|
+
agentId: 'agent-orchestrator',
|
|
420
|
+
importance: 'normal', privacyClass: 'internal',
|
|
421
|
+
payload: {
|
|
422
|
+
lane: 'v2', scope: 'all', indexed: true,
|
|
423
|
+
candidate_count: ctx.tokens(10, 40), returned_count: 2,
|
|
424
|
+
duration_ms: ctx.tokens(5, 20), freshness: {},
|
|
425
|
+
},
|
|
426
|
+
dtMs: 8,
|
|
427
|
+
});
|
|
428
|
+
const resource = { value: ctx.tokens(800, 1200), source: 'PROVIDER' };
|
|
429
|
+
const attempts = modelAttempt(ctx, {
|
|
430
|
+
traceId, parentSpanId: execSpanId, executionId,
|
|
431
|
+
agentId: 'agent-worker',
|
|
432
|
+
attemptIndex: 1, status: 'ok', resource, privacyClass: 'public',
|
|
433
|
+
});
|
|
434
|
+
const end = execEnd(ctx, {
|
|
435
|
+
traceId, spanId: execSpanId, executionId, agentId: 'agent-orchestrator',
|
|
436
|
+
status: 'completed', privacyClass: 'public', extra: { outcome: 'success' },
|
|
437
|
+
});
|
|
438
|
+
return { records: [query, ...attempts, end], attempts, succeeded: true };
|
|
439
|
+
},
|
|
440
|
+
}),
|
|
441
|
+
);
|
|
442
|
+
|
|
326
443
|
return {
|
|
327
444
|
metric_version: METRIC_VERSION,
|
|
328
445
|
seed: SEED,
|
|
@@ -560,52 +677,253 @@ export async function runPerturb({ quiet = false } = {}) {
|
|
|
560
677
|
return { code: outcome.missed.length > 0 ? 1 : 0, report };
|
|
561
678
|
}
|
|
562
679
|
|
|
563
|
-
|
|
564
|
-
|
|
680
|
+
// --anomalies: replay every golden case under each pre-registered policy,
|
|
681
|
+
// summarize each materialized run, and report per-case anomaly detection
|
|
682
|
+
// counts keyed by detector kind. The detector is imported defensively:
|
|
683
|
+
// while src/core/observability/analytics/anomalies.js (TASK-011) does not
|
|
684
|
+
// exist the report degrades to `detector: 'unavailable'` — never a stub.
|
|
685
|
+
// When a detector IS available, per-case declarations in
|
|
686
|
+
// `expect.anomalies.expected_kinds` act as a regression gate: unmet or
|
|
687
|
+
// over-fired kinds exit 1. Evaluation is offline — no model calls.
|
|
688
|
+
async function loadAnomalyDetector() {
|
|
689
|
+
try {
|
|
690
|
+
const mod = await import('../../src/core/observability/analytics/anomalies.js');
|
|
691
|
+
return typeof mod.detectAnomalies === 'function' ? mod.detectAnomalies : null;
|
|
692
|
+
} catch {
|
|
693
|
+
return null;
|
|
694
|
+
}
|
|
695
|
+
}
|
|
696
|
+
|
|
697
|
+
function anomalyKind(record) {
|
|
698
|
+
const kind = record && record.payload && record.payload.kind !== undefined
|
|
699
|
+
? record.payload.kind
|
|
700
|
+
: record && record.kind;
|
|
701
|
+
return typeof kind === 'string' && kind.length > 0 ? kind : 'unknown';
|
|
702
|
+
}
|
|
703
|
+
|
|
704
|
+
async function runAnomaliesBench() {
|
|
705
|
+
const { replayCase } = await import('../../src/core/observability/evaluation/replay.js');
|
|
706
|
+
const { summarizeTrace } = await import('../../src/core/observability/analytics/summary.js');
|
|
707
|
+
const detectAnomalies = await loadAnomalyDetector();
|
|
708
|
+
|
|
709
|
+
const goldenPath = path.join(repoRoot, 'tests/fixtures/observability/golden/cases.json');
|
|
710
|
+
let corpus;
|
|
711
|
+
try {
|
|
712
|
+
corpus = JSON.parse(fs.readFileSync(goldenPath, 'utf8'));
|
|
713
|
+
} catch (err) {
|
|
714
|
+
process.stderr.write(`--anomalies: cannot load ${goldenPath}: ${err && err.message}\n`);
|
|
715
|
+
return 1;
|
|
716
|
+
}
|
|
717
|
+
const policies = Object.keys(
|
|
718
|
+
(corpus.pre_registered && corpus.pre_registered.policies) || {},
|
|
719
|
+
).sort();
|
|
720
|
+
if (policies.length === 0) {
|
|
721
|
+
process.stderr.write('--anomalies: corpus declares no pre_registered.policies\n');
|
|
722
|
+
return 1;
|
|
723
|
+
}
|
|
724
|
+
|
|
725
|
+
const runs = [];
|
|
726
|
+
for (const goldenCase of corpus.cases || []) {
|
|
727
|
+
const expectedKinds = goldenCase && goldenCase.expect && goldenCase.expect.anomalies
|
|
728
|
+
&& Array.isArray(goldenCase.expect.anomalies.expected_kinds)
|
|
729
|
+
? goldenCase.expect.anomalies.expected_kinds
|
|
730
|
+
: null;
|
|
731
|
+
const run = { case_id: goldenCase && goldenCase.case_id, expected_kinds: expectedKinds, policies: {} };
|
|
732
|
+
for (const policyId of policies) {
|
|
733
|
+
const replay = replayCase(goldenCase, { policyId });
|
|
734
|
+
if (!replay.ok) {
|
|
735
|
+
run.policies[policyId] = { ok: false, reason: replay.reason };
|
|
736
|
+
continue;
|
|
737
|
+
}
|
|
738
|
+
const summary = summarizeTrace(replay.records);
|
|
739
|
+
run.policies[policyId] = {
|
|
740
|
+
ok: true,
|
|
741
|
+
trace_id: summary.trace_id,
|
|
742
|
+
summary,
|
|
743
|
+
record_ids: replay.records.map((r) => r.record_id),
|
|
744
|
+
};
|
|
745
|
+
}
|
|
746
|
+
runs.push(run);
|
|
747
|
+
}
|
|
748
|
+
|
|
749
|
+
// Detector wiring: one call over ALL materialized summaries (detectors are
|
|
750
|
+
// cohort-relative), then attribution back to cases by trace_id or any
|
|
751
|
+
// evidence_ref that resolves to a materialized record_id.
|
|
752
|
+
const summaries = runs.flatMap((run) => policies
|
|
753
|
+
.map((policyId) => run.policies[policyId])
|
|
754
|
+
.filter((entry) => entry && entry.ok)
|
|
755
|
+
.map((entry) => entry.summary));
|
|
756
|
+
|
|
757
|
+
const byCase = new Map(runs.map((run) => [run.case_id, new Map()]));
|
|
758
|
+
const byTrace = new Map();
|
|
759
|
+
const byRecordId = new Map();
|
|
760
|
+
for (const run of runs) {
|
|
761
|
+
for (const policyId of policies) {
|
|
762
|
+
const entry = run.policies[policyId];
|
|
763
|
+
if (!entry || !entry.ok) continue;
|
|
764
|
+
if (typeof entry.trace_id === 'string') byTrace.set(entry.trace_id, [run.case_id, policyId]);
|
|
765
|
+
for (const recordId of entry.record_ids) byRecordId.set(recordId, [run.case_id, policyId]);
|
|
766
|
+
}
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
let detection = null;
|
|
770
|
+
if (detectAnomalies) {
|
|
771
|
+
const anomalies = detectAnomalies(summaries) || [];
|
|
772
|
+
const unattributed = [];
|
|
773
|
+
for (const anomaly of anomalies) {
|
|
774
|
+
const refs = anomaly && anomaly.payload && Array.isArray(anomaly.payload.evidence_refs)
|
|
775
|
+
? anomaly.payload.evidence_refs
|
|
776
|
+
: [];
|
|
777
|
+
const traceRef = anomaly && anomaly.payload && typeof anomaly.payload.trace_ref === 'string'
|
|
778
|
+
? anomaly.payload.trace_ref
|
|
779
|
+
: (anomaly && typeof anomaly.trace_id === 'string' ? anomaly.trace_id : null);
|
|
780
|
+
let target = traceRef ? byTrace.get(traceRef) : null;
|
|
781
|
+
if (!target) {
|
|
782
|
+
for (const ref of refs) {
|
|
783
|
+
target = byRecordId.get(ref);
|
|
784
|
+
if (target) break;
|
|
785
|
+
}
|
|
786
|
+
}
|
|
787
|
+
if (!target) { unattributed.push(anomaly); continue; }
|
|
788
|
+
const [caseId, policyId] = target;
|
|
789
|
+
const policyMap = byCase.get(caseId);
|
|
790
|
+
if (!policyMap.has(policyId)) policyMap.set(policyId, {});
|
|
791
|
+
const counts = policyMap.get(policyId);
|
|
792
|
+
const kind = anomalyKind(anomaly);
|
|
793
|
+
counts[kind] = (counts[kind] || 0) + 1;
|
|
794
|
+
}
|
|
795
|
+
detection = { anomalies, unattributed };
|
|
796
|
+
}
|
|
797
|
+
// Per-case signals — the summary fields the detectors consume (TASK-011):
|
|
798
|
+
// attempt/retry counters, cache posture, drops, critical path. Reported
|
|
799
|
+
// per policy run so a detector regression is attributable to a case.
|
|
800
|
+
function signalsOf(entry) {
|
|
801
|
+
const s = entry && entry.summary;
|
|
802
|
+
if (!s) return null;
|
|
803
|
+
return {
|
|
804
|
+
model_attempts: s.retries.model_attempts,
|
|
805
|
+
retries: s.retries.retries,
|
|
806
|
+
failed_spans: s.retries.failed_spans,
|
|
807
|
+
cache_misses: s.cache.misses,
|
|
808
|
+
cache_hits: s.cache.hits,
|
|
809
|
+
hit_rate: s.cache.hit_rate,
|
|
810
|
+
dropped_events: s.drops.events,
|
|
811
|
+
dropped_count: s.drops.dropped_count,
|
|
812
|
+
critical_path_ms: s.critical_path_ms,
|
|
813
|
+
telemetry_complete: s.telemetry_complete,
|
|
814
|
+
spans: s.coverage.spans,
|
|
815
|
+
};
|
|
816
|
+
}
|
|
817
|
+
|
|
818
|
+
const casesReport = {};
|
|
819
|
+
for (const run of runs) {
|
|
820
|
+
const signals = {};
|
|
821
|
+
for (const policyId of policies) {
|
|
822
|
+
signals[policyId] = signalsOf(run.policies[policyId]);
|
|
823
|
+
}
|
|
824
|
+
let anomalies = null;
|
|
825
|
+
if (detection) {
|
|
826
|
+
anomalies = 0;
|
|
827
|
+
for (const counts of byCase.get(run.case_id)?.values() || []) {
|
|
828
|
+
for (const n of Object.values(counts)) anomalies += n;
|
|
829
|
+
}
|
|
830
|
+
}
|
|
831
|
+
casesReport[run.case_id] = { signals, anomalies };
|
|
832
|
+
}
|
|
833
|
+
|
|
834
|
+
// Regression gate — only meaningful when a real detector ran. A case that
|
|
835
|
+
// declares expect.anomalies.expected_kinds must see those kinds fire; a
|
|
836
|
+
// case declaring an empty list must stay clean.
|
|
837
|
+
const unmet = [];
|
|
838
|
+
if (detection) {
|
|
839
|
+
for (const run of runs) {
|
|
840
|
+
if (!Array.isArray(run.expected_kinds)) continue;
|
|
841
|
+
for (const [policyId, counts] of byCase.get(run.case_id) || []) {
|
|
842
|
+
for (const kind of run.expected_kinds) {
|
|
843
|
+
if (!counts[kind]) {
|
|
844
|
+
unmet.push({ case_id: run.case_id, policy: policyId, kind, expected: '>=1', got: 0 });
|
|
845
|
+
}
|
|
846
|
+
}
|
|
847
|
+
for (const kind of Object.keys(counts)) {
|
|
848
|
+
if (!run.expected_kinds.includes(kind)) {
|
|
849
|
+
unmet.push({ case_id: run.case_id, policy: policyId, kind, expected: 'absent', got: counts[kind] });
|
|
850
|
+
}
|
|
851
|
+
}
|
|
852
|
+
}
|
|
853
|
+
}
|
|
854
|
+
unmet.sort((a, b) => String(a.case_id).localeCompare(String(b.case_id))
|
|
855
|
+
|| a.policy.localeCompare(b.policy) || a.kind.localeCompare(b.kind));
|
|
856
|
+
}
|
|
857
|
+
|
|
858
|
+
const report = {
|
|
859
|
+
metric_version: METRIC_VERSION,
|
|
860
|
+
bench: 'anomalies',
|
|
861
|
+
seed: SEED,
|
|
862
|
+
corpus: path.relative(repoRoot, goldenPath),
|
|
863
|
+
policies,
|
|
864
|
+
detector: detectAnomalies ? 'detectAnomalies' : 'unavailable',
|
|
865
|
+
summaries: summaries.length,
|
|
866
|
+
anomalies_total: detection ? detection.anomalies.length : null,
|
|
867
|
+
unattributed: detection ? detection.unattributed.length : null,
|
|
868
|
+
unmet_expectations: detection ? unmet : null,
|
|
869
|
+
cases: casesReport,
|
|
870
|
+
};
|
|
871
|
+
process.stdout.write(`${JSON.stringify(report, null, 2)}\n`);
|
|
872
|
+
if (!detection) return 0; // degraded — nothing to gate on
|
|
873
|
+
return unmet.length > 0 ? 1 : 0;
|
|
874
|
+
}
|
|
875
|
+
|
|
876
|
+
const BENCH_FLAGS = Object.freeze([
|
|
877
|
+
'--fixture',
|
|
878
|
+
'--recorder',
|
|
879
|
+
'--analytics',
|
|
880
|
+
'--evaluate',
|
|
881
|
+
'--perturb',
|
|
882
|
+
'--anomalies',
|
|
883
|
+
]);
|
|
884
|
+
|
|
885
|
+
async function runFlag(flag) {
|
|
565
886
|
switch (flag) {
|
|
566
|
-
case '--fixture':
|
|
567
|
-
|
|
568
|
-
case '--
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
process.stderr.write(`--recorder failed: ${err && err.message}\n`);
|
|
573
|
-
return 1;
|
|
574
|
-
},
|
|
575
|
-
);
|
|
576
|
-
case '--analytics':
|
|
577
|
-
return runAnalyticsBench().then(
|
|
578
|
-
(code) => code,
|
|
579
|
-
(err) => {
|
|
580
|
-
process.stderr.write(`--analytics failed: ${err && err.message}\n`);
|
|
581
|
-
return 1;
|
|
582
|
-
},
|
|
583
|
-
);
|
|
584
|
-
case '--evaluate':
|
|
585
|
-
return runEvaluateBench().then(
|
|
586
|
-
(code) => code,
|
|
587
|
-
(err) => {
|
|
588
|
-
process.stderr.write(`--evaluate failed: ${err && err.message}\n`);
|
|
589
|
-
return 1;
|
|
590
|
-
},
|
|
591
|
-
);
|
|
592
|
-
case '--perturb':
|
|
593
|
-
return runPerturb().then(
|
|
594
|
-
({ code }) => code,
|
|
595
|
-
(err) => {
|
|
596
|
-
process.stderr.write(`--perturb failed: ${err && err.message}\n`);
|
|
597
|
-
return 1;
|
|
598
|
-
},
|
|
599
|
-
);
|
|
887
|
+
case '--fixture': return writeFixture();
|
|
888
|
+
case '--recorder': return runRecorderBench();
|
|
889
|
+
case '--analytics': return runAnalyticsBench();
|
|
890
|
+
case '--evaluate': return runEvaluateBench();
|
|
891
|
+
case '--perturb': return runPerturb().then(({ code }) => code);
|
|
892
|
+
case '--anomalies': return runAnomaliesBench();
|
|
600
893
|
default:
|
|
601
|
-
process.stderr.write(
|
|
602
|
-
`usage: node scripts/bench/data-foundation.mjs --fixture|--recorder|--analytics|--evaluate|--perturb\n`,
|
|
603
|
-
);
|
|
894
|
+
process.stderr.write(`unknown flag: ${flag}\n`);
|
|
604
895
|
return 2;
|
|
605
896
|
}
|
|
606
897
|
}
|
|
898
|
+
|
|
899
|
+
// Flags run in argument order; the exit code is the first non-zero result.
|
|
900
|
+
// (PLAN §5 verification chains --recorder --analytics --evaluate.)
|
|
901
|
+
async function main(argv) {
|
|
902
|
+
const flags = argv.slice(2).filter((arg) => arg.startsWith('--'));
|
|
903
|
+
if (flags.length === 0) {
|
|
904
|
+
process.stderr.write(
|
|
905
|
+
`usage: node scripts/bench/data-foundation.mjs ${BENCH_FLAGS.join('|')}\n`,
|
|
906
|
+
);
|
|
907
|
+
return 2;
|
|
908
|
+
}
|
|
909
|
+
let firstFailure = 0;
|
|
910
|
+
for (const flag of flags) {
|
|
911
|
+
try {
|
|
912
|
+
const code = await runFlag(flag);
|
|
913
|
+
if (code !== 0 && firstFailure === 0) firstFailure = code;
|
|
914
|
+
} catch (err) {
|
|
915
|
+
process.stderr.write(`${flag} failed: ${err && err.message}\n`);
|
|
916
|
+
if (firstFailure === 0) firstFailure = 1;
|
|
917
|
+
}
|
|
918
|
+
}
|
|
919
|
+
return firstFailure;
|
|
920
|
+
}
|
|
607
921
|
const invokedAs = process.argv[1] ? path.resolve(process.argv[1]) : '';
|
|
608
922
|
if (invokedAs === fileURLToPath(import.meta.url)) {
|
|
609
|
-
|
|
923
|
+
// Set the code then let the loop drain stdout — process.exit() hard-kills
|
|
924
|
+
// pending pipe writes, which truncates reports past the 64 KiB buffer.
|
|
925
|
+
Promise.resolve(main(process.argv)).then((code) => {
|
|
926
|
+
process.exitCode = code;
|
|
927
|
+
});
|
|
610
928
|
}
|
|
611
929
|
|