@ngockhoale/ukit 3.0.12 → 3.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. package/CHANGELOG.md +21 -0
  2. package/README.md +1 -0
  3. package/manifests/documentation.yaml +12 -0
  4. package/manifests/platform.full.yaml +24 -0
  5. package/package.json +1 -1
  6. package/scripts/bench/data-foundation.mjs +368 -50
  7. package/src/cli/commands/doctor.js +232 -3
  8. package/src/cli/commands/feedback.js +64 -1
  9. package/src/cli/commands/install.js +18 -0
  10. package/src/cli/commands/memory.js +42 -37
  11. package/src/cli/commands/telemetry.js +460 -0
  12. package/src/cli/index.js +7 -0
  13. package/src/core/agentRuntime/adapters.js +83 -2
  14. package/src/core/agentRuntime/diagnostics.js +104 -0
  15. package/src/core/agentRuntime/supervisor.js +137 -0
  16. package/src/core/agentRuntime/telemetry.js +204 -0
  17. package/src/core/memory/memoryEmit.js +131 -0
  18. package/src/core/memory/memoryHit.js +1 -1
  19. package/src/core/memory/migrate.js +18 -11
  20. package/src/core/memory/migrateMapping.js +15 -7
  21. package/src/core/memory/mutateMemory.js +22 -4
  22. package/src/core/memory/recordIndex.js +10 -3
  23. package/src/core/memory/recordStore.js +28 -3
  24. package/src/core/memory/retrieval.js +79 -38
  25. package/src/core/memory/store.js +37 -37
  26. package/src/core/memory/storeV2.js +28 -25
  27. package/src/core/memory/storeV2Loader.js +2 -2
  28. package/src/core/observability/adapters/ingest.js +576 -0
  29. package/src/core/observability/analytics/anomalies.js +415 -0
  30. package/src/core/observability/analytics/summary.js +16 -1
  31. package/src/core/observability/emit/config.js +69 -1
  32. package/src/core/observability/emit/crash.js +434 -0
  33. package/src/core/observability/emit/lifecycle.js +349 -0
  34. package/src/core/observability/emit/recorder.js +135 -9
  35. package/src/core/observability/evaluation/aiPacket.js +52 -10
  36. package/src/core/observability/evaluation/outcomes.js +95 -0
  37. package/src/core/observability/evaluation/runner.js +225 -0
  38. package/src/core/observability/privacy/allowlist.js +23 -3
  39. package/src/core/observability/schema/compatibility.js +48 -3
  40. package/src/core/observability/schema/constants.js +5 -0
  41. package/src/core/observability/schema/registry.js +57 -0
  42. package/src/core/observability/schema/validate.js +68 -6
  43. package/src/core/observability/segments/internal.js +42 -8
  44. package/src/core/observability/segments/readSegments.js +35 -1
  45. package/src/core/observability/segments/recovery.js +3 -2
  46. package/src/core/observability/segments/retention.js +137 -33
  47. package/src/core/observability/support/projector.js +88 -18
  48. package/src/core/observability/support/provision.js +160 -0
  49. package/src/core/observability/support/renderer.js +2 -2
  50. package/src/core/observability/support/schedule.js +174 -0
  51. package/src/decision/registry.js +144 -0
  52. package/src/decision/reviewVerdict.js +309 -0
  53. package/template_project/.claude/agents/code-reviewer.md +25 -1
  54. package/template_project/.claude/agents/ukit-small-task-maintainer.md +16 -0
  55. package/template_project/.claude/commands/ukit/handoff-fullstack.md +2 -0
  56. package/template_project/.claude/commands/ukit/handoff-review.md +12 -0
  57. package/template_project/.claude/hooks/auto-allow-bash.sh +7 -1
  58. package/template_project/.claude/hooks/auto-prune-bash.sh +16 -7
  59. package/template_project/.claude/hooks/verification-guard.sh +13 -4
  60. package/template_project/.claude/ukit/index/review-verdict.mjs +592 -0
  61. package/template_project/.claude/ukit/index/sidecar-decision.mjs +595 -0
  62. package/template_project/.claude/ukit/index/unic-decision.mjs +10 -1
  63. package/template_project/.claude/ukit/runtime/async-lock.mjs +26 -0
  64. package/template_project/.codex/settings.json +3 -0
  65. package/template_project/.omp/agents/code-reviewer.md +25 -1
  66. package/template_project/.omp/agents/ukit-small-task-maintainer.md +16 -0
package/CHANGELOG.md CHANGED
@@ -2,6 +2,27 @@
2
2
 
3
3
  All notable changes to UKit are documented here.
4
4
 
5
+ ## 3.1.1 - 2026-09-26
6
+
7
+ - **Decision plane — review + workflow families (UNIC_DECISION_MIGRATION S1–S4)** — the reserved `review` and `workflow` registry families are now populated and wired; all keys ship `rolloutStage: 'off'` so deterministic fallbacks stay authoritative (stage promotion is a separate owner decision):
8
+ - **Registry** (`src/decision/registry.js`): `review.verdict.v1`, `review.finding-bucket.v1`, `review.panel-verdict.v1`, `workflow.sidecar-lane.v1`, `workflow.sidecar-risk.v1`, `workflow.routing-needed.v1`, `workflow.step-budget.v1`, `workflow.compact-now.v1`, `workflow.summarize-vs-keep.v1` — all `choice` kind, `off`, deterministic fallback owners recorded.
9
+ - **Review adapter**: `src/decision/reviewVerdict.js` + installed twin `template_project/.claude/ukit/index/review-verdict.mjs` — findings (whitelisted `{id, severity, file, line, claim}`) in, stage-gated verdict + per-finding buckets out; reviewers still generate findings on the general model, only the bounded verdict moves to `unic-decision` when `decisionPlane.families.review.stage` is enabled.
10
+ - **Sidecar adapter**: `template_project/.claude/ukit/index/sidecar-decision.mjs` answers the six `.codex/settings.json` `smallTaskModel.decisionPolicy.decisions` via a bounded `unic-decision` batch (stage-gated `decisionPlane.families.workflow.stage`); adapter failure → deterministic rule answer, never another LLM.
11
+ - **Engine parity**: `.codex/ukit` is a symlink to `.claude/ukit` (installed layout), `decisionAdapter` hints in `.codex/settings.json`; all three index CLIs' `isMainModule` gates now use `realpathSync` + basename equality — symlinked invocation previously compared unresolved paths and silently exited 0 without running `main()`.
12
+ - **Docs/tests**: `docs/pstack/UNIC_DECISION_GUIDE.md` (family keys + adapters), `docs/GATEWAY.md` §8, `docs/MEMORY.md`; parity + symlink-invocation coverage in `tests/index/{unicDecision,reviewVerdictCli,sidecarDecision}.test.js`.
13
+
14
+ ## 3.1.0 - 2026-09-26
15
+
16
+ - **Flight recorder — Data Foundation wave (cycle C81)** — the observability pipeline grows from record/segment plumbing into a full local diagnostics surface; still opt-in behind `observability.stage` (default `off`, `off→shadow→canary→default` per seam, absolute kill switch):
17
+ - **Envelope v1.1 + sampling**: additive `agent_id`/`project_instance_id`/`sampling` stamp fields and new semantic names (`memory.write`, `memory.retrieved`, `retrieval.query`, `outcome.observed`, `anomaly.detected`, `telemetry.crashed`); `observability.sampling` seeded-bucket keep-rates per importance level with `SAMPLED_OUT`/`CONFIG_INVALID` reason codes — `critical` is never sampled.
18
+ - **Segments**: sealed segments compress to `.jsonl.zst` on Node ≥ 22.15 (plain `.jsonl` otherwise); readers, recovery, and quarantine handle both transparently.
19
+ - **Crash capture + lifecycle**: bounded `crashes/crash-<boot>-<seq>.json` capture on uncaughtException/unhandledRejection/SIGINT/SIGTERM (≤10 files, FIFO, never throws); `getRecorder` memoized per (root, boot) with interval flush and deadline-bounded exit flush.
20
+ - **Ingestion + instrumentation**: `ingestStoredTelemetry` pulls stored hook telemetry, route/decision ledgers, and memory/context items into segments; agentRuntime and memory/retrieval paths emit real trace/span ids with honest `UNKNOWN` resource provenance.
21
+ - **CLI**: `ukit telemetry` — `status`, `collect`, `digest`, `export-support`, `import <path>`, `evaluate` (`--json` on status/digest/evaluate; exit 0/2, usage 1). `ukit doctor` gains a flight-recorder section (stage, segments, support lag, crashes — `unknown` when unavailable). `ukit feedback` emits `outcome.observed`.
22
+ - **Support view**: `Documents/UKit Support` provisioned at install (never deleted by uninstall); support refresh is scheduled after collect and on the lifecycle interval, with crash list, anomaly section, and freshness lag stamped in `SUMMARY.md`; `import` validates bundles (manifest/checksums/traversal/decompression bounds).
23
+ - **Anomaly detection + evaluation**: deterministic detectors (duration outliers, retry storms, cache-miss/drop-rate spikes, verification-failure clusters) produce `anomaly.detected` records with evidence refs; `runEvaluation` dispatches to a configured `observability.evaluator` provider — `skipped/no-provider` is the honest default — and appends validated findings to the optimization KB; expanded golden corpus with `--anomalies` bench flag.
24
+ - **Docs**: `docs/OBSERVABILITY.md` — stages, privacy contract (what is never recorded), the support-folder zip-send flow, CLI and config reference.
25
+
5
26
  ## 3.0.12 - 2026-09-26
6
27
 
7
28
  - Re-publish of the 3.0.11 content: npm staged-publish left 3.0.11 in limbo (E409 "previously staged version"); re-published as 3.0.12 — same tree, version bump only.
package/README.md CHANGED
@@ -95,6 +95,7 @@ Maintainer/debug commands (advanced workflows, not team onboarding):
95
95
  - `ukit memory ...` — inspect/export/forget shared memory items
96
96
  - `ukit update` — upgrade the global UKit CLI to the latest published version
97
97
  - `ukit index ...` — run repo indexing/query/triage tools directly
98
+ - `ukit telemetry ...` — inspect the opt-in local flight recorder (segments, digests, `Documents/UKit Support` export) — see [docs/OBSERVABILITY.md](docs/OBSERVABILITY.md)
98
99
  - `ukit build index` — alias for `ukit index build`
99
100
 
100
101
  ## Development
@@ -296,6 +296,18 @@ entries:
296
296
  archive_policy: date-rotate
297
297
  budget: { max_lines: 150, enforcement: error }
298
298
  notes: enforcement flipped to error by TASK-005 (2026-09-19); TASK-217 audit — actual 83 lines (~55% headroom), value kept
299
+ - id: docs-observability
300
+ path: docs/OBSERVABILITY.md
301
+ class: canonical
302
+ audience: [agent, maintainer]
303
+ owner: product
304
+ source_of_truth: docs/OBSERVABILITY.md
305
+ merge_strategy: none
306
+ load_policy: on-demand
307
+ validation: [manual]
308
+ archive_policy: never
309
+ notes: 'registered by C81 fix round 1 — Data Foundation/Flight Recorder surface doc'
310
+
299
311
  - id: docs-worklog
300
312
  path: docs/WORKLOG.md
301
313
  class: runtime
@@ -1750,6 +1750,30 @@ items:
1750
1750
  packs:
1751
1751
  - core
1752
1752
 
1753
+ - id: ukit-index-review-verdict-script
1754
+ type: config
1755
+ sourceTemplate: .claude/ukit/index/review-verdict.mjs
1756
+ targetPath: .claude/ukit/index/review-verdict.mjs
1757
+ requires:
1758
+ - ukit-index-unic-decision-script
1759
+ mergeStrategy: overwrite_with_backup
1760
+ variables: []
1761
+ enabledByDefault: true
1762
+ packs:
1763
+ - core
1764
+
1765
+ - id: ukit-index-sidecar-decision-script
1766
+ type: config
1767
+ sourceTemplate: .claude/ukit/index/sidecar-decision.mjs
1768
+ targetPath: .claude/ukit/index/sidecar-decision.mjs
1769
+ requires:
1770
+ - ukit-index-unic-decision-script
1771
+ mergeStrategy: overwrite_with_backup
1772
+ variables: []
1773
+ enabledByDefault: true
1774
+ packs:
1775
+ - core
1776
+
1753
1777
  - id: ukit-index-extract-image-script
1754
1778
  type: config
1755
1779
  sourceTemplate: .claude/ukit/index/extract-image.mjs
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ngockhoale/ukit",
3
- "version": "3.0.12",
3
+ "version": "3.1.1",
4
4
  "description": "Install/update an index-first AI workspace for Claude Code, OpenAI Codex and omp (Oh My Pi).",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -9,6 +9,15 @@
9
9
  // a fixed-seed variance report (TASK-014).
10
10
  // --perturb Inject seeded synthetic mutations into corpus summaries and measure
11
11
  // detection through computeFingerprint/detectOpportunities (TASK-002).
12
+ // --anomalies Replay the golden cases, summarize each run, and report anomaly
13
+ // detection counts per case (TASK-015). Degrades honestly to
14
+ // `detector: 'unavailable'` while analytics/anomalies.js
15
+ // (TASK-011) does not exist — never a stub detector.
16
+ //
17
+ // Flags run in argument order; the exit code is the first non-zero code.
18
+ // --evaluate exits 1 by design while the corpus carries expected-negative
19
+ // strata (a fail verdict is the honest scorecard), so a combined invocation
20
+ // ending in --evaluate exits 1.
12
21
  //
13
22
  // The corpus is a synthetic, seeded fixture: no real user content. Records are
14
23
  // envelope-shaped per SPEC §7 (DF-FR01) but intentionally NOT schema-validated —
@@ -77,10 +86,9 @@ function makeCtx() {
77
86
  tokens: (lo, hi) => lo + Math.floor(rand() * (hi - lo + 1)),
78
87
  };
79
88
  }
80
-
81
- function record(ctx, { semanticName, traceId, spanId, parentSpanId, executionId, importance, privacyClass, payload, dtMs }) {
89
+ function record(ctx, { semanticName, traceId, spanId, parentSpanId, executionId, agentId, importance, privacyClass, payload, dtMs }) {
82
90
  const t = ctx.tick(dtMs);
83
- return {
91
+ const rec = {
84
92
  record_type: 'fact',
85
93
  semantic_name: semanticName,
86
94
  schema_version: SCHEMA_VERSION,
@@ -100,9 +108,13 @@ function record(ctx, { semanticName, traceId, spanId, parentSpanId, executionId,
100
108
  privacy_class: clampToRegistryFloor(semanticName, privacyClass),
101
109
  payload,
102
110
  };
111
+ // DF2-FR01 v1.1: agent_id is the optional multi-agent correlation field —
112
+ // emitted only when the scenario declares one.
113
+ if (typeof agentId === 'string' && agentId.length > 0) rec.agent_id = agentId;
114
+ return rec;
103
115
  }
104
116
 
105
- function execSpan(ctx, traceId, executionId, privacyClass) {
117
+ function execSpan(ctx, traceId, executionId, privacyClass, agentId) {
106
118
  const spanId = ctx.spanId();
107
119
  return {
108
120
  spanId,
@@ -112,6 +124,7 @@ function execSpan(ctx, traceId, executionId, privacyClass) {
112
124
  spanId,
113
125
  parentSpanId: null,
114
126
  executionId,
127
+ agentId,
115
128
  importance: 'normal',
116
129
  privacyClass,
117
130
  payload: { operation: 'task', duration_ms: null },
@@ -120,7 +133,7 @@ function execSpan(ctx, traceId, executionId, privacyClass) {
120
133
  };
121
134
  }
122
135
 
123
- function modelAttempt(ctx, { traceId, parentSpanId, executionId, attemptIndex, status, resource, privacyClass, extra = {} }) {
136
+ function modelAttempt(ctx, { traceId, parentSpanId, executionId, agentId, attemptIndex, status, resource, privacyClass, extra = {} }) {
124
137
  const spanId = ctx.spanId();
125
138
  const started = record(ctx, {
126
139
  semanticName: 'model.started',
@@ -128,6 +141,7 @@ function modelAttempt(ctx, { traceId, parentSpanId, executionId, attemptIndex, s
128
141
  spanId,
129
142
  parentSpanId,
130
143
  executionId,
144
+ agentId,
131
145
  importance: 'normal',
132
146
  privacyClass,
133
147
  payload: { attempt_index: attemptIndex, ...extra },
@@ -139,6 +153,7 @@ function modelAttempt(ctx, { traceId, parentSpanId, executionId, attemptIndex, s
139
153
  spanId,
140
154
  parentSpanId,
141
155
  executionId,
156
+ agentId,
142
157
  importance: 'normal',
143
158
  privacyClass,
144
159
  payload: {
@@ -152,14 +167,14 @@ function modelAttempt(ctx, { traceId, parentSpanId, executionId, attemptIndex, s
152
167
  });
153
168
  return [started, completed];
154
169
  }
155
-
156
- function execEnd(ctx, { traceId, spanId, executionId, status, privacyClass, extra = {} }) {
170
+ function execEnd(ctx, { traceId, spanId, executionId, agentId, status, privacyClass, extra = {} }) {
157
171
  return record(ctx, {
158
172
  semanticName: `execution.${status}`,
159
173
  traceId,
160
174
  spanId,
161
175
  parentSpanId: null,
162
176
  executionId,
177
+ agentId,
163
178
  importance: 'normal',
164
179
  privacyClass,
165
180
  payload: { duration_ms: 0, ...extra },
@@ -174,11 +189,10 @@ function sumUsage(attempts) {
174
189
  .map((res) => res.value);
175
190
  return values.length ? values.reduce((a, b) => a + b, 0) : null;
176
191
  }
177
-
178
- function buildTrace(ctx, { scenario, privacyClass, build }) {
192
+ function buildTrace(ctx, { scenario, privacyClass, agentId, build }) {
179
193
  const traceId = ctx.traceId();
180
194
  const executionId = `exec-${traceId}`;
181
- const exec = execSpan(ctx, traceId, executionId, privacyClass);
195
+ const exec = execSpan(ctx, traceId, executionId, privacyClass, agentId);
182
196
  const { records, attempts, succeeded } = build({ traceId, executionId, execSpanId: exec.spanId });
183
197
  const all = [exec.started, ...records];
184
198
  return {
@@ -323,6 +337,109 @@ function buildCorpus() {
323
337
  }),
324
338
  );
325
339
 
340
+ // memory_outcome: retrieval.query → memory.retrieved hits → model →
341
+ // memory.write → execution.completed → outcome.observed — metadata-only
342
+ // payloads (counts, kinds, opaque refs), never content.
343
+ traces.push(
344
+ buildTrace(ctx, {
345
+ scenario: 'memory_outcome',
346
+ privacyClass: 'sensitive',
347
+ build: ({ traceId, executionId, execSpanId }) => {
348
+ const query = record(ctx, {
349
+ semanticName: 'retrieval.query',
350
+ traceId, spanId: ctx.spanId(), parentSpanId: execSpanId, executionId,
351
+ importance: 'normal', privacyClass: 'internal',
352
+ payload: {
353
+ lane: 'v2', scope: 'all', indexed: true,
354
+ candidate_count: ctx.tokens(10, 40), returned_count: 3,
355
+ duration_ms: ctx.tokens(5, 20), freshness: {},
356
+ },
357
+ dtMs: 8,
358
+ });
359
+ const retrieved = [
360
+ { kind: 'fact' }, { kind: 'decision' }, { kind: 'decision' },
361
+ ].map((hit) => record(ctx, {
362
+ semanticName: 'memory.retrieved',
363
+ traceId, spanId: ctx.spanId(), parentSpanId: execSpanId, executionId,
364
+ importance: 'normal', privacyClass: 'sensitive',
365
+ payload: { kind: hit.kind, record_ref: `memref-${ctx.tokens(1000, 9999)}` },
366
+ dtMs: 4,
367
+ }));
368
+ const resource = { value: ctx.tokens(800, 1200), source: 'PROVIDER' };
369
+ const attempts = modelAttempt(ctx, {
370
+ traceId, parentSpanId: execSpanId, executionId,
371
+ attemptIndex: 1, status: 'ok', resource, privacyClass: 'public',
372
+ });
373
+ const write = record(ctx, {
374
+ semanticName: 'memory.write',
375
+ traceId, spanId: ctx.spanId(), parentSpanId: execSpanId, executionId,
376
+ importance: 'normal', privacyClass: 'sensitive',
377
+ payload: {
378
+ op: 'write', record_count: ctx.tokens(1, 5), changed_count: 1,
379
+ kinds: { fact: 1 }, record_refs: [`memref-${ctx.tokens(1000, 9999)}`],
380
+ tombstone_count: 0,
381
+ },
382
+ dtMs: 6,
383
+ });
384
+ const end = execEnd(ctx, {
385
+ traceId, spanId: execSpanId, executionId, status: 'completed',
386
+ privacyClass: 'public', extra: { outcome: 'success' },
387
+ });
388
+ const outcome = record(ctx, {
389
+ semanticName: 'outcome.observed',
390
+ traceId, spanId: ctx.spanId(), parentSpanId: execSpanId, executionId,
391
+ importance: 'high', privacyClass: 'internal',
392
+ payload: {
393
+ kind: 'feedback', verdict: 'accepted', source: 'user',
394
+ evidence_refs: [end.record_id],
395
+ },
396
+ dtMs: 30,
397
+ });
398
+ return {
399
+ records: [query, ...retrieved, ...attempts, write, end, outcome],
400
+ attempts,
401
+ succeeded: true,
402
+ };
403
+ },
404
+ }),
405
+ );
406
+
407
+ // multi_agent: one trace spanning two agents — orchestrator runs the
408
+ // retrieval query and owns the execution span, worker runs the model
409
+ // attempt. agent_id is the v1.1 correlation field; trace_id stays single.
410
+ traces.push(
411
+ buildTrace(ctx, {
412
+ scenario: 'multi_agent',
413
+ privacyClass: 'public',
414
+ agentId: 'agent-orchestrator',
415
+ build: ({ traceId, executionId, execSpanId }) => {
416
+ const query = record(ctx, {
417
+ semanticName: 'retrieval.query',
418
+ traceId, spanId: ctx.spanId(), parentSpanId: execSpanId, executionId,
419
+ agentId: 'agent-orchestrator',
420
+ importance: 'normal', privacyClass: 'internal',
421
+ payload: {
422
+ lane: 'v2', scope: 'all', indexed: true,
423
+ candidate_count: ctx.tokens(10, 40), returned_count: 2,
424
+ duration_ms: ctx.tokens(5, 20), freshness: {},
425
+ },
426
+ dtMs: 8,
427
+ });
428
+ const resource = { value: ctx.tokens(800, 1200), source: 'PROVIDER' };
429
+ const attempts = modelAttempt(ctx, {
430
+ traceId, parentSpanId: execSpanId, executionId,
431
+ agentId: 'agent-worker',
432
+ attemptIndex: 1, status: 'ok', resource, privacyClass: 'public',
433
+ });
434
+ const end = execEnd(ctx, {
435
+ traceId, spanId: execSpanId, executionId, agentId: 'agent-orchestrator',
436
+ status: 'completed', privacyClass: 'public', extra: { outcome: 'success' },
437
+ });
438
+ return { records: [query, ...attempts, end], attempts, succeeded: true };
439
+ },
440
+ }),
441
+ );
442
+
326
443
  return {
327
444
  metric_version: METRIC_VERSION,
328
445
  seed: SEED,
@@ -560,52 +677,253 @@ export async function runPerturb({ quiet = false } = {}) {
560
677
  return { code: outcome.missed.length > 0 ? 1 : 0, report };
561
678
  }
562
679
 
563
- function main(argv) {
564
- const flag = argv[2];
680
+ // --anomalies: replay every golden case under each pre-registered policy,
681
+ // summarize each materialized run, and report per-case anomaly detection
682
+ // counts keyed by detector kind. The detector is imported defensively:
683
+ // while src/core/observability/analytics/anomalies.js (TASK-011) does not
684
+ // exist the report degrades to `detector: 'unavailable'` — never a stub.
685
+ // When a detector IS available, per-case declarations in
686
+ // `expect.anomalies.expected_kinds` act as a regression gate: unmet or
687
+ // over-fired kinds exit 1. Evaluation is offline — no model calls.
688
+ async function loadAnomalyDetector() {
689
+ try {
690
+ const mod = await import('../../src/core/observability/analytics/anomalies.js');
691
+ return typeof mod.detectAnomalies === 'function' ? mod.detectAnomalies : null;
692
+ } catch {
693
+ return null;
694
+ }
695
+ }
696
+
697
+ function anomalyKind(record) {
698
+ const kind = record && record.payload && record.payload.kind !== undefined
699
+ ? record.payload.kind
700
+ : record && record.kind;
701
+ return typeof kind === 'string' && kind.length > 0 ? kind : 'unknown';
702
+ }
703
+
704
+ async function runAnomaliesBench() {
705
+ const { replayCase } = await import('../../src/core/observability/evaluation/replay.js');
706
+ const { summarizeTrace } = await import('../../src/core/observability/analytics/summary.js');
707
+ const detectAnomalies = await loadAnomalyDetector();
708
+
709
+ const goldenPath = path.join(repoRoot, 'tests/fixtures/observability/golden/cases.json');
710
+ let corpus;
711
+ try {
712
+ corpus = JSON.parse(fs.readFileSync(goldenPath, 'utf8'));
713
+ } catch (err) {
714
+ process.stderr.write(`--anomalies: cannot load ${goldenPath}: ${err && err.message}\n`);
715
+ return 1;
716
+ }
717
+ const policies = Object.keys(
718
+ (corpus.pre_registered && corpus.pre_registered.policies) || {},
719
+ ).sort();
720
+ if (policies.length === 0) {
721
+ process.stderr.write('--anomalies: corpus declares no pre_registered.policies\n');
722
+ return 1;
723
+ }
724
+
725
+ const runs = [];
726
+ for (const goldenCase of corpus.cases || []) {
727
+ const expectedKinds = goldenCase && goldenCase.expect && goldenCase.expect.anomalies
728
+ && Array.isArray(goldenCase.expect.anomalies.expected_kinds)
729
+ ? goldenCase.expect.anomalies.expected_kinds
730
+ : null;
731
+ const run = { case_id: goldenCase && goldenCase.case_id, expected_kinds: expectedKinds, policies: {} };
732
+ for (const policyId of policies) {
733
+ const replay = replayCase(goldenCase, { policyId });
734
+ if (!replay.ok) {
735
+ run.policies[policyId] = { ok: false, reason: replay.reason };
736
+ continue;
737
+ }
738
+ const summary = summarizeTrace(replay.records);
739
+ run.policies[policyId] = {
740
+ ok: true,
741
+ trace_id: summary.trace_id,
742
+ summary,
743
+ record_ids: replay.records.map((r) => r.record_id),
744
+ };
745
+ }
746
+ runs.push(run);
747
+ }
748
+
749
+ // Detector wiring: one call over ALL materialized summaries (detectors are
750
+ // cohort-relative), then attribution back to cases by trace_id or any
751
+ // evidence_ref that resolves to a materialized record_id.
752
+ const summaries = runs.flatMap((run) => policies
753
+ .map((policyId) => run.policies[policyId])
754
+ .filter((entry) => entry && entry.ok)
755
+ .map((entry) => entry.summary));
756
+
757
+ const byCase = new Map(runs.map((run) => [run.case_id, new Map()]));
758
+ const byTrace = new Map();
759
+ const byRecordId = new Map();
760
+ for (const run of runs) {
761
+ for (const policyId of policies) {
762
+ const entry = run.policies[policyId];
763
+ if (!entry || !entry.ok) continue;
764
+ if (typeof entry.trace_id === 'string') byTrace.set(entry.trace_id, [run.case_id, policyId]);
765
+ for (const recordId of entry.record_ids) byRecordId.set(recordId, [run.case_id, policyId]);
766
+ }
767
+ }
768
+
769
+ let detection = null;
770
+ if (detectAnomalies) {
771
+ const anomalies = detectAnomalies(summaries) || [];
772
+ const unattributed = [];
773
+ for (const anomaly of anomalies) {
774
+ const refs = anomaly && anomaly.payload && Array.isArray(anomaly.payload.evidence_refs)
775
+ ? anomaly.payload.evidence_refs
776
+ : [];
777
+ const traceRef = anomaly && anomaly.payload && typeof anomaly.payload.trace_ref === 'string'
778
+ ? anomaly.payload.trace_ref
779
+ : (anomaly && typeof anomaly.trace_id === 'string' ? anomaly.trace_id : null);
780
+ let target = traceRef ? byTrace.get(traceRef) : null;
781
+ if (!target) {
782
+ for (const ref of refs) {
783
+ target = byRecordId.get(ref);
784
+ if (target) break;
785
+ }
786
+ }
787
+ if (!target) { unattributed.push(anomaly); continue; }
788
+ const [caseId, policyId] = target;
789
+ const policyMap = byCase.get(caseId);
790
+ if (!policyMap.has(policyId)) policyMap.set(policyId, {});
791
+ const counts = policyMap.get(policyId);
792
+ const kind = anomalyKind(anomaly);
793
+ counts[kind] = (counts[kind] || 0) + 1;
794
+ }
795
+ detection = { anomalies, unattributed };
796
+ }
797
+ // Per-case signals — the summary fields the detectors consume (TASK-011):
798
+ // attempt/retry counters, cache posture, drops, critical path. Reported
799
+ // per policy run so a detector regression is attributable to a case.
800
+ function signalsOf(entry) {
801
+ const s = entry && entry.summary;
802
+ if (!s) return null;
803
+ return {
804
+ model_attempts: s.retries.model_attempts,
805
+ retries: s.retries.retries,
806
+ failed_spans: s.retries.failed_spans,
807
+ cache_misses: s.cache.misses,
808
+ cache_hits: s.cache.hits,
809
+ hit_rate: s.cache.hit_rate,
810
+ dropped_events: s.drops.events,
811
+ dropped_count: s.drops.dropped_count,
812
+ critical_path_ms: s.critical_path_ms,
813
+ telemetry_complete: s.telemetry_complete,
814
+ spans: s.coverage.spans,
815
+ };
816
+ }
817
+
818
+ const casesReport = {};
819
+ for (const run of runs) {
820
+ const signals = {};
821
+ for (const policyId of policies) {
822
+ signals[policyId] = signalsOf(run.policies[policyId]);
823
+ }
824
+ let anomalies = null;
825
+ if (detection) {
826
+ anomalies = 0;
827
+ for (const counts of byCase.get(run.case_id)?.values() || []) {
828
+ for (const n of Object.values(counts)) anomalies += n;
829
+ }
830
+ }
831
+ casesReport[run.case_id] = { signals, anomalies };
832
+ }
833
+
834
+ // Regression gate — only meaningful when a real detector ran. A case that
835
+ // declares expect.anomalies.expected_kinds must see those kinds fire; a
836
+ // case declaring an empty list must stay clean.
837
+ const unmet = [];
838
+ if (detection) {
839
+ for (const run of runs) {
840
+ if (!Array.isArray(run.expected_kinds)) continue;
841
+ for (const [policyId, counts] of byCase.get(run.case_id) || []) {
842
+ for (const kind of run.expected_kinds) {
843
+ if (!counts[kind]) {
844
+ unmet.push({ case_id: run.case_id, policy: policyId, kind, expected: '>=1', got: 0 });
845
+ }
846
+ }
847
+ for (const kind of Object.keys(counts)) {
848
+ if (!run.expected_kinds.includes(kind)) {
849
+ unmet.push({ case_id: run.case_id, policy: policyId, kind, expected: 'absent', got: counts[kind] });
850
+ }
851
+ }
852
+ }
853
+ }
854
+ unmet.sort((a, b) => String(a.case_id).localeCompare(String(b.case_id))
855
+ || a.policy.localeCompare(b.policy) || a.kind.localeCompare(b.kind));
856
+ }
857
+
858
+ const report = {
859
+ metric_version: METRIC_VERSION,
860
+ bench: 'anomalies',
861
+ seed: SEED,
862
+ corpus: path.relative(repoRoot, goldenPath),
863
+ policies,
864
+ detector: detectAnomalies ? 'detectAnomalies' : 'unavailable',
865
+ summaries: summaries.length,
866
+ anomalies_total: detection ? detection.anomalies.length : null,
867
+ unattributed: detection ? detection.unattributed.length : null,
868
+ unmet_expectations: detection ? unmet : null,
869
+ cases: casesReport,
870
+ };
871
+ process.stdout.write(`${JSON.stringify(report, null, 2)}\n`);
872
+ if (!detection) return 0; // degraded — nothing to gate on
873
+ return unmet.length > 0 ? 1 : 0;
874
+ }
875
+
876
+ const BENCH_FLAGS = Object.freeze([
877
+ '--fixture',
878
+ '--recorder',
879
+ '--analytics',
880
+ '--evaluate',
881
+ '--perturb',
882
+ '--anomalies',
883
+ ]);
884
+
885
+ async function runFlag(flag) {
565
886
  switch (flag) {
566
- case '--fixture':
567
- return writeFixture();
568
- case '--recorder':
569
- return runRecorderBench().then(
570
- (code) => code,
571
- (err) => {
572
- process.stderr.write(`--recorder failed: ${err && err.message}\n`);
573
- return 1;
574
- },
575
- );
576
- case '--analytics':
577
- return runAnalyticsBench().then(
578
- (code) => code,
579
- (err) => {
580
- process.stderr.write(`--analytics failed: ${err && err.message}\n`);
581
- return 1;
582
- },
583
- );
584
- case '--evaluate':
585
- return runEvaluateBench().then(
586
- (code) => code,
587
- (err) => {
588
- process.stderr.write(`--evaluate failed: ${err && err.message}\n`);
589
- return 1;
590
- },
591
- );
592
- case '--perturb':
593
- return runPerturb().then(
594
- ({ code }) => code,
595
- (err) => {
596
- process.stderr.write(`--perturb failed: ${err && err.message}\n`);
597
- return 1;
598
- },
599
- );
887
+ case '--fixture': return writeFixture();
888
+ case '--recorder': return runRecorderBench();
889
+ case '--analytics': return runAnalyticsBench();
890
+ case '--evaluate': return runEvaluateBench();
891
+ case '--perturb': return runPerturb().then(({ code }) => code);
892
+ case '--anomalies': return runAnomaliesBench();
600
893
  default:
601
- process.stderr.write(
602
- `usage: node scripts/bench/data-foundation.mjs --fixture|--recorder|--analytics|--evaluate|--perturb\n`,
603
- );
894
+ process.stderr.write(`unknown flag: ${flag}\n`);
604
895
  return 2;
605
896
  }
606
897
  }
898
+
899
+ // Flags run in argument order; the exit code is the first non-zero result.
900
+ // (PLAN §5 verification chains --recorder --analytics --evaluate.)
901
+ async function main(argv) {
902
+ const flags = argv.slice(2).filter((arg) => arg.startsWith('--'));
903
+ if (flags.length === 0) {
904
+ process.stderr.write(
905
+ `usage: node scripts/bench/data-foundation.mjs ${BENCH_FLAGS.join('|')}\n`,
906
+ );
907
+ return 2;
908
+ }
909
+ let firstFailure = 0;
910
+ for (const flag of flags) {
911
+ try {
912
+ const code = await runFlag(flag);
913
+ if (code !== 0 && firstFailure === 0) firstFailure = code;
914
+ } catch (err) {
915
+ process.stderr.write(`${flag} failed: ${err && err.message}\n`);
916
+ if (firstFailure === 0) firstFailure = 1;
917
+ }
918
+ }
919
+ return firstFailure;
920
+ }
607
921
  const invokedAs = process.argv[1] ? path.resolve(process.argv[1]) : '';
608
922
  if (invokedAs === fileURLToPath(import.meta.url)) {
609
- Promise.resolve(main(process.argv)).then((code) => process.exit(code));
923
+ // Set the code then let the loop drain stdout — process.exit() hard-kills
924
+ // pending pipe writes, which truncates reports past the 64 KiB buffer.
925
+ Promise.resolve(main(process.argv)).then((code) => {
926
+ process.exitCode = code;
927
+ });
610
928
  }
611
929