@ngockhoale/ukit 3.0.7 → 3.0.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/CHANGELOG.md +18 -1
  2. package/manifests/documentation.yaml +11 -0
  3. package/package.json +1 -1
  4. package/scripts/audit/decision-coverage.mjs +29 -2
  5. package/scripts/bench/data-foundation.mjs +52 -3
  6. package/scripts/bench/decision-runtime-baseline.mjs +427 -0
  7. package/scripts/bench/decision-runtime-metrics.mjs +67 -0
  8. package/scripts/bench/decision-runtime-variant.mjs +626 -0
  9. package/scripts/bench/memory-ablation.mjs +495 -0
  10. package/scripts/bench/memory-baseline.mjs +596 -0
  11. package/scripts/bench/memory-bench.mjs +661 -0
  12. package/scripts/bench/memory-canary.mjs +321 -0
  13. package/scripts/bench/memory-corpus.mjs +354 -0
  14. package/scripts/bench/memory-gate.mjs +389 -0
  15. package/scripts/bench/memory-metrics.mjs +179 -0
  16. package/scripts/bench/parallel-agents.mjs +33 -11
  17. package/scripts/bench/recorder-overhead.mjs +204 -0
  18. package/scripts/bench/sqlite-spike.mjs +451 -0
  19. package/scripts/measure-decision-gateway.mjs +306 -0
  20. package/scripts/perf/audit-perf.mjs +35 -17
  21. package/src/bug/triageBug.js +4 -3
  22. package/src/cli/commands/memory.js +357 -63
  23. package/src/context/detectProjectContext.js +11 -1
  24. package/src/core/agentRuntime/adapters.js +254 -0
  25. package/src/core/agentRuntime/artifacts.js +192 -0
  26. package/src/core/agentRuntime/completionGate.js +176 -0
  27. package/src/core/agentRuntime/context.js +149 -0
  28. package/src/core/agentRuntime/contract.js +247 -0
  29. package/src/core/agentRuntime/diagnostics.js +244 -0
  30. package/src/core/agentRuntime/evaluation.js +163 -0
  31. package/src/core/agentRuntime/eventStore.js +404 -0
  32. package/src/core/agentRuntime/liveness.js +60 -0
  33. package/src/core/agentRuntime/planCompiler.js +322 -0
  34. package/src/core/agentRuntime/promotion.js +53 -0
  35. package/src/core/agentRuntime/qualityComparison.js +112 -0
  36. package/src/core/agentRuntime/recovery.js +266 -0
  37. package/src/core/agentRuntime/resourcePolicy.js +78 -0
  38. package/src/core/agentRuntime/runtimeSupport.js +237 -0
  39. package/src/core/agentRuntime/supervisor.js +565 -0
  40. package/src/core/agentRuntime/vmEngine.js +621 -0
  41. package/src/core/codeintel/analogy.js +3 -2
  42. package/src/core/experiments/dynamicWorkflow.js +17 -2
  43. package/src/core/fileOps.js +21 -3
  44. package/src/core/memory/deltaOverlays.js +75 -30
  45. package/src/core/memory/learningCandidates.js +93 -48
  46. package/src/core/memory/memoryFlags.js +83 -0
  47. package/src/core/memory/memoryFreshness.js +190 -0
  48. package/src/core/memory/memoryHit.js +144 -0
  49. package/src/core/memory/migrate.js +69 -189
  50. package/src/core/memory/migrateMapping.js +232 -0
  51. package/src/core/memory/mutateMemory.js +323 -0
  52. package/src/core/memory/policy.js +96 -0
  53. package/src/core/memory/projectIdentity.js +266 -0
  54. package/src/core/memory/recordIndex.js +178 -0
  55. package/src/core/memory/recordStore.js +133 -20
  56. package/src/core/memory/records.js +144 -6
  57. package/src/core/memory/retrieval.js +259 -125
  58. package/src/core/memory/store.js +16 -5
  59. package/src/core/memory/storeBackup.js +226 -0
  60. package/src/core/memory/storeV2.js +63 -26
  61. package/src/core/memory/storeV2Loader.js +30 -12
  62. package/src/core/memory/userMemory.js +38 -20
  63. package/src/core/memory/writeClassification.js +161 -0
  64. package/src/core/memory/writeGuard.js +129 -0
  65. package/src/core/observability/adapters/hookTelemetryAdapter.js +90 -0
  66. package/src/core/observability/analytics/cohorts.js +148 -0
  67. package/src/core/observability/analytics/storeDigest.js +163 -0
  68. package/src/core/observability/evaluation/experimentPlan.js +95 -0
  69. package/src/core/observability/evaluation/findings.js +99 -0
  70. package/src/core/observability/evaluation/optimizationKnowledge.js +10 -1
  71. package/src/core/observability/evaluation/perturbation.js +273 -0
  72. package/src/core/observability/evaluation/replay.js +7 -1
  73. package/src/core/observability/evaluation/scorecard.js +23 -3
  74. package/src/core/observability/rollout.js +11 -7
  75. package/src/core/observability/schema/compatibility.js +135 -0
  76. package/src/core/observability/schema/registry.js +99 -0
  77. package/src/core/observability/schema/validate.js +7 -0
  78. package/src/core/observability/support/import.js +53 -9
  79. package/src/core/observability/support/paths.js +13 -3
  80. package/src/core/observability/support/projector.js +148 -12
  81. package/src/core/output/index.js +12 -2
  82. package/src/core/runtimeConfig.js +83 -0
  83. package/src/core/runtimePaths.js +3 -0
  84. package/src/core/sensitiveValueScanner.js +40 -0
  85. package/src/core/token/index.js +40 -3
  86. package/src/decision/client.js +37 -13
  87. package/src/decision/protocol.js +1 -1
  88. package/src/decision/registry.js +5 -3
  89. package/src/decision/runtimeDecide.js +242 -0
  90. package/src/decision/runtimeFilter.js +150 -0
  91. package/src/decision/runtimeScheduler.js +239 -0
  92. package/src/index/buildIndex.js +13 -12
  93. package/src/index/queryIndex.js +35 -14
  94. package/src/index/relatedTests.js +50 -8
  95. package/src/index/resolveContext.js +9 -4
  96. package/src/manifest/selectItems.js +7 -3
  97. package/src/render/instructionRenderer.js +17 -5
  98. package/template_project/.claude/ukit/index/lib/index-core.mjs +94 -39
  99. package/template_project/.claude/ukit/index/route-task.mjs +121 -19
  100. package/template_project/.claude/ukit/index/unic-decision.mjs +28 -13
  101. package/template_project/.claude/ukit/runtime/memory-flags.mjs +51 -0
  102. package/template_project/.claude/ukit/runtime/memory-freshness.mjs +155 -0
  103. package/template_project/.claude/ukit/runtime/memory-policy.mjs +286 -0
  104. package/template_project/.claude/ukit/runtime/output-compression.mjs +3 -0
  105. package/template_project/.claude/ukit/runtime/reinject-context.mjs +145 -14
package/CHANGELOG.md CHANGED
@@ -2,6 +2,23 @@
2
2
 
3
3
  All notable changes to UKit are documented here.
4
4
 
5
+ ## 3.0.9 - 2026-09-25
6
+
7
+ - **ukit-memory stack (cycles C57–C62, waves W1–W6)** — the full v2 memory layer lands behind staged rollout flags:
8
+ - **Record store v2**: versioned `records.json` store (`src/core/memory/storeV2.js`, `recordStore.js`, `records.js`) with tombstones, generation counters, fingerprint dedupe, and a locked read-modify-write path in `mutateMemory`; automatic v1→v2 migration (`migrate.js`, `migrateMapping.js`) with `autoMigrate` config.
9
+ - **Retrieval**: deny-by-default eligibility policy (`policy.js` — status/expiry/provenance/trust-tier/project-binding), generation-bound in-memory `recordIndex`, ranked hits (`memoryHit.js`, `retrieval.js`), and labeled `## Previous Context` injection that renders record text as data, never instructions.
10
+ - **Freshness**: `memoryFreshness.js` resolves `evidence[].locator` with `lstat`+`realpath` containment — symlinks escaping `projectRoot` report `locator-unverifiable`, never `fresh`.
11
+ - **Write safety**: `writeGuard.js`/`writeClassification.js` reject secret-bearing payloads before persist; `sensitiveValueScanner` shared.
12
+ - **Backup/restore/purge**: `storeBackup.js` byte-copy backups with sha256 manifest (mode 0600), manifest boundary check (`backup-file-out-of-boundary`), tombstone precedence on restore; CLI `memory purge/export` lanes sweep `deletePromptCacheEntries` so purged content cannot resurrect from the prompt cache.
13
+ - **Rollout flags (W6)**: `memoryV2.{eligibility,writer,index,decision}.stage` on the shared `off→shadow→canary→default` ladder, `canaryProjects`, and absolute `killSwitch` — read live on every operation, config-only reversal, no storage-format change. Ships at `default` for eligibility/writer/index, `off` for decision, `killSwitch: false`.
14
+ - **Security review (W6)**: `tests/security/memorySecurity.test.js` (11 tests) — fixed symlink-follow in `resolveLocator` and manifest path escape in `restoreBackup`; 0 open high-severity findings.
15
+ - **Gate G6**: canary delta clean (0 hit/query divergence on 2000-record seeded corpus), kill-switch drill pass on read and write legs — verdict **GO** (`docs/AI_HANDOFF/inventory/g6-verdict.md`).
16
+ - **Installed mirror**: `template_project/.claude/ukit/runtime/memory-{flags,freshness,policy}.mjs` shipped for hook-side parity.
17
+
18
+ ## 3.0.8 - 2026-09-24
19
+
20
+ - Re-release of the 3.0.7 content: npm staged-publish left 3.0.7 in limbo (E409 "previously staged version"); re-published as 3.0.8. Tag v3.0.7 exists on the same tree.
21
+
5
22
  ## 3.0.7 - 2026-09-24
6
23
 
7
24
  - **Data foundation & flight recorder (cycle C54)** — new `src/core/observability/` pipeline, all behind `observability.stage` (default `off`, staged `off→shadow→canary→default` per seam with kill switch):
@@ -105,7 +122,7 @@ All notable changes to UKit are documented here.
105
122
  ## 2.7.8 - 2026-09-22
106
123
 
107
124
  Stall/hang fixes — cycle C43 (TASK-001..008): the indefinite-stall producers
108
- root-caused in `docs/AI_REPORT/2026-09-21-NORMAL_CHAT_STALL.md` are fixed
125
+ root-caused in `docs/AI_REPORT/archive/2026-09-21-NORMAL_CHAT_STALL.md` are fixed
109
126
  across Claude Code, omp, and Codex runtimes.
110
127
 
111
128
  - **CX-8**: route-state writes are atomic + locked — torn `skill-router-state.json`
@@ -763,6 +763,17 @@ entries:
763
763
  validation: [manual]
764
764
  archive_policy: immutable
765
765
  notes: vendor evidence extracted from docs/PROMPT_CACHING.md (DOC-206, TASK-212)
766
+ - id: docs-decision-runtime-replay
767
+ path: docs/decision-runtime-replay.md
768
+ class: canonical
769
+ audience: [maintainer, user]
770
+ owner: runtime
771
+ source_of_truth: docs/decision-runtime-replay.md
772
+ merge_strategy: none
773
+ load_policy: on-demand
774
+ validation: [manual]
775
+ archive_policy: never
776
+ notes: C69 G7 operator doc — timeline render, support bundle contents/exclusions, replay limits
766
777
 
767
778
  # Declared complexity→docs mapping consumed by the router's `docs=` summary segment
768
779
  # (src/index/taskRouting.js CONTEXT_LAYER_DOCS mirrors this block; DOC-201 FR-001/002).
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ngockhoale/ukit",
3
- "version": "3.0.7",
3
+ "version": "3.0.9",
4
4
  "description": "Install/update an index-first AI workspace for Claude Code, OpenAI Codex and omp (Oh My Pi).",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -200,6 +200,17 @@ function scanFiles(root, extraScan) {
200
200
  if (entry.endsWith('.js')) files.push(path.join(absDir, entry));
201
201
  }
202
202
  }
203
+ for (const extra of extraScan) {
204
+ let stat;
205
+ try {
206
+ stat = fs.statSync(extra);
207
+ } catch (error) {
208
+ return { error: `extra-scan path is not a file: ${extra} (${error.code ?? error.message})` };
209
+ }
210
+ if (!stat.isFile()) {
211
+ return { error: `extra-scan path is not a file: ${extra}` };
212
+ }
213
+ }
203
214
  files.push(...extraScan);
204
215
  return { files };
205
216
  }
@@ -260,7 +271,16 @@ function main() {
260
271
  console.error(`decision-coverage: scan file missing: ${path.relative(args.root, file)}`);
261
272
  process.exit(1);
262
273
  }
263
- const source = fs.readFileSync(file, 'utf8');
274
+ let source;
275
+ try {
276
+ source = fs.readFileSync(file, 'utf8');
277
+ } catch (error) {
278
+ const rel = path.relative(args.root, file);
279
+ console.error(
280
+ `decision-coverage: cannot read ${rel}: ${error.code ?? error.message}`,
281
+ );
282
+ process.exit(1);
283
+ }
264
284
  discovered.push(...discoverInSource(source, path.basename(file)));
265
285
  }
266
286
 
@@ -292,4 +312,11 @@ function main() {
292
312
  );
293
313
  }
294
314
 
295
- main();
315
+ try {
316
+ main();
317
+ } catch (error) {
318
+ console.error(
319
+ `decision-coverage: ${error instanceof Error ? error.message : String(error)}`,
320
+ );
321
+ process.exit(1);
322
+ }
@@ -7,6 +7,8 @@
7
7
  // --evaluate Replay the golden corpus (tests/fixtures/observability/golden/cases.json)
8
8
  // under both pre-registered policies and print paired scorecards +
9
9
  // a fixed-seed variance report (TASK-014).
10
+ // --perturb Inject seeded synthetic mutations into corpus summaries and measure
11
+ // detection through computeFingerprint/detectOpportunities (TASK-002).
10
12
  //
11
13
  // The corpus is a synthetic, seeded fixture: no real user content. Records are
12
14
  // envelope-shaped per SPEC §7 (DF-FR01) but intentionally NOT schema-validated —
@@ -521,6 +523,43 @@ async function runEvaluateBench() {
521
523
  return failed ? 1 : 0;
522
524
  }
523
525
 
526
+ // --perturb: summarize the corpus, inject seeded synthetic mutations via
527
+ // mutateSummaries, then measure detection through the real
528
+ // computeFingerprint/detectOpportunities path. Exit 0 when every injected
529
+ // mutation is detected; 1 when missed[] is non-empty or the corpus is
530
+ // unloadable. Deterministic at SEED; evaluation is offline — no model calls.
531
+ export async function runPerturb({ quiet = false } = {}) {
532
+ const { summarizeTrace } = await import('../../src/core/observability/analytics/summary.js');
533
+ const { mutateSummaries, detectPerturbations } = await import(
534
+ '../../src/core/observability/evaluation/perturbation.js'
535
+ );
536
+
537
+ let corpus;
538
+ try {
539
+ corpus = JSON.parse(fs.readFileSync(corpusPath, 'utf8'));
540
+ } catch (err) {
541
+ process.stderr.write(`--perturb: cannot load ${corpusPath}: ${err && err.message}\n`);
542
+ return { code: 1, report: null };
543
+ }
544
+
545
+ const baseline = corpus.traces.map((t) => summarizeTrace(t.records));
546
+ const { summaries: mutated, manifest } = mutateSummaries(baseline, { seed: SEED });
547
+ const outcome = detectPerturbations(mutated, baseline, manifest);
548
+
549
+ const report = {
550
+ metric_version: METRIC_VERSION,
551
+ bench: 'perturb',
552
+ seed: SEED,
553
+ corpus: path.relative(repoRoot, corpusPath),
554
+ injected: outcome.injected,
555
+ detected: outcome.detected,
556
+ missed: outcome.missed,
557
+ manifest,
558
+ };
559
+ if (!quiet) process.stdout.write(`${JSON.stringify(report, null, 2)}\n`);
560
+ return { code: outcome.missed.length > 0 ? 1 : 0, report };
561
+ }
562
+
524
563
  function main(argv) {
525
564
  const flag = argv[2];
526
565
  switch (flag) {
@@ -550,13 +589,23 @@ function main(argv) {
550
589
  return 1;
551
590
  },
552
591
  );
592
+ case '--perturb':
593
+ return runPerturb().then(
594
+ ({ code }) => code,
595
+ (err) => {
596
+ process.stderr.write(`--perturb failed: ${err && err.message}\n`);
597
+ return 1;
598
+ },
599
+ );
553
600
  default:
554
601
  process.stderr.write(
555
- `usage: node scripts/bench/data-foundation.mjs --fixture|--recorder|--analytics|--evaluate\n`,
602
+ `usage: node scripts/bench/data-foundation.mjs --fixture|--recorder|--analytics|--evaluate|--perturb\n`,
556
603
  );
557
604
  return 2;
558
605
  }
559
606
  }
560
-
561
- Promise.resolve(main(process.argv)).then((code) => process.exit(code));
607
+ const invokedAs = process.argv[1] ? path.resolve(process.argv[1]) : '';
608
+ if (invokedAs === fileURLToPath(import.meta.url)) {
609
+ Promise.resolve(main(process.argv)).then((code) => process.exit(code));
610
+ }
562
611
 
@@ -0,0 +1,427 @@
1
+ #!/usr/bin/env node
2
+ // TASK-004 — DR-01b decision-runtime baseline runner (SPEC §5 FR-006, §8, §10).
3
+ //
4
+ // Replays the versioned synthetic corpus (docs/AI_HANDOFF/benchmark/corpus.yaml)
5
+ // through the TASK-003 metric module and writes a schema-valid JSON report:
6
+ // {version, env{sha,package,node,os,arch,fs}, runs,
7
+ // scenarios[{id,verdict,wallMs,wakes,eligible,quality,tokens}],
8
+ // summary{externalWakeRate,qualityScore,p50,p95}}
9
+ //
10
+ // The baseline replays deterministic inline commands only — no model/agent
11
+ // wake is ever made — so `wakes` is measured as 0 and `tokens` is 'UNKNOWN'
12
+ // (no provider data source exists; never 0 or an estimate, metric-spec §5).
13
+ //
14
+ // Exit codes: 0 = ran to completion (verdicts are data), 2 = malformed
15
+ // corpus/args or harness crash. Per-step timeout default 30s.
16
+
17
+ import { spawn, execSync } from 'node:child_process';
18
+ import fs from 'node:fs';
19
+ import os from 'node:os';
20
+ import path from 'node:path';
21
+ import { fileURLToPath } from 'node:url';
22
+ import yaml from 'yaml';
23
+
24
+ import {
25
+ UNKNOWN,
26
+ computeExternalWakeRate,
27
+ computeQualityScore,
28
+ summarizeRuns,
29
+ } from './decision-runtime-metrics.mjs';
30
+
31
+ const REPO_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..');
32
+ const DEFAULT_RUNS = 9;
33
+ const DEFAULT_TIMEOUT_MS = 30_000;
34
+ const ORACLE_TYPES = ['exit-code', 'output-match', 'file-exists'];
35
+
36
+ const USAGE = `Usage: node scripts/bench/decision-runtime-baseline.mjs --corpus <file.yaml> --out <file.json> [options]
37
+
38
+ Replays the DR-01b synthetic corpus and writes a schema-valid JSON report
39
+ (SPEC §5 FR-006). Verdicts are data: a failing scenario does not fail the run.
40
+
41
+ Options:
42
+ --corpus <file.yaml> Corpus path (required)
43
+ --out <file.json> Report output path (required)
44
+ --runs <n> Repetitions per scenario (default ${DEFAULT_RUNS})
45
+ --timeout <ms> Per-step timeout (default ${DEFAULT_TIMEOUT_MS})
46
+ --help Show this help
47
+
48
+ Exit codes: 0 = completed, 2 = malformed corpus/args or crash.`;
49
+
50
+ // ---------- args ----------
51
+
52
+ function parseArgs(argv) {
53
+ const opts = { runs: DEFAULT_RUNS, timeoutMs: DEFAULT_TIMEOUT_MS };
54
+ for (let i = 0; i < argv.length; i += 1) {
55
+ const arg = argv[i];
56
+ if (arg === '--help' || arg === '-h') return { help: true };
57
+ const takeValue = () => {
58
+ i += 1;
59
+ if (i >= argv.length) throw new Error(`Missing value for ${arg}`);
60
+ return argv[i];
61
+ };
62
+ if (arg === '--corpus') opts.corpus = takeValue();
63
+ else if (arg === '--out') opts.out = takeValue();
64
+ else if (arg === '--runs') opts.runs = Number(takeValue());
65
+ else if (arg === '--timeout') opts.timeoutMs = Number(takeValue());
66
+ else throw new Error(`Unknown argument: ${arg}`);
67
+ }
68
+ return opts;
69
+ }
70
+
71
+ // ---------- corpus loading + validation ----------
72
+
73
+ function isNonEmptyString(value) {
74
+ return typeof value === 'string' && value.trim() !== '';
75
+ }
76
+
77
+ function isNonNegativeInt(value) {
78
+ return typeof value === 'number' && Number.isInteger(value) && value >= 0;
79
+ }
80
+
81
+ /** Validate a parsed corpus doc; returns problem strings (empty = valid). */
82
+ function validateCorpus(doc) {
83
+ const problems = [];
84
+ if (!doc || typeof doc !== 'object' || Array.isArray(doc)) {
85
+ return ['corpus is not an object'];
86
+ }
87
+ if (doc.version !== 1) problems.push('missing or invalid `version` (must be 1)');
88
+ const scenarios = doc.scenarios;
89
+ if (!Array.isArray(scenarios) || scenarios.length === 0) {
90
+ problems.push('`scenarios` must be a non-empty array');
91
+ return problems;
92
+ }
93
+ const seen = new Set();
94
+ scenarios.forEach((sc, idx) => {
95
+ const where = `scenarios[${idx}]`;
96
+ if (!sc || typeof sc !== 'object') {
97
+ problems.push(`${where}: not an object`);
98
+ return;
99
+ }
100
+ const label = isNonEmptyString(sc.id) ? `scenario ${sc.id}` : where;
101
+ if (!isNonEmptyString(sc.id)) problems.push(`${where}: missing required field \`id\``);
102
+ else if (seen.has(sc.id)) problems.push(`duplicate scenario id: ${sc.id}`);
103
+ else seen.add(sc.id);
104
+ if (!isNonEmptyString(sc.class)) problems.push(`${label}: missing required field \`class\``);
105
+ if (!isNonEmptyString(sc.description)) problems.push(`${label}: missing required field \`description\``);
106
+ if (!Array.isArray(sc.steps) || sc.steps.length === 0) {
107
+ problems.push(`${label}: \`steps\` must be a non-empty array`);
108
+ } else {
109
+ sc.steps.forEach((step, sIdx) => {
110
+ if (!step || typeof step !== 'object' || !isNonEmptyString(step.run)) {
111
+ problems.push(`${label} steps[${sIdx}]: missing required field \`run\``);
112
+ } else if (step.expectExit !== undefined && !Number.isInteger(step.expectExit)) {
113
+ problems.push(`${label} steps[${sIdx}]: \`expectExit\` must be an integer`);
114
+ }
115
+ });
116
+ }
117
+ const oracle = sc.oracle;
118
+ if (!oracle || typeof oracle !== 'object' || !ORACLE_TYPES.includes(oracle.type)) {
119
+ problems.push(`${label}: \`oracle.type\` must be one of ${ORACLE_TYPES.join('|')}`);
120
+ } else if (oracle.type === 'exit-code' && !Number.isInteger(oracle.expect)) {
121
+ problems.push(`${label}: exit-code oracle requires integer \`expect\``);
122
+ } else if (oracle.type === 'output-match' && !isNonEmptyString(oracle.pattern)) {
123
+ problems.push(`${label}: output-match oracle requires \`pattern\``);
124
+ } else if (oracle.type === 'file-exists' && !isNonEmptyString(oracle.path)) {
125
+ problems.push(`${label}: file-exists oracle requires \`path\``);
126
+ }
127
+ if (!isNonNegativeInt(sc.eligibleTransitions)) {
128
+ problems.push(`${label}: missing required field \`eligibleTransitions\` (non-negative integer)`);
129
+ }
130
+ if (!isNonNegativeInt(sc.expectedExternalWakes)) {
131
+ problems.push(`${label}: missing required field \`expectedExternalWakes\` (non-negative integer)`);
132
+ }
133
+ });
134
+ return problems;
135
+ }
136
+
137
+ function loadCorpus(corpusPath) {
138
+ if (!fs.existsSync(corpusPath)) {
139
+ throw new Error(`corpus not found: ${corpusPath}`);
140
+ }
141
+ let doc;
142
+ try {
143
+ doc = yaml.parse(fs.readFileSync(corpusPath, 'utf8'));
144
+ } catch (error) {
145
+ throw new Error(`corpus YAML parse error: ${error.message}`);
146
+ }
147
+ const problems = validateCorpus(doc);
148
+ if (problems.length > 0) {
149
+ throw new Error(`malformed corpus:\n - ${problems.join('\n - ')}`);
150
+ }
151
+ return doc;
152
+ }
153
+
154
+ // ---------- step execution ----------
155
+
156
+ const SHELL = process.platform === 'win32'
157
+ ? { cmd: 'cmd.exe', args: ['/d', '/s', '/c'] }
158
+ : { cmd: '/bin/sh', args: ['-c'] };
159
+
160
+ /** Run one inline step command; captures output for output-match oracles. */
161
+ function runStep(run, { cwd, timeoutMs }) {
162
+ return new Promise((resolve) => {
163
+ const started = Date.now();
164
+ let timedOut = false;
165
+ let spawnError = null;
166
+ let output = '';
167
+ let child;
168
+ try {
169
+ // Bounded child: `timeout` arms the spawn-level deadline (SIGTERM) and
170
+ // the watchdog timer below escalates to SIGKILL — a wedged step can
171
+ // never hang the runner.
172
+ child = spawn(SHELL.cmd, [...SHELL.args, run], {
173
+ cwd,
174
+ env: process.env,
175
+ stdio: ['ignore', 'pipe', 'pipe'],
176
+ timeout: timeoutMs,
177
+ });
178
+ } catch (error) {
179
+ resolve({ exitCode: -1, durationMs: Date.now() - started, timedOut: false, error: String(error?.message ?? error), output: '' });
180
+ return;
181
+ }
182
+ // Watchdog: escalates the spawn timeout's SIGTERM to SIGKILL so a step
183
+ // that ignores termination still dies at the deadline.
184
+ const timer = setTimeout(() => {
185
+ timedOut = true;
186
+ child.kill('SIGKILL');
187
+ }, timeoutMs);
188
+ child.stdout.on('data', (chunk) => { output += chunk; });
189
+ child.stderr.on('data', (chunk) => { output += chunk; });
190
+ child.on('error', (error) => { spawnError = error; });
191
+ child.on('close', (code) => {
192
+ clearTimeout(timer);
193
+ resolve({
194
+ exitCode: timedOut || code == null ? -1 : code,
195
+ durationMs: Date.now() - started,
196
+ timedOut,
197
+ error: spawnError ? String(spawnError.message ?? spawnError) : null,
198
+ output,
199
+ });
200
+ });
201
+ });
202
+ }
203
+
204
+ // ---------- oracle ----------
205
+
206
+ function checkOracle(oracle, { lastExit, output, cwd }) {
207
+ switch (oracle.type) {
208
+ case 'exit-code':
209
+ return lastExit === oracle.expect;
210
+ case 'output-match':
211
+ return output.includes(oracle.pattern);
212
+ case 'file-exists':
213
+ return fs.existsSync(path.resolve(cwd, oracle.path));
214
+ default:
215
+ return false;
216
+ }
217
+ }
218
+
219
+ // ---------- scenario replay ----------
220
+
221
+ /**
222
+ * Replay one scenario `runs` times. Each run executes steps in order and stops
223
+ * at the first step whose exit code differs from `expectExit` (default 0) or
224
+ * that times out — a crash mid-operation means later declared-eligible
225
+ * transitions are never reached (coverage gap, metric-spec §1).
226
+ */
227
+ async function replayScenario(scenario, { runs, timeoutMs, workDir }) {
228
+ const durations = [];
229
+ let oraclePasses = 0;
230
+ let forbiddenFailures = 0;
231
+ let reachedTransitions = 0;
232
+ let stepMismatches = 0;
233
+ let stepTimeouts = 0;
234
+
235
+ for (let r = 0; r < runs; r += 1) {
236
+ const runDir = path.join(workDir, `run-${r}`);
237
+ fs.mkdirSync(runDir, { recursive: true });
238
+ const started = Date.now();
239
+ let lastExit = -1;
240
+ let output = '';
241
+ let stepsOk = true;
242
+ let reached = 0;
243
+
244
+ for (const step of scenario.steps) {
245
+ const expected = step.expectExit ?? 0;
246
+ const result = await runStep(step.run, { cwd: runDir, timeoutMs });
247
+ lastExit = result.exitCode;
248
+ output += result.output;
249
+ if (result.timedOut) stepTimeouts += 1;
250
+ if (result.exitCode !== expected || result.timedOut || result.error) {
251
+ stepsOk = false;
252
+ stepMismatches += 1;
253
+ break;
254
+ }
255
+ reached += 1;
256
+ }
257
+
258
+ durations.push(Date.now() - started);
259
+ reachedTransitions += Math.min(reached, scenario.eligibleTransitions);
260
+
261
+ const oraclePass = checkOracle(scenario.oracle, { lastExit, output, cwd: runDir });
262
+ if (oraclePass) oraclePasses += 1;
263
+ // Forbidden failure (metric-spec §2): every step completed as declared
264
+ // (exit-code success) while the oracle rejects the run — false completion.
265
+ if (stepsOk && !oraclePass) forbiddenFailures += 1;
266
+ }
267
+
268
+ const quality = computeQualityScore({
269
+ satisfied: oraclePasses,
270
+ eligible: runs,
271
+ forbiddenFailures,
272
+ });
273
+ const { p50, p95 } = summarizeRuns(durations);
274
+ const coverageGap = Math.max(0, scenario.eligibleTransitions * runs - reachedTransitions);
275
+
276
+ return {
277
+ entry: {
278
+ id: scenario.id,
279
+ class: scenario.class,
280
+ verdict: quality.verdict,
281
+ wallMs: { n: runs, p50, p95 },
282
+ wakes: 0,
283
+ eligible: reachedTransitions,
284
+ coverageGap,
285
+ quality: {
286
+ score: quality.score,
287
+ satisfied: oraclePasses,
288
+ eligible: runs,
289
+ forbiddenFailures,
290
+ },
291
+ tokens: UNKNOWN,
292
+ stepMismatches,
293
+ stepTimeouts,
294
+ },
295
+ durations,
296
+ };
297
+ }
298
+
299
+ // ---------- env ----------
300
+
301
+ function detectFsType(dir) {
302
+ try {
303
+ if (typeof fs.statfsSync === 'function') {
304
+ const stats = fs.statfsSync(dir);
305
+ if (stats && typeof stats.type === 'number') {
306
+ return `0x${stats.type.toString(16)}`;
307
+ }
308
+ }
309
+ } catch { /* fall through */ }
310
+ return 'local';
311
+ }
312
+
313
+ function collectEnv() {
314
+ const env = {
315
+ sha: 'unknown',
316
+ package: 'unknown',
317
+ node: process.version,
318
+ os: process.platform,
319
+ arch: process.arch,
320
+ fs: detectFsType(REPO_ROOT),
321
+ };
322
+ try {
323
+ env.sha = execSync('git rev-parse HEAD', { cwd: REPO_ROOT, encoding: 'utf8', timeout: 10_000 }).trim();
324
+ } catch { /* leave 'unknown' */ }
325
+ try {
326
+ env.package = JSON.parse(fs.readFileSync(path.join(REPO_ROOT, 'package.json'), 'utf8')).version ?? 'unknown';
327
+ } catch { /* leave 'unknown' */ }
328
+ return env;
329
+ }
330
+
331
+ // ---------- main ----------
332
+
333
+ async function main() {
334
+ let opts;
335
+ try {
336
+ opts = parseArgs(process.argv.slice(2));
337
+ } catch (error) {
338
+ console.error(`[decision-runtime-baseline] ${error.message}\n\n${USAGE}`);
339
+ process.exit(2);
340
+ }
341
+ if (opts.help) {
342
+ console.log(USAGE);
343
+ process.exit(0);
344
+ }
345
+ if (!opts.corpus || !opts.out) {
346
+ console.error(`[decision-runtime-baseline] --corpus and --out are required.\n\n${USAGE}`);
347
+ process.exit(2);
348
+ }
349
+ if (!Number.isInteger(opts.runs) || opts.runs < 1) {
350
+ console.error('[decision-runtime-baseline] --runs must be a positive integer.');
351
+ process.exit(2);
352
+ }
353
+ if (!Number.isFinite(opts.timeoutMs) || opts.timeoutMs < 1) {
354
+ console.error('[decision-runtime-baseline] --timeout must be a positive number of ms.');
355
+ process.exit(2);
356
+ }
357
+
358
+ let corpus;
359
+ try {
360
+ corpus = loadCorpus(path.resolve(opts.corpus));
361
+ } catch (error) {
362
+ console.error(`[decision-runtime-baseline] ${error.message}`);
363
+ process.exit(2);
364
+ }
365
+
366
+ const tmpRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'ukit-dr-baseline-'));
367
+ const scenarios = [];
368
+ const allDurations = [];
369
+ for (const scenario of corpus.scenarios) {
370
+ const workDir = path.join(tmpRoot, scenario.id);
371
+ fs.mkdirSync(workDir, { recursive: true });
372
+ const { entry, durations } = await replayScenario(scenario, {
373
+ runs: opts.runs,
374
+ timeoutMs: opts.timeoutMs,
375
+ workDir,
376
+ });
377
+ scenarios.push(entry);
378
+ allDurations.push(...durations);
379
+ }
380
+
381
+ const totals = scenarios.reduce((acc, sc) => ({
382
+ wakes: acc.wakes + sc.wakes,
383
+ eligible: acc.eligible + sc.eligible,
384
+ satisfied: acc.satisfied + sc.quality.satisfied,
385
+ qualityEligible: acc.qualityEligible + sc.quality.eligible,
386
+ forbiddenFailures: acc.forbiddenFailures + sc.quality.forbiddenFailures,
387
+ }), { wakes: 0, eligible: 0, satisfied: 0, qualityEligible: 0, forbiddenFailures: 0 });
388
+
389
+ const wallSummary = summarizeRuns(allDurations);
390
+ const summaryQuality = computeQualityScore({
391
+ satisfied: totals.satisfied,
392
+ eligible: totals.qualityEligible,
393
+ forbiddenFailures: totals.forbiddenFailures,
394
+ });
395
+
396
+ const report = {
397
+ version: 1,
398
+ env: collectEnv(),
399
+ corpus: { path: path.resolve(opts.corpus), version: corpus.version, scenarios: corpus.scenarios.length },
400
+ runs: opts.runs,
401
+ scenarios,
402
+ summary: {
403
+ externalWakeRate: computeExternalWakeRate({
404
+ externalWakes: totals.wakes,
405
+ eligibleTransitions: totals.eligible,
406
+ }),
407
+ qualityScore: summaryQuality.score,
408
+ qualityVerdict: summaryQuality.verdict,
409
+ p50: wallSummary.p50,
410
+ p95: wallSummary.p95,
411
+ },
412
+ };
413
+
414
+ const outPath = path.resolve(opts.out);
415
+ fs.mkdirSync(path.dirname(outPath), { recursive: true });
416
+ fs.writeFileSync(outPath, `${JSON.stringify(report, null, 2)}\n`);
417
+
418
+ console.log(`[decision-runtime-baseline] report written: ${outPath}`);
419
+ console.log(`[decision-runtime-baseline] scenarios=${scenarios.length} runs=${opts.runs} `
420
+ + `externalWakeRate=${report.summary.externalWakeRate} quality=${report.summary.qualityScore}`);
421
+ process.exit(0);
422
+ }
423
+
424
+ main().catch((error) => {
425
+ console.error(`[decision-runtime-baseline] harness crash: ${error?.stack ?? error}`);
426
+ process.exit(2);
427
+ });
@@ -0,0 +1,67 @@
1
+ // TASK-003 — DR-01a metric computation module (SPEC §5 FR-004, §8, §10).
2
+ //
3
+ // Pure ESM, `node:` builtins only (in fact no imports at all). The baseline
4
+ // runner (TASK-004) and every later gate import these functions so all metric
5
+ // math is computed by exactly one implementation — see
6
+ // docs/AI_HANDOFF/benchmark/metric-spec.md for the pre-registered definitions.
7
+ //
8
+ // UNKNOWN semantics: the literal string 'UNKNOWN' is returned wherever the
9
+ // denominator or source data is absent/invalid. It is never 0, NaN, Infinity,
10
+ // or an estimate — a string cannot silently coerce into numeric aggregates.
11
+
12
+ export const UNKNOWN = 'UNKNOWN';
13
+
14
+ function isValidCount(value) {
15
+ return typeof value === 'number' && Number.isFinite(value) && value >= 0;
16
+ }
17
+
18
+ /**
19
+ * external_wake_rate = external_reasoning_continuations / eligible_known_transitions.
20
+ *
21
+ * @param {{externalWakes: number, eligibleTransitions: number}} input
22
+ * @returns {number|'UNKNOWN'} ratio in [0,1], or 'UNKNOWN' when the denominator
23
+ * is <= 0 or either input is not a finite non-negative number.
24
+ */
25
+ export function computeExternalWakeRate({ externalWakes, eligibleTransitions } = {}) {
26
+ if (!isValidCount(externalWakes) || !isValidCount(eligibleTransitions)) return UNKNOWN;
27
+ if (eligibleTransitions <= 0) return UNKNOWN;
28
+ // wakes > eligible means instrumentation error upstream; clamp to keep the
29
+ // contract range [0,1] rather than emitting an impossible rate.
30
+ return Math.min(1, externalWakes / eligibleTransitions);
31
+ }
32
+
33
+ /**
34
+ * Quality score = satisfied / eligible requirements; verdict is 'fail' when any
35
+ * forbidden failure occurred (false completion, fabricated support, estimated
36
+ * metric) regardless of score, and fail-closed when the score is UNKNOWN.
37
+ *
38
+ * @param {{satisfied: number, eligible: number, forbiddenFailures: number}} input
39
+ * @returns {{score: number|'UNKNOWN', verdict: 'pass'|'fail'}}
40
+ */
41
+ export function computeQualityScore({ satisfied, eligible, forbiddenFailures } = {}) {
42
+ const forbidden = isValidCount(forbiddenFailures) ? forbiddenFailures : 0;
43
+ const verdict = forbidden > 0 ? 'fail' : 'pass';
44
+ if (!isValidCount(satisfied) || !isValidCount(eligible) || eligible <= 0) {
45
+ // Fail-closed: an unmeasurable quality score never passes.
46
+ return { score: UNKNOWN, verdict: 'fail' };
47
+ }
48
+ return { score: Math.min(1, satisfied / eligible), verdict };
49
+ }
50
+
51
+ /**
52
+ * Nearest-rank percentiles over run durations (same convention as
53
+ * scripts/bench/memory-baseline.mjs). Non-finite or negative entries are
54
+ * dropped; `n` counts valid durations only.
55
+ *
56
+ * @param {number[]} durations
57
+ * @returns {{n: number, p50: number|'UNKNOWN', p95: number|'UNKNOWN'}}
58
+ */
59
+ export function summarizeRuns(durations) {
60
+ const valid = Array.isArray(durations)
61
+ ? durations.filter((d) => typeof d === 'number' && Number.isFinite(d) && d >= 0)
62
+ : [];
63
+ if (valid.length === 0) return { n: 0, p50: UNKNOWN, p95: UNKNOWN };
64
+ const sorted = [...valid].sort((a, b) => a - b);
65
+ const rank = (q) => sorted[Math.max(1, Math.ceil((q / 100) * sorted.length)) - 1];
66
+ return { n: sorted.length, p50: rank(50), p95: rank(95) };
67
+ }