@ngockhoale/ukit 3.0.7 → 3.0.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/CHANGELOG.md +18 -1
  2. package/manifests/documentation.yaml +11 -0
  3. package/package.json +1 -1
  4. package/scripts/audit/decision-coverage.mjs +29 -2
  5. package/scripts/bench/data-foundation.mjs +52 -3
  6. package/scripts/bench/decision-runtime-baseline.mjs +427 -0
  7. package/scripts/bench/decision-runtime-metrics.mjs +67 -0
  8. package/scripts/bench/decision-runtime-variant.mjs +626 -0
  9. package/scripts/bench/memory-ablation.mjs +495 -0
  10. package/scripts/bench/memory-baseline.mjs +596 -0
  11. package/scripts/bench/memory-bench.mjs +661 -0
  12. package/scripts/bench/memory-canary.mjs +321 -0
  13. package/scripts/bench/memory-corpus.mjs +354 -0
  14. package/scripts/bench/memory-gate.mjs +389 -0
  15. package/scripts/bench/memory-metrics.mjs +179 -0
  16. package/scripts/bench/parallel-agents.mjs +33 -11
  17. package/scripts/bench/recorder-overhead.mjs +204 -0
  18. package/scripts/bench/sqlite-spike.mjs +451 -0
  19. package/scripts/measure-decision-gateway.mjs +306 -0
  20. package/scripts/perf/audit-perf.mjs +35 -17
  21. package/src/bug/triageBug.js +4 -3
  22. package/src/cli/commands/memory.js +357 -63
  23. package/src/context/detectProjectContext.js +11 -1
  24. package/src/core/agentRuntime/adapters.js +254 -0
  25. package/src/core/agentRuntime/artifacts.js +192 -0
  26. package/src/core/agentRuntime/completionGate.js +176 -0
  27. package/src/core/agentRuntime/context.js +149 -0
  28. package/src/core/agentRuntime/contract.js +247 -0
  29. package/src/core/agentRuntime/diagnostics.js +244 -0
  30. package/src/core/agentRuntime/evaluation.js +163 -0
  31. package/src/core/agentRuntime/eventStore.js +404 -0
  32. package/src/core/agentRuntime/liveness.js +60 -0
  33. package/src/core/agentRuntime/planCompiler.js +322 -0
  34. package/src/core/agentRuntime/promotion.js +53 -0
  35. package/src/core/agentRuntime/qualityComparison.js +112 -0
  36. package/src/core/agentRuntime/recovery.js +266 -0
  37. package/src/core/agentRuntime/resourcePolicy.js +78 -0
  38. package/src/core/agentRuntime/runtimeSupport.js +237 -0
  39. package/src/core/agentRuntime/supervisor.js +565 -0
  40. package/src/core/agentRuntime/vmEngine.js +621 -0
  41. package/src/core/codeintel/analogy.js +3 -2
  42. package/src/core/experiments/dynamicWorkflow.js +17 -2
  43. package/src/core/fileOps.js +21 -3
  44. package/src/core/memory/deltaOverlays.js +75 -30
  45. package/src/core/memory/learningCandidates.js +93 -48
  46. package/src/core/memory/memoryFlags.js +83 -0
  47. package/src/core/memory/memoryFreshness.js +190 -0
  48. package/src/core/memory/memoryHit.js +144 -0
  49. package/src/core/memory/migrate.js +69 -189
  50. package/src/core/memory/migrateMapping.js +232 -0
  51. package/src/core/memory/mutateMemory.js +323 -0
  52. package/src/core/memory/policy.js +96 -0
  53. package/src/core/memory/projectIdentity.js +266 -0
  54. package/src/core/memory/recordIndex.js +178 -0
  55. package/src/core/memory/recordStore.js +133 -20
  56. package/src/core/memory/records.js +144 -6
  57. package/src/core/memory/retrieval.js +259 -125
  58. package/src/core/memory/store.js +16 -5
  59. package/src/core/memory/storeBackup.js +226 -0
  60. package/src/core/memory/storeV2.js +63 -26
  61. package/src/core/memory/storeV2Loader.js +30 -12
  62. package/src/core/memory/userMemory.js +38 -20
  63. package/src/core/memory/writeClassification.js +161 -0
  64. package/src/core/memory/writeGuard.js +129 -0
  65. package/src/core/observability/adapters/hookTelemetryAdapter.js +90 -0
  66. package/src/core/observability/analytics/cohorts.js +148 -0
  67. package/src/core/observability/analytics/storeDigest.js +163 -0
  68. package/src/core/observability/evaluation/experimentPlan.js +95 -0
  69. package/src/core/observability/evaluation/findings.js +99 -0
  70. package/src/core/observability/evaluation/optimizationKnowledge.js +10 -1
  71. package/src/core/observability/evaluation/perturbation.js +273 -0
  72. package/src/core/observability/evaluation/replay.js +7 -1
  73. package/src/core/observability/evaluation/scorecard.js +23 -3
  74. package/src/core/observability/rollout.js +11 -7
  75. package/src/core/observability/schema/compatibility.js +135 -0
  76. package/src/core/observability/schema/registry.js +99 -0
  77. package/src/core/observability/schema/validate.js +7 -0
  78. package/src/core/observability/support/import.js +53 -9
  79. package/src/core/observability/support/paths.js +13 -3
  80. package/src/core/observability/support/projector.js +148 -12
  81. package/src/core/output/index.js +12 -2
  82. package/src/core/runtimeConfig.js +83 -0
  83. package/src/core/runtimePaths.js +3 -0
  84. package/src/core/sensitiveValueScanner.js +40 -0
  85. package/src/core/token/index.js +40 -3
  86. package/src/decision/client.js +37 -13
  87. package/src/decision/protocol.js +1 -1
  88. package/src/decision/registry.js +5 -3
  89. package/src/decision/runtimeDecide.js +242 -0
  90. package/src/decision/runtimeFilter.js +150 -0
  91. package/src/decision/runtimeScheduler.js +239 -0
  92. package/src/index/buildIndex.js +13 -12
  93. package/src/index/queryIndex.js +35 -14
  94. package/src/index/relatedTests.js +50 -8
  95. package/src/index/resolveContext.js +9 -4
  96. package/src/manifest/selectItems.js +7 -3
  97. package/src/render/instructionRenderer.js +17 -5
  98. package/template_project/.claude/ukit/index/lib/index-core.mjs +94 -39
  99. package/template_project/.claude/ukit/index/route-task.mjs +121 -19
  100. package/template_project/.claude/ukit/index/unic-decision.mjs +28 -13
  101. package/template_project/.claude/ukit/runtime/memory-flags.mjs +51 -0
  102. package/template_project/.claude/ukit/runtime/memory-freshness.mjs +155 -0
  103. package/template_project/.claude/ukit/runtime/memory-policy.mjs +286 -0
  104. package/template_project/.claude/ukit/runtime/output-compression.mjs +3 -0
  105. package/template_project/.claude/ukit/runtime/reinject-context.mjs +145 -14
@@ -0,0 +1,389 @@
1
+ #!/usr/bin/env node
2
+ // TASK-006 — G3 decision gate (SPEC §5 FR-011/FR-012, M03-04).
3
+ //
4
+ // Consumes the bench report (TASK-003/004) and the SQLite spike report
5
+ // (TASK-005), applies the signed G3 decision rule, and writes the ADR +
6
+ // machine-readable verdict to docs/AI_HANDOFF/inventory/ (or --out).
7
+ //
8
+ // Decision rule (blueprint §W3/G3, SPEC FR-011): `sqlite` iff ALL of —
9
+ // (a) the JSON/lexical variant demonstrably fails a signed threshold:
10
+ // correctness floor breach (scopeLeakRate != 0 OR falseAuthorityRate
11
+ // != 0 OR staleHitRate != 0 OR precisionAt5 < 0.9) OR medium-tier
12
+ // CLI-search p95 > 150ms OR write-burst lostUpdates > 0;
13
+ // (b) the spike has no `failed` cell on availability / WAL / backup /
14
+ // migration;
15
+ // (c) the sqlite-fts5 variant shows a meaningful gain over lexical —
16
+ // strictly higher precisionAt5 or recallAt5, or strictly lower p95.
17
+ // Otherwise `json`. Inconclusive evidence (missing/UNKNOWN metrics,
18
+ // complete:false, missing spike cells) resolves to `json` with each gap
19
+ // recorded in `inconclusiveGap`.
20
+ //
21
+ // Exit codes: 0 = verdict written (the decision is data), 1 = usage error,
22
+ // 2 = harness crash.
23
+
24
+ import fs from 'node:fs';
25
+ import path from 'node:path';
26
+ import { fileURLToPath } from 'node:url';
27
+
28
+ const REPO_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..');
29
+ const DEFAULT_OUT = path.join(REPO_ROOT, 'docs/AI_HANDOFF/inventory/g3-verdict.md');
30
+
31
+ // FR-012 signed thresholds — emitted verbatim into the ADR.
32
+ export const THRESHOLDS = Object.freeze({
33
+ scopeLeakRate: 0,
34
+ falseAuthorityRate: 0,
35
+ staleHitRate: 0,
36
+ precisionAt5Min: 0.9,
37
+ cliSearchP95MsMax: 150,
38
+ writeBurstLostUpdatesMax: 0,
39
+ });
40
+
41
+ // Spike cells gated by condition (b).
42
+ const GATED_SPIKE_CELLS = [
43
+ 'node-sqlite-availability', 'wal-mode', 'backup', 'migration-roundtrip',
44
+ ];
45
+
46
+ const USAGE = `Usage: node scripts/bench/memory-gate.mjs --bench <report.json> --spike <spike.json> [--out <g3-verdict.md>]
47
+
48
+ Consumes the W3 bench report + SQLite spike report, applies the signed G3
49
+ decision rule (SPEC FR-011), and writes the ADR (<out>.md) plus a sibling
50
+ verdict JSON (<out>.json).
51
+
52
+ --bench <f> Bench report JSON from memory-bench.mjs (required).
53
+ --spike <f> Spike report JSON from sqlite-spike.mjs (required).
54
+ --out <f> ADR path (default ${path.relative(REPO_ROOT, DEFAULT_OUT)}); the
55
+ sibling .json verdict is always written next to it.
56
+
57
+ Exit codes: 0 = verdict written, 1 = usage error, 2 = crash.`;
58
+
59
+ function parseArgs(argv) {
60
+ const opts = { out: DEFAULT_OUT };
61
+ for (let i = 0; i < argv.length; i += 1) {
62
+ const arg = argv[i];
63
+ if (arg === '--help' || arg === '-h') return { help: true };
64
+ const takeValue = () => {
65
+ i += 1;
66
+ if (i >= argv.length) throw new Error(`Missing value for ${arg}`);
67
+ return argv[i];
68
+ };
69
+ if (arg === '--bench') opts.bench = takeValue();
70
+ else if (arg === '--spike') opts.spike = takeValue();
71
+ else if (arg === '--out') opts.out = takeValue();
72
+ else throw new Error(`Unknown argument: ${arg}`);
73
+ }
74
+ return opts;
75
+ }
76
+
77
+ const isNum = (v) => typeof v === 'number' && Number.isFinite(v);
78
+
79
+ /** CLI-search p95: prefer warm, fall back to cold. */
80
+ function cliP95(bench) {
81
+ const warm = bench?.surfaces?.cli?.warm?.p95;
82
+ if (isNum(warm)) return { value: warm, source: 'cli.warm' };
83
+ const cold = bench?.surfaces?.cli?.cold?.p95;
84
+ if (isNum(cold)) return { value: cold, source: 'cli.cold' };
85
+ return { value: 'UNKNOWN', source: 'cli' };
86
+ }
87
+
88
+ /**
89
+ * Evaluate the G3 rule. Returns
90
+ * {decision, rationale[], inconclusiveGap[], evidence:{breaches[],
91
+ * spikeFailedCells[], ftsGain}}.
92
+ */
93
+ export function evaluateGate(bench, spike) {
94
+ const rationale = [];
95
+ const gaps = [];
96
+ const breaches = [];
97
+
98
+ if (bench?.complete === false) {
99
+ gaps.push('bench report marked complete:false (interrupted/failed probes)');
100
+ }
101
+
102
+ // ---- condition (a): JSON/lexical demonstrably fails a signed threshold ----
103
+ const lexical = bench?.variants?.lexical;
104
+ if (!lexical || lexical.verdict !== 'measured') {
105
+ gaps.push('lexical variant missing or not measured in bench report');
106
+ } else {
107
+ const c = lexical.correctness ?? {};
108
+ if (!isNum(c.scopeLeakRate)) gaps.push('lexical scopeLeakRate UNKNOWN/missing');
109
+ else if (c.scopeLeakRate !== THRESHOLDS.scopeLeakRate) {
110
+ breaches.push(`scopeLeakRate=${c.scopeLeakRate} breaches floor ${THRESHOLDS.scopeLeakRate}`);
111
+ }
112
+ if (!isNum(c.falseAuthorityRate)) gaps.push('lexical falseAuthorityRate UNKNOWN/missing');
113
+ else if (c.falseAuthorityRate !== THRESHOLDS.falseAuthorityRate) {
114
+ breaches.push(`falseAuthorityRate=${c.falseAuthorityRate} breaches floor ${THRESHOLDS.falseAuthorityRate}`);
115
+ }
116
+ if (!isNum(c.staleHitRate)) gaps.push('lexical staleHitRate UNKNOWN/missing');
117
+ else if (c.staleHitRate !== THRESHOLDS.staleHitRate) {
118
+ breaches.push(`staleHitRate=${c.staleHitRate} breaches floor ${THRESHOLDS.staleHitRate}`);
119
+ }
120
+ if (!isNum(c.precisionAt5)) gaps.push('lexical precisionAt5 UNKNOWN/missing');
121
+ else if (c.precisionAt5 < THRESHOLDS.precisionAt5Min) {
122
+ breaches.push(`precisionAt5=${c.precisionAt5} below floor ${THRESHOLDS.precisionAt5Min}`);
123
+ }
124
+ }
125
+
126
+ const cli = cliP95(bench);
127
+ if (!isNum(cli.value)) {
128
+ gaps.push('CLI-search p95 UNKNOWN/missing (surfaces.cli warm+cold)');
129
+ } else if (cli.value > THRESHOLDS.cliSearchP95MsMax) {
130
+ breaches.push(`CLI-search p95=${cli.value}ms (${cli.source}) exceeds budget ${THRESHOLDS.cliSearchP95MsMax}ms`);
131
+ }
132
+
133
+ const wb = bench?.writeBurst;
134
+ if (!wb || !isNum(wb.lostUpdates)) {
135
+ gaps.push('writeBurst lostUpdates UNKNOWN/missing');
136
+ } else if (wb.lostUpdates > THRESHOLDS.writeBurstLostUpdatesMax) {
137
+ breaches.push(`write burst lostUpdates=${wb.lostUpdates} exceeds ${THRESHOLDS.writeBurstLostUpdatesMax}`);
138
+ }
139
+
140
+ const conditionA = breaches.length > 0;
141
+ rationale.push(conditionA
142
+ ? `(a) JSON/lexical fails signed threshold(s): ${breaches.join('; ')}`
143
+ : '(a) JSON/lexical meets every signed threshold — no demonstrable failure');
144
+
145
+ // ---- condition (b): no failed spike cell on availability/WAL/backup/migration ----
146
+ const spikeFailedCells = [];
147
+ const cells = spike?.cells ?? {};
148
+ for (const name of GATED_SPIKE_CELLS) {
149
+ const verdict = cells[name]?.verdict;
150
+ if (verdict === 'failed') spikeFailedCells.push(name);
151
+ else if (verdict === undefined) gaps.push(`spike cell '${name}' missing`);
152
+ else if (verdict === 'unsupported') gaps.push(`spike cell '${name}' unsupported: ${cells[name].detail}`);
153
+ }
154
+ const conditionB = spikeFailedCells.length === 0;
155
+ rationale.push(conditionB
156
+ ? '(b) spike shows no failed cell on availability/WAL/backup/migration'
157
+ : `(b) unmet: failed spike cell(s) ${spikeFailedCells.join(', ')}`);
158
+
159
+ // ---- condition (c): meaningful FTS5 gain over lexical ----
160
+ const fts = bench?.variants?.['sqlite-fts5'];
161
+ let ftsGain = null;
162
+ if (!fts || fts.verdict !== 'measured') {
163
+ gaps.push(`sqlite-fts5 variant missing or not measured (verdict=${fts?.verdict ?? 'absent'})`);
164
+ rationale.push('(c) unmet: sqlite-fts5 variant not measured — no gain evidence');
165
+ } else if (!lexical || lexical.verdict !== 'measured') {
166
+ gaps.push('cannot assess FTS5 gain: lexical baseline not measured');
167
+ rationale.push('(c) unmet: no lexical baseline to compare against');
168
+ } else {
169
+ const fp = fts.correctness?.precisionAt5;
170
+ const fr = fts.correctness?.recallAt5;
171
+ const lp = lexical.correctness?.precisionAt5;
172
+ const lr = lexical.correctness?.recallAt5;
173
+ const fp95 = fts.latency?.p95;
174
+ const lp95 = lexical.latency?.p95;
175
+ const gains = [];
176
+ if (isNum(fp) && isNum(lp) && fp > lp) gains.push(`precisionAt5 ${lp}→${fp}`);
177
+ if (isNum(fr) && isNum(lr) && fr > lr) gains.push(`recallAt5 ${lr}→${fr}`);
178
+ if (isNum(fp95) && isNum(lp95) && fp95 < lp95) gains.push(`p95 ${lp95}ms→${fp95}ms`);
179
+ if ((!isNum(fp) || !isNum(lp)) && (!isNum(fp95) || !isNum(lp95))) {
180
+ gaps.push('FTS5-vs-lexical comparison metrics UNKNOWN/missing');
181
+ }
182
+ ftsGain = gains.length > 0 ? gains : null;
183
+ rationale.push(ftsGain
184
+ ? `(c) FTS5 shows meaningful gain over lexical: ${gains.join('; ')}`
185
+ : '(c) unmet: no meaningful FTS5 quality or latency gain over lexical');
186
+ }
187
+
188
+ const conditionC = ftsGain !== null;
189
+ // Inconclusive evidence resolves to json: any recorded gap blocks sqlite
190
+ // even when all three conditions are met (FR-011).
191
+ const decision = conditionA && conditionB && conditionC && gaps.length === 0
192
+ ? 'sqlite' : 'json';
193
+ rationale.push(decision === 'sqlite'
194
+ ? 'decision=sqlite — all three conditions of FR-011 satisfied, no gaps'
195
+ : `decision=json — breaches=${conditionA} spikeClean=${conditionB} `
196
+ + `ftsGain=${conditionC} gaps=${gaps.length}; inconclusive evidence resolves to json`);
197
+
198
+ return {
199
+ decision,
200
+ rationale,
201
+ inconclusiveGap: gaps,
202
+ evidence: { breaches, spikeFailedCells, ftsGain, cliP95: cli },
203
+ };
204
+ }
205
+
206
+ // ---------- ADR rendering ----------
207
+
208
+ function renderAdr({ bench, spike, benchRef, spikeRef, result }) {
209
+ const v = result.decision;
210
+ const lines = [];
211
+ lines.push('# ADR — Memory storage/retrieval engine: guarded JSON vs SQLite+FTS5 (gate G3)');
212
+ lines.push('');
213
+ lines.push(`- Status: ${v === 'sqlite' ? 'accepted — migrate to SQLite+FTS5' : 'accepted — stay on guarded JSON'} (C59, TASK-006)`);
214
+ lines.push(`- Date: ${new Date().toISOString().slice(0, 10)}`);
215
+ lines.push('- Spec: SPEC §5 FR-011/FR-012, §7; blueprint §W3/G3');
216
+ lines.push(`- Evidence: bench \`${benchRef}\` · spike \`${spikeRef}\``);
217
+ lines.push('');
218
+ lines.push('## Context');
219
+ lines.push('');
220
+ lines.push('W1+W2 made ukit-memory reads policy-correct and writes guarded on the');
221
+ lines.push('JSON v2 store. Blueprint gate G3 requires measured evidence before');
222
+ lines.push('choosing the long-term storage/retrieval engine: guarded JSON +');
223
+ lines.push('lexical scan is the incumbent; SQLite+FTS5 is the challenger. The');
224
+ lines.push(`bench ran tier=\`${bench?.corpus?.tier ?? 'UNKNOWN'}\` `
225
+ + `(records=${bench?.corpus?.recordCount ?? 'UNKNOWN'}, `
226
+ + `seed=${bench?.corpus?.seed ?? 'UNKNOWN'}) on node \`${bench?.env?.node ?? '?'}\` `
227
+ + `${bench?.env?.os ?? '?'}/${bench?.env?.arch ?? '?'}.`);
228
+ lines.push('');
229
+ lines.push('Signed thresholds (FR-012, recorded verbatim):');
230
+ lines.push('');
231
+ lines.push(`- correctness floor: \`scopeLeakRate == ${THRESHOLDS.scopeLeakRate}\`, `
232
+ + `\`falseAuthorityRate == ${THRESHOLDS.falseAuthorityRate}\`, `
233
+ + `\`staleHitRate == ${THRESHOLDS.staleHitRate}\`, `
234
+ + `\`precisionAt5 ≥ ${THRESHOLDS.precisionAt5Min}\` on medium tier`);
235
+ lines.push(`- latency budget: CLI-search p95 ≤ ${THRESHOLDS.cliSearchP95MsMax}ms on medium tier `
236
+ + '(~3× the W0 small-corpus p95 of 56ms — provisional, first signed budget)');
237
+ lines.push(`- write burst: ${THRESHOLDS.writeBurstLostUpdatesMax} lost updates`);
238
+ lines.push('');
239
+ lines.push('## Options');
240
+ lines.push('');
241
+ lines.push('1. **Guarded JSON v2 + lexical scan (incumbent)** — single-file store,');
242
+ lines.push(' generation-guarded writes, in-process token-overlap retrieval.');
243
+ lines.push(' No new dependency, no migration surface.');
244
+ lines.push('2. **SQLite + FTS5** — embedded index over eligible text via');
245
+ lines.push(' `node:sqlite` (Node ≥22.5 unflagged; Node 20 needs');
246
+ lines.push(' `--experimental-sqlite`). Adds a file format, WAL semantics, and');
247
+ lines.push(' a migration path.');
248
+ lines.push('');
249
+ lines.push('Measured variants (bench report):');
250
+ lines.push('');
251
+ lines.push('| variant | verdict | p@5 | r@5 | scopeLeak | falseAuth | stale | p95 ms |');
252
+ lines.push('|---------|---------|-----|-----|-----------|-----------|-------|--------|');
253
+ for (const [name, entry] of Object.entries(bench?.variants ?? {})) {
254
+ const c = entry.correctness ?? {};
255
+ const fmt = (x) => (isNum(x) ? String(Math.round(x * 1000) / 1000) : (x ?? 'n/a'));
256
+ lines.push(`| ${name} | ${entry.verdict} | ${fmt(c.precisionAt5)} | ${fmt(c.recallAt5)} `
257
+ + `| ${fmt(c.scopeLeakRate)} | ${fmt(c.falseAuthorityRate)} | ${fmt(c.staleHitRate)} `
258
+ + `| ${fmt(entry.latency?.p95)} |`);
259
+ }
260
+ lines.push('');
261
+ const cli = result.evidence.cliP95;
262
+ const wb = bench?.writeBurst ?? {};
263
+ lines.push(`CLI-search p95 (${cli.source}): \`${cli.value}\` ms · `
264
+ + `write burst: expected=${wb.expected ?? 'n/a'} persisted=${wb.persisted ?? 'n/a'} `
265
+ + `lostUpdates=${wb.lostUpdates ?? 'n/a'} verdict=${wb.verdict ?? 'n/a'}`);
266
+ lines.push('');
267
+ const s = spike?.summary ?? {};
268
+ lines.push(`Spike cells: tested=${s.tested ?? 'n/a'} unsupported=${s.unsupported ?? 'n/a'} `
269
+ + `failed=${s.failed ?? 'n/a'} of ${s.total ?? 'n/a'}`);
270
+ if (result.evidence.breaches.length > 0) {
271
+ const measured = Object.values(bench?.variants ?? {}).filter((e) => e?.verdict === 'measured');
272
+ const ref = measured[0]?.correctness ?? {};
273
+ const identical = measured.length > 1 && measured.every((e) => {
274
+ const c = e.correctness ?? {};
275
+ return c.scopeLeakRate === ref.scopeLeakRate
276
+ && c.falseAuthorityRate === ref.falseAuthorityRate
277
+ && c.staleHitRate === ref.staleHitRate
278
+ && c.precisionAt5 === ref.precisionAt5;
279
+ });
280
+ if (identical) {
281
+ lines.push('Note: the correctness breaches above are identical across all');
282
+ lines.push('measured variants — they originate in the shared `eligible()`');
283
+ lines.push('policy/corpus labeling, not in the JSON storage layer. They are');
284
+ lines.push('recorded as a W4 policy-fix item, not as evidence for SQLite.');
285
+ lines.push('');
286
+ }
287
+ }
288
+ lines.push('');
289
+ lines.push('## Decision');
290
+ lines.push('');
291
+ lines.push(`**${v === 'sqlite' ? 'Option 2 — SQLite+FTS5' : 'Option 1 — stay on guarded JSON'}** `
292
+ + `(verdict \`${v}\`, ruleVersion 1). Rule evaluation:`);
293
+ lines.push('');
294
+ for (const r of result.rationale) lines.push(`- ${r}`);
295
+ lines.push('');
296
+ if (result.inconclusiveGap.length > 0) {
297
+ lines.push('Inconclusive gaps (resolved to `json` per FR-011):');
298
+ lines.push('');
299
+ for (const g of result.inconclusiveGap) lines.push(`- ${g}`);
300
+ lines.push('');
301
+ }
302
+ lines.push('## Reversal trigger');
303
+ lines.push('');
304
+ if (v === 'sqlite') {
305
+ lines.push('Revisit if the SQLite path shows a `failed` spike cell on a');
306
+ lines.push('supported engine matrix (Node 20 flag gating, network-FS WAL),');
307
+ lines.push('or if W4 implementation reveals migration cost exceeding the');
308
+ lines.push('measured gain recorded here.');
309
+ } else {
310
+ lines.push('Revisit when a measured bench run shows the JSON variant');
311
+ lines.push('breaching a signed threshold — correctness floor breach,');
312
+ lines.push(`CLI-search p95 > ${THRESHOLDS.cliSearchP95MsMax}ms on the medium tier, or lost`);
313
+ lines.push('updates in the write burst — AND the spike stays clean AND FTS5');
314
+ lines.push('shows a meaningful gain. The trigger is a number from');
315
+ lines.push('`scripts/bench/memory-bench.mjs`, not a guess.');
316
+ }
317
+ lines.push('');
318
+ return lines.join('\n');
319
+ }
320
+
321
+ // ---------- main ----------
322
+
323
+ async function main() {
324
+ let opts;
325
+ try {
326
+ opts = parseArgs(process.argv.slice(2));
327
+ } catch (error) {
328
+ console.error(`[memory-gate] ${error.message}\n\n${USAGE}`);
329
+ process.exit(1);
330
+ }
331
+ if (opts.help) {
332
+ console.log(USAGE);
333
+ process.exit(0);
334
+ }
335
+ if (!opts.bench || !opts.spike) {
336
+ console.error(`[memory-gate] --bench and --spike are required.\n\n${USAGE}`);
337
+ process.exit(1);
338
+ }
339
+
340
+ const benchPath = path.resolve(opts.bench);
341
+ const spikePath = path.resolve(opts.spike);
342
+ let bench;
343
+ let spike;
344
+ try {
345
+ bench = JSON.parse(fs.readFileSync(benchPath, 'utf8'));
346
+ } catch (error) {
347
+ console.error(`[memory-gate] cannot read bench report: ${error.message}`);
348
+ process.exit(1);
349
+ }
350
+ try {
351
+ spike = JSON.parse(fs.readFileSync(spikePath, 'utf8'));
352
+ } catch (error) {
353
+ console.error(`[memory-gate] cannot read spike report: ${error.message}`);
354
+ process.exit(1);
355
+ }
356
+
357
+ const result = evaluateGate(bench, spike);
358
+ const benchRef = path.relative(REPO_ROOT, benchPath);
359
+ const spikeRef = path.relative(REPO_ROOT, spikePath);
360
+
361
+ const verdict = {
362
+ decision: result.decision,
363
+ ruleVersion: 1,
364
+ evidence: { benchRef, spikeRef },
365
+ thresholds: THRESHOLDS,
366
+ rationale: result.rationale,
367
+ };
368
+ if (result.inconclusiveGap.length > 0) verdict.inconclusiveGap = result.inconclusiveGap;
369
+
370
+ const outPath = path.resolve(opts.out);
371
+ fs.mkdirSync(path.dirname(outPath), { recursive: true });
372
+ fs.writeFileSync(outPath, renderAdr({ bench, spike, benchRef, spikeRef, result }));
373
+ const jsonPath = outPath.replace(/\.md$/i, '') + '.json';
374
+ fs.writeFileSync(jsonPath, `${JSON.stringify(verdict, null, 2)}\n`);
375
+
376
+ console.log(`[memory-gate] decision=${verdict.decision} gaps=${result.inconclusiveGap.length}`);
377
+ console.log(`[memory-gate] ADR: ${outPath}`);
378
+ console.log(`[memory-gate] verdict: ${jsonPath}`);
379
+ process.exit(0);
380
+ }
381
+
382
+ const invokedAsScript = process.argv[1]
383
+ && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
384
+ if (invokedAsScript) {
385
+ main().catch((error) => {
386
+ console.error(`[memory-gate] harness crash: ${error?.stack ?? error}`);
387
+ process.exit(2);
388
+ });
389
+ }
@@ -0,0 +1,179 @@
1
+ // TASK-002 — memory bench correctness/latency metric module
2
+ // (SPEC §5 FR-003/FR-004, §8, §10).
3
+ //
4
+ // Pure ESM; the only I/O is `footprintBytes` (recursive byte total). The bench
5
+ // runner (TASK-003), ablation (TASK-004), and gate (TASK-006) all import these
6
+ // functions so metric math lives in exactly one implementation — see
7
+ // docs/AI_HANDOFF/benchmark/metric-spec.md for pre-registered definitions.
8
+ //
9
+ // UNKNOWN semantics: the literal string 'UNKNOWN' is returned wherever the
10
+ // denominator or source data is absent/invalid. It is never 0, NaN, Infinity,
11
+ // or an estimate — a string cannot silently coerce into numeric aggregates.
12
+ //
13
+ // Hit shape (contract emitted by TASK-004 variants):
14
+ // {id, project_id?, status?, labels?[]}
15
+
16
+ import { readdir, stat } from 'node:fs/promises';
17
+ import { join } from 'node:path';
18
+
19
+ export const UNKNOWN = 'UNKNOWN';
20
+
21
+ // Labels that mark a record as non-authoritative when surfaced as a hit.
22
+ const NON_AUTHORITY_LABELS = new Set(['candidate', 'observation', 'injection']);
23
+ // Manifest query kinds whose relevantIds form the stale id set.
24
+ const STALE_QUERY_KINDS = new Set(['stale', 'superseded', 'expired']);
25
+
26
+ function isNonEmptyArray(value) {
27
+ return Array.isArray(value) && value.length > 0;
28
+ }
29
+
30
+ function isPositiveInt(value) {
31
+ return typeof value === 'number' && Number.isInteger(value) && value > 0;
32
+ }
33
+
34
+ /**
35
+ * precision@k = |top-k ∩ relevant| / k.
36
+ *
37
+ * @param {string[]} rankedIds ranked hit ids (best first)
38
+ * @param {string[]} relevantIds ids considered relevant for the query
39
+ * @param {number} k cutoff
40
+ * @returns {number|'UNKNOWN'} ratio in [0,1], or 'UNKNOWN' when relevantIds is
41
+ * empty/invalid or k is not a positive integer.
42
+ */
43
+ export function precisionAtK(rankedIds, relevantIds, k) {
44
+ if (!Array.isArray(rankedIds) || !isNonEmptyArray(relevantIds) || !isPositiveInt(k)) {
45
+ return UNKNOWN;
46
+ }
47
+ const relevant = new Set(relevantIds);
48
+ const hits = rankedIds.slice(0, k).filter((id) => relevant.has(id)).length;
49
+ return hits / k;
50
+ }
51
+
52
+ /**
53
+ * recall@k = |top-k ∩ relevant| / |relevant|.
54
+ *
55
+ * @returns {number|'UNKNOWN'} 'UNKNOWN' on empty/invalid relevantIds or k.
56
+ */
57
+ export function recallAtK(rankedIds, relevantIds, k) {
58
+ if (!Array.isArray(rankedIds) || !isNonEmptyArray(relevantIds) || !isPositiveInt(k)) {
59
+ return UNKNOWN;
60
+ }
61
+ const relevant = new Set(relevantIds);
62
+ const hits = rankedIds.slice(0, k).filter((id) => relevant.has(id)).length;
63
+ return hits / relevant.size;
64
+ }
65
+
66
+ /**
67
+ * scopeLeakRate = fraction of hits whose project_id is bound to a different
68
+ * project than `boundProjectId`. Hits with no project_id are not leaks (they
69
+ * are not bound to a *different* project); the denominator is total hits.
70
+ *
71
+ * @param {Array<{id: string, project_id?: string}>} hits
72
+ * @param {string} boundProjectId
73
+ * @returns {number|'UNKNOWN'} 'UNKNOWN' when hits is empty/invalid or
74
+ * boundProjectId is absent.
75
+ */
76
+ export function scopeLeakRate(hits, boundProjectId) {
77
+ if (!isNonEmptyArray(hits) || typeof boundProjectId !== 'string' || boundProjectId === '') {
78
+ return UNKNOWN;
79
+ }
80
+ const leaked = hits.filter(
81
+ (h) => h && typeof h.project_id === 'string' && h.project_id !== boundProjectId,
82
+ ).length;
83
+ return leaked / hits.length;
84
+ }
85
+
86
+ /**
87
+ * falseAuthorityRate = fraction of hits that are non-active / candidate /
88
+ * observation / injection-labeled records presented as current. A hit is
89
+ * flagged when `status` is a string other than 'active', or when `labels`
90
+ * contains a non-authority label.
91
+ *
92
+ * @param {Array<{id: string, status?: string, labels?: string[]}>} hits
93
+ * @returns {number|'UNKNOWN'} 'UNKNOWN' when hits is empty/invalid.
94
+ */
95
+ export function falseAuthorityRate(hits) {
96
+ if (!isNonEmptyArray(hits)) return UNKNOWN;
97
+ const flagged = hits.filter((h) => {
98
+ if (!h || typeof h !== 'object') return false;
99
+ if (typeof h.status === 'string' && h.status !== 'active') return true;
100
+ return Array.isArray(h.labels) && h.labels.some((l) => NON_AUTHORITY_LABELS.has(l));
101
+ }).length;
102
+ return flagged / hits.length;
103
+ }
104
+
105
+ /**
106
+ * staleHitRate = fraction of hits whose id is in the manifest's
107
+ * stale/superseded/expired id set. The set comes from `manifest.staleIds`
108
+ * when present, else is derived from `manifest.queries` entries whose kind is
109
+ * stale/superseded/expired (their relevantIds are the stale records).
110
+ *
111
+ * @param {Array<{id: string}>} hits
112
+ * @param {{staleIds?: string[], queries?: Array<{kind: string, relevantIds?: string[]}>}} manifest
113
+ * @returns {number|'UNKNOWN'} 'UNKNOWN' when hits or manifest is empty/invalid.
114
+ */
115
+ export function staleHitRate(hits, manifest) {
116
+ if (!isNonEmptyArray(hits) || !manifest || typeof manifest !== 'object') return UNKNOWN;
117
+ let staleIds;
118
+ if (Array.isArray(manifest.staleIds)) {
119
+ staleIds = manifest.staleIds;
120
+ } else if (Array.isArray(manifest.queries)) {
121
+ staleIds = manifest.queries
122
+ .filter((q) => q && STALE_QUERY_KINDS.has(q.kind) && Array.isArray(q.relevantIds))
123
+ .flatMap((q) => q.relevantIds);
124
+ } else {
125
+ return UNKNOWN;
126
+ }
127
+ const stale = new Set(staleIds);
128
+ const flagged = hits.filter((h) => h && stale.has(h.id)).length;
129
+ return flagged / hits.length;
130
+ }
131
+
132
+ /**
133
+ * Nearest-rank percentiles over run durations (same convention as
134
+ * scripts/bench/memory-baseline.mjs and metric-spec §7). Non-finite or
135
+ * negative entries are dropped; `n` counts valid durations only.
136
+ *
137
+ * @param {number[]} durations
138
+ * @returns {{n: number, p50: number|'UNKNOWN', p95: number|'UNKNOWN', p99: number|'UNKNOWN'}}
139
+ */
140
+ export function summarizeRuns(durations) {
141
+ const valid = Array.isArray(durations)
142
+ ? durations.filter((d) => typeof d === 'number' && Number.isFinite(d) && d >= 0)
143
+ : [];
144
+ if (valid.length === 0) return { n: 0, p50: UNKNOWN, p95: UNKNOWN, p99: UNKNOWN };
145
+ const sorted = [...valid].sort((a, b) => a - b);
146
+ const rank = (q) => sorted[Math.max(1, Math.ceil((q / 100) * sorted.length)) - 1];
147
+ return { n: sorted.length, p50: rank(50), p95: rank(95), p99: rank(99) };
148
+ }
149
+
150
+ /**
151
+ * Recursive byte total of a directory — the only async/fs function in the
152
+ * module. Follows no symlinks (lstat-style via dirent), so cycles and links
153
+ * escaping the fixture cannot inflate the measurement.
154
+ *
155
+ * @param {string} dir
156
+ * @returns {Promise<number|'UNKNOWN'>} total bytes, or 'UNKNOWN' when dir is
157
+ * not a readable directory.
158
+ */
159
+ export async function footprintBytes(dir) {
160
+ if (typeof dir !== 'string' || dir === '') return UNKNOWN;
161
+ let total = 0;
162
+ const walk = async (current) => {
163
+ const entries = await readdir(current, { withFileTypes: true });
164
+ for (const entry of entries) {
165
+ const path = join(current, entry.name);
166
+ if (entry.isDirectory()) {
167
+ await walk(path);
168
+ } else if (entry.isFile()) {
169
+ total += (await stat(path)).size;
170
+ }
171
+ }
172
+ };
173
+ try {
174
+ await walk(dir);
175
+ } catch {
176
+ return UNKNOWN;
177
+ }
178
+ return total;
179
+ }
@@ -15,6 +15,7 @@ import fs from 'node:fs';
15
15
  import os from 'node:os';
16
16
  import path from 'node:path';
17
17
  import { spawn } from 'node:child_process';
18
+ import { fileURLToPath } from 'node:url';
18
19
 
19
20
  const SLOWDOWN_LIMIT = 2.0;
20
21
 
@@ -121,6 +122,28 @@ function fail(msg) {
121
122
  process.exit(1);
122
123
  }
123
124
 
125
+ // BUG-C27-24: `wallClockMs === 0` (a sub-ms --cmd) made the base `base` 0 →
126
+ // every slowdownFactor NaN → `eligible` empty → `eligible.reduce(..., eligible[0])`
127
+ // dereffed undefined and threw a cryptic TypeError AFTER the whole benchmark ran.
128
+ // `wallClockMs === 0` is below Date.now() resolution for a spawned child, so it is
129
+ // marked unmeasurable and excluded; an empty `eligible` yields recommended=null so
130
+ // main() fails with a clear message, and the reduce is seeded with rows[0] so it can
131
+ // never deref undefined.
132
+ export function computeRecommendation(rows) {
133
+ const base = rows[0].wallClockMs;
134
+ for (const row of rows) {
135
+ row.unmeasurable = row.wallClockMs === 0 || base === 0;
136
+ row.slowdownFactor = base === 0 ? NaN : row.wallClockMs / base;
137
+ row.perRunCost = row.wallClockMs / row.n;
138
+ }
139
+ const eligible = rows.filter((r) => !r.unmeasurable && r.slowdownFactor <= SLOWDOWN_LIMIT);
140
+ const recommended = eligible.length === 0
141
+ ? null
142
+ : eligible.reduce((best, r) => (r.perRunCost < best.perRunCost ? r : best), rows[0]).n;
143
+ return { recommended, eligible };
144
+ }
145
+
146
+
124
147
  async function main() {
125
148
  const opts = (() => {
126
149
  try {
@@ -145,22 +168,18 @@ async function main() {
145
168
  // eslint-disable-next-line no-await-in-loop -- levels are measured sequentially on purpose
146
169
  rows.push(await runLevel(n, opts.cmd));
147
170
  }
148
- const base = rows[0].wallClockMs;
149
- for (const row of rows) {
150
- row.slowdownFactor = row.wallClockMs / base;
151
- row.perRunCost = row.wallClockMs / row.n;
171
+ const { recommended } = computeRecommendation(rows);
172
+ if (recommended === null) {
173
+ fail('no level met slowdown limit (base wall clock 0ms?) — benchmark is unmeasurable; rerun with a slower --cmd');
152
174
  }
153
175
 
154
- const eligible = rows.filter((r) => r.slowdownFactor <= SLOWDOWN_LIMIT);
155
- const recommended = eligible.reduce((best, r) => (r.perRunCost < best.perRunCost ? r : best), eligible[0]).n;
156
-
157
176
  const loadAvgEnd = os.loadavg();
158
177
  const finishedAt = new Date().toISOString();
159
178
 
160
179
  const result = {
161
180
  cpuCount,
162
- levels: rows.map(({ n, wallClockMs, slowdownFactor, perRunCost, failures, concurrencyHighWaterMark, timedOut }) => ({
163
- n, wallClockMs, slowdownFactor, perRunCost, failures, concurrencyHighWaterMark, timedOut,
181
+ levels: rows.map(({ n, wallClockMs, slowdownFactor, perRunCost, failures, concurrencyHighWaterMark, timedOut, unmeasurable }) => ({
182
+ n, wallClockMs, slowdownFactor, perRunCost, failures, concurrencyHighWaterMark, timedOut, unmeasurable,
164
183
  })),
165
184
  recommended,
166
185
  measurementConditions: { loadAvgStart, loadAvgEnd, startedAt, finishedAt, cpuCount },
@@ -173,7 +192,7 @@ async function main() {
173
192
  console.log(
174
193
  String(r.n).padStart(4)
175
194
  + ' ' + String(r.wallClockMs).padStart(12)
176
- + ' ' + r.slowdownFactor.toFixed(2).padStart(15)
195
+ + ' ' + (r.unmeasurable ? 'unmeasurable'.padStart(15) : r.slowdownFactor.toFixed(2).padStart(15))
177
196
  + ' ' + r.perRunCost.toFixed(1).padStart(11)
178
197
  + ' ' + String(r.failures).padStart(8)
179
198
  + ' ' + String(r.concurrencyHighWaterMark).padStart(23),
@@ -199,4 +218,7 @@ async function main() {
199
218
  }
200
219
  }
201
220
 
202
- main().catch((err) => fail(err && err.message ? err.message : String(err)));
221
+ const isMain = process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url);
222
+ if (isMain) {
223
+ main().catch((err) => fail(err && err.message ? err.message : String(err)));
224
+ }