@ngockhoale/ukit 2.7.6 → 2.7.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/CHANGELOG.md +105 -0
  2. package/package.json +1 -1
  3. package/scripts/install/sync-installed-mirror.mjs +250 -0
  4. package/scripts/perf/audit-perf.mjs +287 -36
  5. package/scripts/perf/diff-perf-findings.mjs +136 -0
  6. package/scripts/perf/perf-findings.json +260 -206
  7. package/scripts/perf/perf-measure.md +271 -0
  8. package/src/context/detectProjectContext.js +5 -0
  9. package/src/core/codeintel/invalidation.js +4 -0
  10. package/src/core/fileOps.js +40 -117
  11. package/src/core/hookChainDoctor.js +65 -2
  12. package/src/core/memory/store.js +22 -1
  13. package/src/core/taskBudgetValidator.js +9 -6
  14. package/src/core/unattendedDoctor.js +8 -1
  15. package/src/render/buildVariables.js +10 -0
  16. package/templates/.claude/agents/bug-debugger.md +1 -1
  17. package/templates/.claude/agents/feature-implementer.md +2 -2
  18. package/templates/.claude/commands/ukit/handoff-create.md +1 -1
  19. package/templates/.claude/commands/ukit/handoff-fullstack.md +1 -1
  20. package/templates/.claude/commands/ukit/handoff-implement.md +1 -1
  21. package/templates/.claude/commands/ukit/handoff-review.md +1 -1
  22. package/templates/.claude/hooks/auto-prune-bash.sh +19 -0
  23. package/templates/.claude/hooks/context-hardcap-gate.sh +4 -1
  24. package/templates/.claude/hooks/handoff-model-guard.sh +22 -11
  25. package/templates/.claude/hooks/reinject-context.sh +22 -0
  26. package/templates/.claude/hooks/reset-compact-pressure.sh +29 -0
  27. package/templates/.claude/hooks/session-episode.sh +20 -0
  28. package/templates/.claude/hooks/skill-router.sh +15 -8
  29. package/templates/.claude/hooks/verification-guard.sh +3 -0
  30. package/templates/.claude/ukit/index/route-task.mjs +237 -32
  31. package/templates/.claude/ukit/index/task-budget-validator.mjs +6 -2
  32. package/templates/.claude/ukit/runtime/async-lock.mjs +144 -10
  33. package/templates/.claude/ukit/runtime/compact-threshold.mjs +5 -2
  34. package/templates/.claude/ukit/runtime/execution-ledger.mjs +217 -17
  35. package/templates/.claude/ukit/runtime/hook-chain-runner.mjs +156 -24
  36. package/templates/.claude/ukit/runtime/hook-payload-store.mjs +57 -0
  37. package/templates/.claude/ukit/runtime/hook-telemetry.mjs +84 -12
  38. package/templates/.claude/ukit/runtime/hook-telemetry.sh +50 -0
  39. package/templates/.claude/ukit/runtime/stop-coordinator.mjs +35 -20
  40. package/templates/.claude/ukit/runtime/token-utils.mjs +37 -126
  41. package/templates/.codex/settings.json +1 -5
  42. package/templates/.omp/agents/bug-debugger.md +1 -1
  43. package/templates/.omp/agents/feature-implementer.md +2 -2
  44. package/templates/.omp/hooks/pre/ukit-bridge.js +157 -26
  45. package/templates/docs/AI_HANDOFF/INDEX.md +1 -1
  46. package/templates/docs/AI_HANDOFF/RULES.md +6 -6
  47. package/templates/ukit/storage/config.json +2 -2
@@ -17,7 +17,9 @@
17
17
  * count per (event, matcher) pair.
18
18
  * 3. micro-benchmark — live spawn measurements: the `bash` and `node` boot
19
19
  * floor, the advisory telemetry child, each settings row's script, and the
20
- * index/router helpers invoked per prompt.
20
+ * index/router helpers invoked per prompt — plus an in-process measurement
21
+ * of `appendTelemetryRow` itself, so the observability layer's own I/O is
22
+ * measured rather than assumed (O5).
21
23
  * 4. git-log — hot-path files added in the regression window (SPEC §2.3).
22
24
  *
23
25
  * Honesty contract: a number is emitted only when it was measured. An area with
@@ -33,9 +35,10 @@
33
35
  */
34
36
 
35
37
  import fs from 'node:fs';
38
+ import os from 'node:os';
36
39
  import path from 'node:path';
37
40
  import { execFileSync, spawnSync } from 'node:child_process';
38
- import { fileURLToPath } from 'node:url';
41
+ import { fileURLToPath, pathToFileURL } from 'node:url';
39
42
 
40
43
  const SCRIPT_DIR = path.dirname(fileURLToPath(import.meta.url));
41
44
  const DEFAULT_ROOT = path.resolve(SCRIPT_DIR, '..', '..');
@@ -56,6 +59,8 @@ const PROCESSES_PER_HOOK_ROW = 3;
56
59
 
57
60
  const HOOKS_SUBPATH = ['templates', '.claude', 'hooks'];
58
61
  const CHAIN_RUNNER_LABEL = 'hook-chain-runner';
62
+ // The ONE append API every hook writes through — the subject of the O5 benchmark.
63
+ const TELEMETRY_MODULE_SUBPATH = ['templates', '.claude', 'ukit', 'runtime', 'hook-telemetry.mjs'];
59
64
 
60
65
  // Index/router helpers a prompt or an edit pays a cold start for (SPEC §2.5).
61
66
  const ROUTER_HELPERS = [
@@ -316,6 +321,62 @@ function nestedScriptStats(rows) {
316
321
  return out;
317
322
  }
318
323
 
324
+ /**
325
+ * O3 (SPEC §FR-009): the exec-vs-stage split. A hook's wall time is two
326
+ * different things added together — time it spent waiting for the producer to
327
+ * finish writing stdin, and time it spent working — and only the second is the
328
+ * hook's cost. Rows carry the stage half as `stdinStageMs` (chain rows) or
329
+ * `stageMs` (direct rows); the exec half is derived from the row's own
330
+ * `elapsedMs`, and both halves come from ONE row so they always sum to it.
331
+ *
332
+ * Null-safe by construction: a row carrying neither field is skipped entirely
333
+ * rather than counted as a zero stage, and a bucket with no usable numbers
334
+ * summarises to nulls (`summarise` filters non-finite values), so a tree whose
335
+ * dataset predates the fields renders as "not measured", never as 0ms.
336
+ */
337
+ function stageSplit(rows) {
338
+ const buckets = new Map();
339
+ const allStage = [];
340
+ const allExec = [];
341
+ for (const row of rows) {
342
+ const chainStage = row.stdinStageMs;
343
+ const directStage = row.stageMs;
344
+ const hasChain = typeof chainStage === 'number' && Number.isFinite(chainStage);
345
+ const hasDirect = typeof directStage === 'number' && Number.isFinite(directStage);
346
+ if (!hasChain && !hasDirect) continue;
347
+ const stage = hasChain ? chainStage : directStage;
348
+ const key = row.hook || '(unattributed)';
349
+ if (!buckets.has(key)) {
350
+ buckets.set(key, { field: hasChain ? 'stdinStageMs' : 'stageMs', stage: [], exec: [] });
351
+ }
352
+ const bucket = buckets.get(key);
353
+ bucket.stage.push(stage);
354
+ allStage.push(stage);
355
+ if (typeof row.elapsedMs === 'number' && Number.isFinite(row.elapsedMs)) {
356
+ const exec = Math.max(0, row.elapsedMs - stage);
357
+ bucket.exec.push(exec);
358
+ allExec.push(exec);
359
+ }
360
+ }
361
+ const hooks = new Map();
362
+ for (const [hook, bucket] of buckets) {
363
+ hooks.set(hook, {
364
+ field: bucket.field,
365
+ stage: summarise(bucket.stage),
366
+ exec: summarise(bucket.exec),
367
+ // The rows belong to `hooks`, but `measured` counts the whole dataset so a
368
+ // reader can tell a fully-split snapshot from a partially-migrated one.
369
+ measured: bucket.stage.length,
370
+ });
371
+ }
372
+ return {
373
+ hooks,
374
+ measured: allStage.length,
375
+ stage: summarise(allStage),
376
+ exec: summarise(allExec),
377
+ };
378
+ }
379
+
319
380
  // --- source 2: settings.json structure -------------------------------------
320
381
 
321
382
  function loadSettings(root, override) {
@@ -365,23 +426,83 @@ function benchSpawn(file, args, iterations, env) {
365
426
  return summarise(samples);
366
427
  }
367
428
 
429
+ /**
430
+ * Every benchmark that exercises the telemetry writer is pointed at a throwaway
431
+ * project root: the audit must never add rows to the dataset it is measuring,
432
+ * or two snapshots of the same tree would differ by the act of auditing them
433
+ * (O2/O5).
434
+ */
435
+ function scratchRoot(label) {
436
+ return fs.mkdtempSync(path.join(os.tmpdir(), `ukit-perf-${label}-`));
437
+ }
438
+
368
439
  function benchmarkFloor(root, iterations) {
369
440
  const env = { ...process.env, CLAUDE_PROJECT_DIR: root };
370
441
  const telemetryTwin = path.join(root, 'templates', '.claude', 'ukit', 'runtime', 'hook-telemetry.mjs');
371
- return {
372
- floor: {
373
- bash: benchSpawn('bash', ['-c', 'true'], iterations, env),
374
- node: benchSpawn('node', ['-e', '0'], iterations, env),
375
- },
376
- telemetryFinish: fs.existsSync(telemetryTwin)
377
- ? benchSpawn('node', [telemetryTwin, '--finish'], iterations, {
378
- ...env,
379
- UKIT_TEL_HOOK: 'audit-bench.sh',
380
- UKIT_TEL_RC: '0',
381
- PROJECT_ROOT: root,
382
- })
383
- : null,
384
- };
442
+ const scratch = scratchRoot('floor');
443
+ try {
444
+ return {
445
+ floor: {
446
+ bash: benchSpawn('bash', ['-c', 'true'], iterations, env),
447
+ node: benchSpawn('node', ['-e', '0'], iterations, env),
448
+ },
449
+ telemetryFinish: fs.existsSync(telemetryTwin)
450
+ ? benchSpawn('node', [telemetryTwin, '--finish'], iterations, {
451
+ ...env,
452
+ UKIT_TEL_HOOK: 'audit-bench.sh',
453
+ UKIT_TEL_RC: '0',
454
+ PROJECT_ROOT: scratch,
455
+ CLAUDE_PROJECT_DIR: scratch,
456
+ })
457
+ : null,
458
+ };
459
+ } finally {
460
+ fs.rmSync(scratch, { recursive: true, force: true });
461
+ }
462
+ }
463
+
464
+ /** The telemetry writer a tree ships, preferring the template (like hooksDir). */
465
+ function telemetryModulePath(root) {
466
+ const candidates = [
467
+ path.join(root, 'templates', '.claude', 'ukit', 'runtime', 'hook-telemetry.mjs'),
468
+ path.join(root, '.claude', 'ukit', 'runtime', 'hook-telemetry.mjs'),
469
+ ];
470
+ return candidates.find((candidate) => fs.existsSync(candidate)) || null;
471
+ }
472
+
473
+ /**
474
+ * O5: cost of one telemetry row as the writer actually pays it — the same
475
+ * `appendTelemetryRow` the hooks call, invoked in-process against a temp project
476
+ * root. The function is async only because the ESM writer must be imported
477
+ * dynamically; the measured region is synchronous I/O, not module loading.
478
+ */
479
+ async function benchmarkAppend(root, iterations) {
480
+ const module = telemetryModulePath(root);
481
+ if (!module) return { stats: null, iterations, module: null, reason: 'telemetry writer not resolvable from this root' };
482
+ let dir = null;
483
+ try {
484
+ dir = scratchRoot('append');
485
+ const { appendTelemetryRow } = await import(pathToFileURL(module).href);
486
+ const row = {
487
+ v: 1,
488
+ ts: Date.now(),
489
+ hookEvent: 'PreToolUse',
490
+ hook: 'audit-append-bench.sh',
491
+ elapsedMs: 1,
492
+ outcome: 'ok',
493
+ };
494
+ const samples = [];
495
+ for (let i = 0; i < iterations; i += 1) {
496
+ const start = process.hrtime.bigint();
497
+ appendTelemetryRow(dir, 'audit-append-bench', row);
498
+ samples.push(Number(process.hrtime.bigint() - start) / 1e6);
499
+ }
500
+ return { stats: summarise(samples), iterations, module, reason: null };
501
+ } catch (error) {
502
+ return { stats: null, iterations, module, reason: `append measurement failed: ${error.message}` };
503
+ } finally {
504
+ if (dir) fs.rmSync(dir, { recursive: true, force: true });
505
+ }
385
506
  }
386
507
 
387
508
  /** Resolves the directory holding the shipped hooks, preferring the template. */
@@ -479,11 +600,11 @@ function finding(entry) {
479
600
  function slug(value) {
480
601
  return String(value).replace(/[^a-zA-Z0-9]+/g, '-').replace(/^-|-$/g, '').toLowerCase();
481
602
  }
482
-
483
603
  function buildFindings(context) {
484
604
  const {
485
605
  telemetry, telemetryDetail, groups, floor, telemetryFinish, hookBench,
486
606
  routerBench, perHook, records, events, multiplicity, nested, additions, spawns,
607
+ appendBench, root, split, telemetryRows,
487
608
  } = context;
488
609
 
489
610
  const findings = [];
@@ -691,7 +812,96 @@ function buildFindings(context) {
691
812
  status: 'not-a-bug',
692
813
  }));
693
814
 
694
- // 9. Registered hooks with no rows at all — evidenced absence, not a guess.
815
+ // 9. The observability layer's own cost (O5) — measured in-process, against a
816
+ // throwaway project root, so auditing a tree never writes into the dataset
817
+ // whose rows the audit is counting. Below the hotspot bar this is a documented
818
+ // non-issue; above it the append joins the same per-call cost reducible by
819
+ // TASK-234's consolidation.
820
+ const appendStats = appendBench.stats;
821
+ if (appendStats && appendStats.p50 !== null) {
822
+ const hotspot = appendStats.p50 > HOTSPOT_P50_MS;
823
+ findings.push(finding({
824
+ id: 'telemetry-append-overhead',
825
+ area: 'telemetry append overhead — appendTelemetryRow, in-process against a throwaway project root',
826
+ evidence: 'micro-benchmark',
827
+ evidenceDetail: `appendTelemetryRow over ${appendBench.iterations} iterations against a temp project root (the audited telemetry dir is never written): p50 ${appendStats.p50.toFixed(3)}ms, p95 ${appendStats.p95.toFixed(3)}ms, max ${appendStats.max.toFixed(3)}ms, total ${appendStats.total.toFixed(3)}ms; writer ${appendBench.module ? path.relative(root, appendBench.module) : 'n/a'}`,
828
+ p50ms: appendStats.p50,
829
+ p95ms: appendStats.p95,
830
+ totalMs: appendStats.total,
831
+ invocations: appendBench.iterations,
832
+ rootCause: hotspot
833
+ ? `one synchronous append costs p50 ${appendStats.p50.toFixed(3)}ms — above the ${HOTSPOT_P50_MS}ms hotspot bar — so the observability layer is paying a measurable per-hook term on the hot path`
834
+ : `one synchronous append costs p50 ${appendStats.p50.toFixed(3)}ms (under the ${HOTSPOT_P50_MS}ms hotspot bar): the observability layer's own I/O is bounded by a single append plus an amortized 1-in-16 sampled sweep, so it is not the hook chain's dominant term`,
835
+ fixTask: hotspot ? 'TASK-234' : null,
836
+ status: hotspot ? 'confirmed' : 'not-a-bug',
837
+ }));
838
+ } else {
839
+ findings.push(finding({
840
+ id: 'telemetry-append-overhead',
841
+ area: 'telemetry append overhead — appendTelemetryRow, in-process against a throwaway project root',
842
+ evidence: 'unavailable',
843
+ evidenceDetail: `${telemetryDetail}; the append benchmark needs the telemetry writer (${TELEMETRY_MODULE_SUBPATH.join('/')}) resolvable from this root — ${appendBench.reason || 'not resolvable'}, so the observability layer's own cost stays unset rather than estimated`,
844
+ p50ms: null,
845
+ p95ms: null,
846
+ totalMs: null,
847
+ invocations: 0,
848
+ rootCause: 'the observability layer cannot be measured on a tree that ships no telemetry writer; the absence is reported instead of a plausible-looking number',
849
+ fixTask: null,
850
+ status: 'deferred',
851
+ }));
852
+ }
853
+
854
+ // 10. O3: the exec-vs-stage split. The reported number here is the EXEC half
855
+ // — the hook's own cost — because the stage half is the producer's, and
856
+ // treating wait-on-stdin as a hook defect is exactly the misreading O3 names.
857
+ // Both halves are stated so the tail can be attributed rather than debated.
858
+ if (split.measured > 0) {
859
+ const stageText = split.stage.p50 === null
860
+ ? 'n/a'
861
+ : `${split.stage.p50}ms p50 / ${split.stage.p95}ms p95`;
862
+ const execText = split.exec.p50 === null
863
+ ? 'n/a'
864
+ : `${split.exec.p50}ms p50 / ${split.exec.p95}ms p95`;
865
+ const top = [...split.hooks.entries()]
866
+ .sort((a, b) => (b[1].stage.total || 0) - (a[1].stage.total || 0))
867
+ .slice(0, 5)
868
+ .map(([hook, entry]) => `${hook} (${entry.field}: stage ${entry.stage.p50}ms p50, exec ${entry.exec.p50}ms p50, n=${entry.measured})`)
869
+ .join(', ');
870
+ const execHotspot = split.exec.p50 !== null && split.exec.p50 > HOTSPOT_P50_MS;
871
+ findings.push(finding({
872
+ id: 'stdin-stage-vs-exec-split',
873
+ area: 'stdin stage vs execution split (O3 — the wall-time tail attributed)',
874
+ evidence: measuredEvidence,
875
+ evidenceDetail: `${telemetryDetail}; ${split.measured} of ${telemetryRows} rows carry the split (a row the producer's stdin held open carries the stage half; a row measured before this field existed carries neither and is excluded, not counted as 0ms). Across those rows, stage ${stageText} vs exec ${execText}; the exec half is the hook's own cost. Per hook: ${top}`,
876
+ p50ms: split.exec.p50,
877
+ p95ms: split.exec.p95,
878
+ totalMs: split.exec.total,
879
+ invocations: split.measured,
880
+ rootCause: execHotspot
881
+ ? `once the stdin stage is attributed to the producer, the remaining execution p50 of ${split.exec.p50}ms is still above the ${HOTSPOT_P50_MS}ms hotspot bar, so the tail is NOT fully explained by stdin staging and the hook's own work needs TASK-234's consolidation`
882
+ : `the tail artifact is a measurement artifact, not hook work: the wall time a hook reports is the stdin stage plus its execution, and only the execution half is the hook's cost. Attributing the stage half to the producer is what makes project-important.sh's reported p95 tail interpretable instead of looking like a slow script`,
883
+ fixTask: execHotspot ? 'TASK-234' : null,
884
+ status: execHotspot ? 'confirmed' : 'not-a-bug',
885
+ }));
886
+ } else {
887
+ // Rows predating the fields are the expected state on an unmigrated tree —
888
+ // reported as an evidenced absence, never as a 0ms stage.
889
+ findings.push(finding({
890
+ id: 'stdin-stage-vs-exec-split',
891
+ area: 'stdin stage vs execution split (O3 — the wall-time tail attributed)',
892
+ evidence: 'unavailable',
893
+ evidenceDetail: `${telemetryDetail}; no row carries \`stdinStageMs\` or \`stageMs\`, so the exec-vs-stage split is unmeasured on this tree. Old rows keep aggregating exactly as before — the fields are additive and optional, so their absence is reported rather than guessed at as 0ms.`,
894
+ p50ms: null,
895
+ p95ms: null,
896
+ totalMs: null,
897
+ invocations: 0,
898
+ rootCause: 'the split needs rows written after the fields shipped; a dataset predating them yields no numbers at all (null, not zero), which is the null-safe posture the audit reports here',
899
+ fixTask: null,
900
+ status: 'deferred',
901
+ }));
902
+ }
903
+
904
+ // 11. Registered hooks with no rows at all — evidenced absence, not a guess.
695
905
  const settingsScripts = [...new Set(groups.flatMap((group) => group.scripts.map((script) => script.script)))];
696
906
  for (const script of settingsScripts) {
697
907
  if (perHook.has(script)) continue;
@@ -720,6 +930,7 @@ function renderMeasureDoc(context) {
720
930
  generatedAt, root, telemetry, telemetryDetail, perHook, records, events,
721
931
  multiplicity, nested, groups, floor, telemetryFinish, hookBench, routerBench,
722
932
  additions, findings, settingsFile, iterations, hooksResolvedFrom,
933
+ telemetryRows, telemetryFiles, appendBench, split,
723
934
  } = context;
724
935
  const lines = [];
725
936
  lines.push('# perf-measure.md — raw measurement notes (TASK-233)');
@@ -728,7 +939,8 @@ function renderMeasureDoc(context) {
728
939
  lines.push('');
729
940
  lines.push('Appendix source for `docs/AI_REPORT/AI_REVIEW_BUGS_REPORT.md`. Every number in');
730
941
  lines.push('`perf-findings.json` comes from one of the four sources below. TASK-236 re-runs');
731
- lines.push('the identical command for the before/after table (SPEC §5).');
942
+ lines.push('the identical command for the before/after table (SPEC §5). Two numbers from two');
943
+ lines.push('different snapshots are only comparable once their provenance line agrees.');
732
944
  lines.push('');
733
945
  lines.push('## 1. Reproduce');
734
946
  lines.push('');
@@ -736,13 +948,14 @@ function renderMeasureDoc(context) {
736
948
  lines.push('node scripts/perf/audit-perf.mjs \\');
737
949
  lines.push(` --telemetry-dir "${telemetry.dir || path.join(root, '.ukit', 'storage', 'cache', 'hook-latency')}" \\`);
738
950
  lines.push(` --out "${root}/scripts/perf/perf-findings.json"`);
951
+ lines.push('node scripts/perf/diff-perf-findings.mjs <earlier-snapshot>.json scripts/perf/perf-findings.json');
739
952
  lines.push('node --test tests/handoff/c33/perfFindings.test.js');
740
- lines.push('```');
741
953
  lines.push('');
742
954
  lines.push(`- root: \`${root}\``);
743
955
  lines.push(`- settings.json: \`${settingsFile || 'NOT FOUND'}\``);
744
956
  lines.push(`- hooks resolved from: \`${hooksResolvedFrom || 'NOT FOUND'}\``);
745
957
  lines.push(`- telemetry: ${telemetry.available ? `\`${telemetry.dir}\` — ${telemetry.rows.length} rows in ${telemetry.files} files${telemetry.malformed ? ` (${telemetry.malformed} malformed lines skipped)` : ''}` : 'NOT PRESENT — every telemetry-sourced finding is `evidence: "unavailable"`'}`);
958
+ lines.push(`- snapshot provenance: generatedAt \`${generatedAt}\`, telemetryRows: ${telemetryRows}, telemetryFiles: ${telemetryFiles} — hooks append continuously, so a report is a point-in-time view, not a stable dataset`);
746
959
  lines.push(`- benchmark iterations per subject: ${iterations}`);
747
960
  lines.push('');
748
961
  lines.push('## 2. Process floor (live spawnSync, ms)');
@@ -757,8 +970,19 @@ function renderMeasureDoc(context) {
757
970
  lines.push('');
758
971
  lines.push(`A hook row = wrapper bash + node runtime + advisory telemetry child = **${PROCESSES_PER_HOOK_ROW} processes**.`);
759
972
  lines.push('');
760
- lines.push('## 3. Per-hook standalone re-run (`/bin/bash <script>` with `{}` on stdin, ms)');
973
+ lines.push('## 3. Telemetry append overhead (O5 — in-process `appendTelemetryRow`, ms)');
761
974
  lines.push('');
975
+ if (appendBench.stats) {
976
+ lines.push('| subject | iterations | p50 | p95 | max |');
977
+ lines.push('|---|---|---|---|---|');
978
+ lines.push(`| \`appendTelemetryRow\` → throwaway temp root | ${appendBench.iterations} | ${appendBench.stats.p50.toFixed(3)} | ${appendBench.stats.p95.toFixed(3)} | ${appendBench.stats.max.toFixed(3)} |`);
979
+ lines.push('');
980
+ lines.push(`Writer: \`${appendBench.module ? path.relative(root, appendBench.module) : 'n/a'}\`. Measured against a temp project root, so auditing a tree never adds rows to the dataset it is counting.`);
981
+ } else {
982
+ lines.push(`NOT MEASURED — ${appendBench.reason || 'the telemetry writer is not resolvable from this root'}. The observability layer's own cost stays unset rather than estimated.`);
983
+ }
984
+ lines.push('');
985
+ lines.push('## 4. Per-hook standalone re-run (`/bin/bash <script>` with `{}` on stdin, ms)');
762
986
  lines.push('| hook | p50 | p95 | max |');
763
987
  lines.push('|---|---|---|---|');
764
988
  for (const [name, stats] of [...hookBench.entries()].sort((a, b) => b[1].p50 - a[1].p50)) {
@@ -766,7 +990,7 @@ function renderMeasureDoc(context) {
766
990
  }
767
991
  if (!hookBench.size) lines.push('| _(no settings hook resolvable from this root)_ | | | |');
768
992
  lines.push('');
769
- lines.push('## 4. Router/index helper cold start (ms)');
993
+ lines.push('## 5. Router/index helper cold start (ms)');
770
994
  lines.push('');
771
995
  lines.push('| helper | p50 | p95 | max |');
772
996
  lines.push('|---|---|---|---|');
@@ -775,7 +999,7 @@ function renderMeasureDoc(context) {
775
999
  }
776
1000
  if (!routerBench.size) lines.push('| _(no helper resolvable from this root)_ | | | |');
777
1001
  lines.push('');
778
- lines.push('## 5. Per-hook telemetry (every hook with rows)');
1002
+ lines.push('## 6. Per-hook telemetry (every hook with rows)');
779
1003
  lines.push('');
780
1004
  lines.push('| hook | n | p50 | p95 | max | total |');
781
1005
  lines.push('|---|---|---|---|---|---|');
@@ -783,7 +1007,25 @@ function renderMeasureDoc(context) {
783
1007
  lines.push(`| \`${hook}\` | ${stats.n} | ${stats.p50 ?? 'n/a'} | ${stats.p95 ?? 'n/a'} | ${stats.max ?? 'n/a'} | ${stats.total ?? 'n/a'} |`);
784
1008
  }
785
1009
  lines.push('');
786
- lines.push('## 6. Per tool / per event cost (chain-runner total where present, else direct sum)');
1010
+ lines.push('## 7. Stdin stage vs execution split (O3 — the wall-time tail attributed)');
1011
+ lines.push('');
1012
+ lines.push('A row\'s `elapsedMs` is the stdin stage plus the hook\'s own execution. `stageMs`');
1013
+ lines.push('(direct rows) and `stdinStageMs` (chain rows) are the first half — the producer\'s');
1014
+ lines.push('time, not the hook\'s — and `exec` below is the remainder, derived from the SAME');
1015
+ lines.push('row so the two halves always sum to it. A row carrying neither field predates the');
1016
+ lines.push('split or never staged stdin: it is excluded, never counted as a 0ms stage.');
1017
+ lines.push('');
1018
+ lines.push(`- rows with the split: ${split.measured} of ${telemetryRows}`);
1019
+ lines.push(`- overall: stage ${split.stage.p50 ?? 'n/a'}ms p50 / ${split.stage.p95 ?? 'n/a'}ms p95, exec ${split.exec.p50 ?? 'n/a'}ms p50 / ${split.exec.p95 ?? 'n/a'}ms p95`);
1020
+ lines.push('');
1021
+ lines.push('| hook | field | n | stage p50 | stage p95 | exec p50 | exec p95 |');
1022
+ lines.push('|---|---|---|---|---|---|---|');
1023
+ for (const [hook, entry] of [...split.hooks.entries()].sort((a, b) => (b[1].stage.total || 0) - (a[1].stage.total || 0))) {
1024
+ lines.push(`| \`${hook}\` | \`${entry.field}\` | ${entry.measured} | ${entry.stage.p50 ?? 'n/a'} | ${entry.stage.p95 ?? 'n/a'} | ${entry.exec.p50 ?? 'n/a'} | ${entry.exec.p95 ?? 'n/a'} |`);
1025
+ }
1026
+ if (!split.hooks.size) lines.push('| _(no row carries the split on this tree)_ | | | | | | |');
1027
+ lines.push('');
1028
+ lines.push('## 8. Per tool / per event cost (chain-runner total where present, else direct sum)');
787
1029
  lines.push('');
788
1030
  lines.push('| scope | calls | cost p50 | cost p95 | hook rows/call p50 | direct / chain-runner calls |');
789
1031
  lines.push('|---|---|---|---|---|---|');
@@ -798,7 +1040,7 @@ function renderMeasureDoc(context) {
798
1040
  lines.push(`| event ${event} (clustered, gap ≤ ${CLUSTER_GAP_MS}ms) | ${entry.calls} | ${entry.cost.p50 ?? 'n/a'} | ${entry.cost.p95 ?? 'n/a'} | ${entry.hookRows.p50 ?? 'n/a'} | ${entry.byPath.direct} / ${entry.byPath.chainRunner} |`);
799
1041
  }
800
1042
  lines.push('');
801
- lines.push('## 7. Hook multiplicity proof (repeat-fire check)');
1043
+ lines.push('## 9. Hook multiplicity proof (repeat-fire check)');
802
1044
  lines.push('');
803
1045
  lines.push('| hook | max fires in one tool call | calls observed | fire-count distribution |');
804
1046
  lines.push('|---|---|---|---|');
@@ -806,7 +1048,7 @@ function renderMeasureDoc(context) {
806
1048
  lines.push(`| \`${hook}\` | ${entry.maxPerCall} | ${entry.calls} | ${JSON.stringify(entry.distribution)} |`);
807
1049
  }
808
1050
  lines.push('');
809
- lines.push('## 8. hook-chain-runner nested per-script (already-consolidated omp path)');
1051
+ lines.push('## 10. hook-chain-runner nested per-script (already-consolidated omp path)');
810
1052
  lines.push('');
811
1053
  lines.push('| script | n | p50 | p95 | max |');
812
1054
  lines.push('|---|---|---|---|---|');
@@ -814,7 +1056,7 @@ function renderMeasureDoc(context) {
814
1056
  lines.push(`| \`${name}\` | ${stats.n} | ${stats.p50} | ${stats.p95} | ${stats.max} |`);
815
1057
  }
816
1058
  lines.push('');
817
- lines.push('## 9. settings.json groups (spawn budget per call)');
1059
+ lines.push('## 11. settings.json groups (spawn budget per call)');
818
1060
  lines.push('');
819
1061
  lines.push('| event | matcher | hooks | registered timeouts | spawns/call |');
820
1062
  lines.push('|---|---|---|---|---|');
@@ -822,7 +1064,7 @@ function renderMeasureDoc(context) {
822
1064
  lines.push(`| ${group.event} | ${group.matcher || '(all)'} | ${group.scripts.map((script) => script.script).join(', ')} | ${group.scripts.map((script) => `${script.timeout}s`).join(', ')} | ${group.scripts.length * PROCESSES_PER_HOOK_ROW} |`);
823
1065
  }
824
1066
  lines.push('');
825
- lines.push('## 10. Recent hot-path additions (`git log --diff-filter=A`)');
1067
+ lines.push('## 12. Recent hot-path additions (`git log --diff-filter=A`)');
826
1068
  lines.push('');
827
1069
  lines.push('| commit | date | subject | files added |');
828
1070
  lines.push('|---|---|---|---|');
@@ -831,7 +1073,7 @@ function renderMeasureDoc(context) {
831
1073
  }
832
1074
  if (!additions.length) lines.push('| _(none in window)_ | | | |');
833
1075
  lines.push('');
834
- lines.push('## 11. Findings emitted');
1076
+ lines.push('## 13. Findings emitted');
835
1077
  lines.push('');
836
1078
  lines.push('| id | evidence | p50 | p95 | status | fixTask |');
837
1079
  lines.push('|---|---|---|---|---|---|');
@@ -844,7 +1086,7 @@ function renderMeasureDoc(context) {
844
1086
 
845
1087
  // --- main ------------------------------------------------------------------
846
1088
 
847
- function main() {
1089
+ async function main() {
848
1090
  const options = parseArgs(process.argv.slice(2));
849
1091
  const root = options.root;
850
1092
  const telemetryDir = options.telemetryDir || path.join(root, '.ukit', 'storage', 'cache', 'hook-latency');
@@ -860,10 +1102,12 @@ function main() {
860
1102
  const events = eventAggregates(records);
861
1103
  const multiplicity = hookMultiplicity(records);
862
1104
  const nested = nestedScriptStats(telemetry.rows);
1105
+ const split = stageSplit(telemetry.rows);
863
1106
 
864
1107
  const { floor, telemetryFinish } = benchmarkFloor(root, options.benchIterations);
865
1108
  const { results: hookBench, dir: hooksResolvedFrom } = benchmarkHooks(root, groups, options.benchIterations);
866
1109
  const routerBench = benchmarkRouterHelpers(root, options.benchIterations);
1110
+ const appendBench = await benchmarkAppend(root, options.benchIterations);
867
1111
  const additions = recentHotPathAdditions(root, options.sinceDays);
868
1112
  const chainScripts = (() => {
869
1113
  const lengths = new Map();
@@ -874,28 +1118,35 @@ function main() {
874
1118
  return [...lengths.entries()].sort((a, b) => b[1] - a[1]).map(([length]) => length)[0] || 0;
875
1119
  })();
876
1120
 
1121
+ // Snapshot provenance (O2): the dataset grows while it is being audited
1122
+ // (30k → 34k rows in two hours in the finding), so a report that does not
1123
+ // state how many rows and files it looked at cannot be compared with another.
1124
+ const telemetryRows = telemetry.available ? telemetry.rows.length : 0;
1125
+ const telemetryFiles = telemetry.available ? telemetry.files : 0;
877
1126
  const telemetryDetail = telemetry.available
878
- ? `.ukit/storage/cache/hook-latency — ${telemetry.rows.length} rows across ${telemetry.files} session files, span ${telemetry.span || 'n/a'}`
1127
+ ? `.ukit/storage/cache/hook-latency — ${telemetryRows} rows across ${telemetryFiles} session files, span ${telemetry.span || 'n/a'}`
879
1128
  : `telemetry directory not present (${telemetry.dir || 'n/a'})`;
880
1129
 
881
1130
  const generatedAt = new Date().toISOString();
882
1131
  const context = {
883
1132
  telemetry, telemetryDetail, groups, floor, telemetryFinish, hookBench, routerBench,
884
1133
  root,
885
- perHook, records, events, multiplicity, nested, additions, settingsFile,
1134
+ perHook, records, events, multiplicity, nested, additions, settingsFile, split,
886
1135
  iterations: options.benchIterations,
887
1136
  sinceDays: options.sinceDays,
1137
+ telemetryRows, telemetryFiles, appendBench,
888
1138
  spawns: { chainScripts },
889
- hooksResolvedFrom,
890
1139
  };
891
1140
  const findings = buildFindings({ ...context, generatedAt });
892
1141
 
893
1142
  const report = {
894
1143
  generatedAt,
895
1144
  schema: 'ukit-perf-findings/1',
1145
+ telemetryRows,
1146
+ telemetryFiles,
896
1147
  sources: {
897
1148
  telemetry: telemetry.available
898
- ? { dir: telemetryDir, rows: telemetry.rows.length, files: telemetry.files }
1149
+ ? { dir: telemetryDir, rows: telemetryRows, files: telemetryFiles }
899
1150
  : { dir: telemetryDir, available: false },
900
1151
  settings: settingsFile,
901
1152
  hooksDir: hooksResolvedFrom,
@@ -911,10 +1162,10 @@ function main() {
911
1162
 
912
1163
  process.stdout.write(
913
1164
  `${findings.length} findings → ${path.relative(root, outFile)}`
914
- + ` (telemetry ${telemetry.available ? `${telemetry.rows.length} rows` : 'absent'},`
1165
+ + ` (generatedAt ${generatedAt}, telemetry ${telemetry.available ? `${telemetryRows} rows / ${telemetryFiles} files` : 'absent'},`
915
1166
  + ` settings ${settingsFile ? 'found' : 'missing'},`
916
1167
  + ` ${groups.length} groups, ${hookBench.size} hooks benchmarked)\n`,
917
1168
  );
918
1169
  }
919
1170
 
920
- main();
1171
+ await main();
@@ -0,0 +1,136 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * diff-perf-findings.mjs — findings-level diff between two audit snapshots
4
+ * (TASK-006, SPEC §5 FR-008).
5
+ *
6
+ * Why a separate script and not a mode of `audit-perf.mjs`: the audit takes live
7
+ * measurements, so "compare two snapshots" is a read-only operation that must
8
+ * work against archived JSON long after the telemetry that produced it rotated
9
+ * away. This script never touches telemetry at all — it reads two reports and
10
+ * prints their deltas.
11
+ *
12
+ * Every snapshot is a point in time over a live dataset (see the O2 finding:
13
+ * 30k → 34k rows in two hours), so both snapshots' `generatedAt` /
14
+ * `telemetryRows` / `telemetryFiles` provenance is printed before any delta —
15
+ * two reports are only comparable once their provenance is known.
16
+ *
17
+ * Deltas reported, per finding id:
18
+ * + <id> added — present in b, absent from a
19
+ * - <id> removed — present in a, absent from b (never a crash)
20
+ * ~ <id> changed — same id, different status and/or different measured numbers
21
+ *
22
+ * Usage:
23
+ * node scripts/perf/diff-perf-findings.mjs <a.json> <b.json>
24
+ *
25
+ * Exit: 0 = compared (with or without deltas, including a self-diff); 2 = usage,
26
+ * unreadable file, or a report that is not a findings snapshot.
27
+ */
28
+ import fs from 'node:fs';
29
+
30
+ const METRIC_KEYS = ['p50ms', 'p95ms', 'totalMs', 'invocations'];
31
+
32
+ function usage() {
33
+ process.stderr.write('usage: node scripts/perf/diff-perf-findings.mjs <a.json> <b.json>\n');
34
+ }
35
+
36
+ function loadSnapshot(file) {
37
+ let report;
38
+ try {
39
+ report = JSON.parse(fs.readFileSync(file, 'utf8'));
40
+ } catch (error) {
41
+ throw new Error(`cannot read findings snapshot ${file}: ${error.message}`);
42
+ }
43
+ if (!report || !Array.isArray(report.findings)) {
44
+ throw new Error(`${file} is not a findings snapshot (no findings[] array)`);
45
+ }
46
+ return report;
47
+ }
48
+
49
+ function formatNumber(value) {
50
+ if (typeof value !== 'number' || !Number.isFinite(value)) return 'n/a';
51
+ return Number.isInteger(value) ? String(value) : value.toFixed(1);
52
+ }
53
+
54
+ function provenanceLine(label, file, report) {
55
+ const rows = formatNumber(report.telemetryRows);
56
+ const files = formatNumber(report.telemetryFiles);
57
+ const generatedAt = typeof report.generatedAt === 'string' ? report.generatedAt : 'n/a';
58
+ return `${label}: ${file} — generatedAt ${generatedAt}, telemetryRows ${rows} / telemetryFiles ${files}`;
59
+ }
60
+
61
+ /** Metric drift for one finding pair, as `key a → b` fragments. */
62
+ function metricDrift(before, after) {
63
+ return METRIC_KEYS
64
+ .filter((key) => before[key] !== after[key])
65
+ .map((key) => `${key} ${formatNumber(before[key])} → ${formatNumber(after[key])}`);
66
+ }
67
+
68
+ function diffFindings(a, b) {
69
+ const before = new Map(a.findings.map((item) => [item.id, item]));
70
+ const after = new Map(b.findings.map((item) => [item.id, item]));
71
+ const deltas = [];
72
+
73
+ for (const [id, item] of after) {
74
+ if (!before.has(id)) {
75
+ deltas.push({ kind: 'added', id, line: `+ ${id} added (status ${item.status})` });
76
+ continue;
77
+ }
78
+ const previous = before.get(id);
79
+ const drift = metricDrift(previous, item);
80
+ if (previous.status !== item.status) {
81
+ const suffix = drift.length ? ` (${drift.join(', ')})` : '';
82
+ deltas.push({
83
+ kind: 'changed',
84
+ id,
85
+ line: `~ ${id} status ${previous.status} → ${item.status}${suffix}`,
86
+ });
87
+ } else if (drift.length) {
88
+ deltas.push({ kind: 'changed', id, line: `~ ${id} ${drift.join(', ')}` });
89
+ }
90
+ }
91
+
92
+ for (const [id, item] of before) {
93
+ if (after.has(id)) continue;
94
+ deltas.push({ kind: 'removed', id, line: `- ${id} removed (was ${item.status})` });
95
+ }
96
+
97
+ return deltas;
98
+ }
99
+
100
+ function main(argv) {
101
+ const [fileA, fileB] = argv;
102
+ if (!fileA || !fileB || argv.length > 2) {
103
+ usage();
104
+ return 2;
105
+ }
106
+
107
+ let snapshotA;
108
+ let snapshotB;
109
+ try {
110
+ snapshotA = loadSnapshot(fileA);
111
+ snapshotB = loadSnapshot(fileB);
112
+ } catch (error) {
113
+ process.stderr.write(`${error.message}\n`);
114
+ return 2;
115
+ }
116
+
117
+ const deltas = diffFindings(snapshotA, snapshotB);
118
+ const count = (kind) => deltas.filter((delta) => delta.kind === kind).length;
119
+ const lines = [
120
+ '# perf-findings diff',
121
+ provenanceLine('a', fileA, snapshotA),
122
+ provenanceLine('b', fileB, snapshotB),
123
+ '',
124
+ ];
125
+ if (deltas.length) {
126
+ lines.push(...deltas.map((delta) => delta.line), '');
127
+ }
128
+ lines.push(
129
+ `${deltas.length} delta${deltas.length === 1 ? '' : 's'}: `
130
+ + `${count('added')} added, ${count('removed')} removed, ${count('changed')} changed`,
131
+ );
132
+ process.stdout.write(`${lines.join('\n')}\n`);
133
+ return 0;
134
+ }
135
+
136
+ process.exitCode = main(process.argv.slice(2));