@ngockhoale/ukit 2.7.6 → 2.7.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/CHANGELOG.md +105 -0
  2. package/package.json +1 -1
  3. package/scripts/install/sync-installed-mirror.mjs +250 -0
  4. package/scripts/perf/audit-perf.mjs +287 -36
  5. package/scripts/perf/diff-perf-findings.mjs +136 -0
  6. package/scripts/perf/perf-findings.json +260 -206
  7. package/scripts/perf/perf-measure.md +271 -0
  8. package/src/context/detectProjectContext.js +5 -0
  9. package/src/core/codeintel/invalidation.js +4 -0
  10. package/src/core/fileOps.js +40 -117
  11. package/src/core/hookChainDoctor.js +65 -2
  12. package/src/core/memory/store.js +22 -1
  13. package/src/core/taskBudgetValidator.js +9 -6
  14. package/src/core/unattendedDoctor.js +8 -1
  15. package/src/render/buildVariables.js +10 -0
  16. package/templates/.claude/agents/bug-debugger.md +1 -1
  17. package/templates/.claude/agents/feature-implementer.md +2 -2
  18. package/templates/.claude/commands/ukit/handoff-create.md +1 -1
  19. package/templates/.claude/commands/ukit/handoff-fullstack.md +1 -1
  20. package/templates/.claude/commands/ukit/handoff-implement.md +1 -1
  21. package/templates/.claude/commands/ukit/handoff-review.md +1 -1
  22. package/templates/.claude/hooks/auto-prune-bash.sh +19 -0
  23. package/templates/.claude/hooks/context-hardcap-gate.sh +4 -1
  24. package/templates/.claude/hooks/handoff-model-guard.sh +22 -11
  25. package/templates/.claude/hooks/reinject-context.sh +22 -0
  26. package/templates/.claude/hooks/reset-compact-pressure.sh +29 -0
  27. package/templates/.claude/hooks/session-episode.sh +20 -0
  28. package/templates/.claude/hooks/skill-router.sh +15 -8
  29. package/templates/.claude/hooks/verification-guard.sh +3 -0
  30. package/templates/.claude/ukit/index/route-task.mjs +237 -32
  31. package/templates/.claude/ukit/index/task-budget-validator.mjs +6 -2
  32. package/templates/.claude/ukit/runtime/async-lock.mjs +144 -10
  33. package/templates/.claude/ukit/runtime/compact-threshold.mjs +5 -2
  34. package/templates/.claude/ukit/runtime/execution-ledger.mjs +217 -17
  35. package/templates/.claude/ukit/runtime/hook-chain-runner.mjs +156 -24
  36. package/templates/.claude/ukit/runtime/hook-payload-store.mjs +57 -0
  37. package/templates/.claude/ukit/runtime/hook-telemetry.mjs +84 -12
  38. package/templates/.claude/ukit/runtime/hook-telemetry.sh +50 -0
  39. package/templates/.claude/ukit/runtime/stop-coordinator.mjs +35 -20
  40. package/templates/.claude/ukit/runtime/token-utils.mjs +37 -126
  41. package/templates/.codex/settings.json +1 -5
  42. package/templates/.omp/agents/bug-debugger.md +1 -1
  43. package/templates/.omp/agents/feature-implementer.md +2 -2
  44. package/templates/.omp/hooks/pre/ukit-bridge.js +157 -26
  45. package/templates/docs/AI_HANDOFF/INDEX.md +1 -1
  46. package/templates/docs/AI_HANDOFF/RULES.md +6 -6
  47. package/templates/ukit/storage/config.json +2 -2
@@ -294,7 +294,8 @@ export async function readRouteState(projectRoot, payload = {}) {
294
294
  }
295
295
 
296
296
  export async function readExecutionLedger(projectRoot, payload = {}) {
297
- return readJson(ledgerPath(projectRoot, payload), null);
297
+ const target = ledgerPath(projectRoot, payload);
298
+ return mergeContinuationSidecar(await readJson(target, null), target);
298
299
  }
299
300
 
300
301
  function promptKeyFromText(promptText) {
@@ -313,7 +314,7 @@ function hasUnfinishedCompletion(state = {}, ledger = {}) {
313
314
  if (!IMPLEMENT_MODES.has(mode)) return false;
314
315
  const required = requiredEvidence(state);
315
316
  if (required.length === 0) return false;
316
- return required.some((item) => !evidenceSatisfied(item, ledger, state?.routeSummary || {}));
317
+ return required.some((item) => !evidenceSatisfied(item, ledger, state));
317
318
  }
318
319
 
319
320
  function resumeSessionHash(sessionId) {
@@ -400,9 +401,38 @@ function explicitError(payload = {}) {
400
401
  payload.tool_result?.is_error,
401
402
  payload.tool_response?.isError,
402
403
  payload.tool_response?.is_error,
404
+ // omp tool details carry their own error flag (eval cells, bash timeouts).
405
+ payload.details?.isError,
406
+ payload.tool_output?.details?.isError,
407
+ payload.tool_result?.details?.isError,
403
408
  ].some((value) => value === true);
404
409
  }
405
410
 
411
+ // omp tool results do not all spell the exit code the same way: bash details
412
+ // carry `exitCode` (non-zero only), other producers serialize `exit_code`, and
413
+ // eval results report per-cell codes under `details.cells[].exitCode`. A missed
414
+ // shape used to journal a failed verification as exitCode null → success, so
415
+ // evidence never satisfied and the Stop gate bounced to the cap.
416
+ function detailsExitCode(details) {
417
+ if (!details || typeof details !== 'object') return null;
418
+ for (const value of [details.exitCode, details.exit_code]) {
419
+ // Number(null) === 0 — a JSON `"exitCode": null` must fall through to the
420
+ // next probe (exit_code, then cells[]), never journal as a successful exit.
421
+ if (value === undefined || value === null) continue;
422
+ const number = Number(value);
423
+ if (Number.isFinite(number)) return number;
424
+ }
425
+ const cells = Array.isArray(details.cells) ? details.cells : [];
426
+ let lastCode = null;
427
+ for (const cell of cells) {
428
+ const number = Number(cell?.exitCode ?? cell?.exit_code);
429
+ if (!Number.isFinite(number)) continue;
430
+ if (number !== 0) return number;
431
+ lastCode = number;
432
+ }
433
+ return lastCode;
434
+ }
435
+
406
436
  function extractExitCode(payload = {}) {
407
437
  const candidates = [
408
438
  payload.tool_output?.exitCode,
@@ -415,8 +445,20 @@ function extractExitCode(payload = {}) {
415
445
  payload.tool_response?.exit_code,
416
446
  payload.exitCode,
417
447
  payload.exit_code,
448
+ // omp details shapes: the bridge mirrors event.details under BOTH
449
+ // tool_output.details and tool_result.details, and eval results nest the
450
+ // real code under details.cells[].
451
+ detailsExitCode(payload.tool_output?.details),
452
+ detailsExitCode(payload.tool_result?.details),
453
+ detailsExitCode(payload.details),
454
+ detailsExitCode(payload.tool_response?.details),
455
+ detailsExitCode(payload.tool_output),
456
+ detailsExitCode(payload.tool_result),
418
457
  ];
419
458
  for (const value of candidates) {
459
+ // Number(null) === 0 — a JSON `"exitCode": null` (or a details probe that
460
+ // found nothing) must read as "no code", never as a successful exit.
461
+ if (value === undefined || value === null) continue;
420
462
  const number = Number(value);
421
463
  if (Number.isFinite(number)) return number;
422
464
  }
@@ -801,6 +843,96 @@ function journalQuarantinePathFor(target) {
801
843
  return `${journalPathFor(target)}.quarantine`;
802
844
  }
803
845
 
846
+ // F-4: the continuation counter is STATE, not a journal row. When the journal is
847
+ // full (or its lock unavailable) a continuation/notified event used to be rejected
848
+ // outright, so MAX_CONTINUATIONS fired several bounces late — or never while the
849
+ // ledger lock stayed contended. Rejected counter events land in this small sidecar
850
+ // instead; readExecutionLedger merges it into the view every evaluator sees, and
851
+ // the next committed ledger write folds it in and removes it.
852
+ function continuationSidecarPathFor(target) {
853
+ return `${target}.continuations.json`;
854
+ }
855
+
856
+ // Overlay sidecar counter state onto the ledger when the sidecar is newer than the
857
+ // ledger's last continuation. A sidecar at or behind the ledger is already folded
858
+ // (or written for a superseded request) and is ignored — never double-counted.
859
+ // When the main file is absent the sidecar IS the counter state: returning null
860
+ // hid every tick from evaluators and the next committed write deleted the sidecar
861
+ // unread, so the cap fired late. A sidecar without counter state still yields null
862
+ // — the no-ledger contract is preserved.
863
+ async function mergeContinuationSidecar(ledger, target) {
864
+ let sidecar = null;
865
+ try {
866
+ sidecar = JSON.parse(await fs.readFile(continuationSidecarPathFor(target), 'utf8'));
867
+ } catch {
868
+ return ledger;
869
+ }
870
+ if (!sidecar || typeof sidecar !== 'object') return ledger;
871
+ const sidecarAt = Number(sidecar.lastContinuationAt) || 0;
872
+ if (!ledger) {
873
+ const hasState = sidecarAt > 0
874
+ || Number(sidecar.continuationCount) > 0
875
+ || sidecar.notified === true;
876
+ if (!hasState) return null;
877
+ return {
878
+ requestKey: sidecar.requestKey ?? null,
879
+ promptKey: sidecar.promptKey ?? null,
880
+ continuationRequestKey: sidecar.requestKey ?? null,
881
+ continuationCount: Number(sidecar.continuationCount) || 0,
882
+ noProgressCount: Number(sidecar.noProgressCount) || 0,
883
+ lastProgressDigest: sidecar.lastProgressDigest ?? null,
884
+ notified: sidecar.notified === true,
885
+ lastContinuationAt: sidecarAt,
886
+ updatedAt: Number(sidecar.updatedAt) || sidecarAt,
887
+ };
888
+ }
889
+ const ledgerAt = Number(ledger.lastContinuationAt) || 0;
890
+ if (!(sidecarAt > ledgerAt)) return ledger;
891
+ return {
892
+ ...ledger,
893
+ continuationCount: Number(sidecar.continuationCount) || 0,
894
+ noProgressCount: Number(sidecar.noProgressCount) || 0,
895
+ continuationRequestKey: sidecar.requestKey ?? ledger.continuationRequestKey ?? null,
896
+ lastProgressDigest: sidecar.lastProgressDigest ?? ledger.lastProgressDigest ?? null,
897
+ notified: ledger.notified === true || sidecar.notified === true,
898
+ lastContinuationAt: sidecarAt,
899
+ updatedAt: Math.max(Number(ledger.updatedAt) || 0, sidecarAt),
900
+ };
901
+ }
902
+
903
+ // Best-effort counter tick for a continuation/notified event the journal rejected.
904
+ // Reads the merged ledger view (main file + existing sidecar) so consecutive ticks
905
+ // accumulate even while the main lock stays busy. Returns { ok } — a sidecar write
906
+ // failure means the tick is genuinely lost and the caller reports the rejection.
907
+ async function bumpContinuationSidecar(target, event) {
908
+ try {
909
+ const base = (await mergeContinuationSidecar(await readJson(target, null), target)) || {};
910
+ const applied = event.type === 'notified'
911
+ ? { ...base, notified: true, updatedAt: Date.now() }
912
+ : applyContinuationToLedger(base, event);
913
+ // The merge gate requires sidecarAt > ledgerAt. A notified tick inherits the
914
+ // base stamp unchanged, so it must be bumped past the base it was computed
915
+ // from — otherwise the tick is invisible to every evaluator and the cap's
916
+ // final notice loops forever while contention persists. The +1 floor also
917
+ // covers a continuation landing in the same millisecond as the base stamp.
918
+ const baseAt = Number(base.lastContinuationAt) || 0;
919
+ const stampedAt = Math.max(Number(applied.lastContinuationAt) || 0, baseAt + 1);
920
+ await writeJsonAtomic(continuationSidecarPathFor(target), {
921
+ requestKey: event?.requestKey ?? applied.requestKey ?? applied.continuationRequestKey ?? null,
922
+ promptKey: event?.promptKey ?? applied.promptKey ?? null,
923
+ continuationCount: Number(applied.continuationCount) || 0,
924
+ noProgressCount: Number(applied.noProgressCount) || 0,
925
+ lastProgressDigest: applied.lastProgressDigest ?? null,
926
+ notified: applied.notified === true,
927
+ lastContinuationAt: stampedAt,
928
+ updatedAt: Date.now(),
929
+ });
930
+ return { ok: true };
931
+ } catch {
932
+ return { ok: false };
933
+ }
934
+ }
935
+
804
936
  function newEventId() {
805
937
  return `${Date.now().toString(36)}-${crypto.randomBytes(8).toString('hex')}`;
806
938
  }
@@ -1214,6 +1346,11 @@ export async function recordLedgerEvent(event, {
1214
1346
  // Consume the journal only after the ledger write landed: a crash before this line
1215
1347
  // leaves the journal intact and the next drain re-applies idempotently by eventId.
1216
1348
  if (drained.commit) await drained.commit();
1349
+ // The continuation sidecar was already merged into `current` (and therefore into
1350
+ // the ledger just written): its ticks are now durable state, so the sidecar is
1351
+ // retired. A tick landing between the read and this rm is lost — same accepted
1352
+ // window as the journal drain — but the breaker still advances.
1353
+ try { await fs.rm(continuationSidecarPathFor(target), { force: true }); } catch {}
1217
1354
  return { committed: true, eventId, value, drained: drained.applied, quarantined: drained.quarantined };
1218
1355
  });
1219
1356
  if (outcome.ok) {
@@ -1228,9 +1365,16 @@ export async function recordLedgerEvent(event, {
1228
1365
  // NEVER mutated unlocked — journal the event once; the next acquired lock reconciles it.
1229
1366
  const record = buildJournalRecord({ event, eventId, payload, routeState: routeState || null });
1230
1367
  const journalResult = await appendJournalRecord(target, record, { signal, deadlineMs });
1231
- return journalResult.ok
1232
- ? { journaled: true, eventId }
1233
- : { rejected: true, eventId, reason: journalResult.reason };
1368
+ if (journalResult.ok) return { journaled: true, eventId };
1369
+ // F-4: continuation/notified are counter state, not journal rows — a journal-full
1370
+ // (or journal-unavailable) rejection must still advance the counter or
1371
+ // MAX_CONTINUATIONS fires late/never. Land the tick in the sidecar the evaluator
1372
+ // merges; only a sidecar write failure is a real rejection.
1373
+ if (event.type === 'continuation' || event.type === 'notified') {
1374
+ const sidecar = await bumpContinuationSidecar(target, event);
1375
+ if (sidecar.ok) return { counted: true, eventId, reason: journalResult.reason };
1376
+ }
1377
+ return { rejected: true, eventId, reason: journalResult.reason };
1234
1378
  }
1235
1379
 
1236
1380
  // TASK-027: the receipt is classified (kind, file, command, targeted/broad scope) from the
@@ -1291,6 +1435,44 @@ export async function recordExecutionReceipt({
1291
1435
  );
1292
1436
  }
1293
1437
 
1438
+ // F-15: mirror of isExplicitBroadVerificationRequest in
1439
+ // templates/.claude/hooks/verification-guard.sh — the guard uses this prompt signal to
1440
+ // let a broad suite run under a targeted policy, so the completion gate must honor the
1441
+ // same signal when it classifies the resulting broad receipt. Keep both lists in sync.
1442
+ function isExplicitBroadVerificationRequest(promptText) {
1443
+ const text = String(promptText || '').toLowerCase();
1444
+ return [
1445
+ /run (the )?full test suite/,
1446
+ /run all tests/,
1447
+ /full verification/,
1448
+ /verify everything/,
1449
+ /run the whole suite/,
1450
+ /run broad verification/,
1451
+ /full suite/,
1452
+ /chạy (?:toàn bộ|full) test/,
1453
+ /chạy hết test/,
1454
+ /chạy full test suite/,
1455
+ /verify toàn bộ/,
1456
+ /kiểm tra toàn bộ/,
1457
+ // Plan-driven execution prompts often authorize the plan's final broad
1458
+ // verification without spelling it as "full test suite".
1459
+ /(?:run|execute|implement|do)\b.*\bfull\b.*\b(?:prd\s+)?plan\b/,
1460
+ /\bfull\b.*\b(?:prd\s+)?plan\b.*\b(?:verify|verification|test|tests)\b/,
1461
+ /(?:chạy|lam|làm|thực hiện|triển khai)\b.*\bfull\b.*\b(?:prd|plan|ke hoach|kế hoạch)\b/,
1462
+ /\bfull\b.*\b(?:prd|plan|ke hoach|kế hoạch)\b.*\b(?:verify|verification|test|tests|kiểm tra)\b/,
1463
+ ].some((pattern) => pattern.test(text));
1464
+ }
1465
+
1466
+ // The route cache strips prompt text (route-task.mjs *ForCache deletes it), so the live
1467
+ // state is the only place the explicit-broad signal survives for the evaluator.
1468
+ function explicitBroadVerificationRequested(state = {}) {
1469
+ const routingContext = state?.routingContext || {};
1470
+ return [
1471
+ routingContext.lastExplicitUserPromptText,
1472
+ routingContext.promptText,
1473
+ ].some((promptText) => isExplicitBroadVerificationRequest(promptText));
1474
+ }
1475
+
1294
1476
  function requiredEvidence(state = {}) {
1295
1477
  const routeSummary = state?.routeSummary || {};
1296
1478
  const contractEvidence = routeSummary.executionContract?.completionEvidence;
@@ -1300,14 +1482,19 @@ function requiredEvidence(state = {}) {
1300
1482
  return [...new Set(routeSummary.completionState?.missingEvidence || [])];
1301
1483
  }
1302
1484
 
1303
- function evidenceSatisfied(evidence, ledger = {}, routeSummary = {}) {
1485
+ function evidenceSatisfied(evidence, ledger = {}, state = {}) {
1486
+ const routeSummary = state?.routeSummary || {};
1304
1487
  if (evidence === 'write-evidence') return ledger.writeSucceeded === true;
1305
1488
  if (evidence === 'verification-evidence') {
1306
1489
  // WS-C: when the route names concrete verification commands, only a receipt that ran
1307
1490
  // one of them counts — an unrelated `yarn test` no longer satisfies the gate.
1491
+ // F-15: unless the user explicitly requested/approved a broad suite — the same
1492
+ // prompt signal verification-guard.sh honors — in which case a passing
1493
+ // scope='broad' receipt (verificationSucceeded) satisfies the gate too.
1308
1494
  const routedCommands = routedVerificationCommands(routeSummary);
1309
1495
  if (routedCommands.length > 0) {
1310
- return ledger.targetedVerificationSucceeded === true;
1496
+ return ledger.targetedVerificationSucceeded === true
1497
+ || (explicitBroadVerificationRequested(state) && ledger.verificationSucceeded === true);
1311
1498
  }
1312
1499
  return ledger.verificationSucceeded === true;
1313
1500
  }
@@ -1433,7 +1620,7 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
1433
1620
  || ledger.requestKey === state.requestKey
1434
1621
  || (ledger.promptKey && evidencePromptKey(state) === ledger.promptKey);
1435
1622
  const effectiveLedger = sameRequest ? ledger : {};
1436
- const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger, routeSummary));
1623
+ const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger, state));
1437
1624
  if (missingEvidence.length === 0) {
1438
1625
  // Silent success: the route is present and every required evidence is satisfied. Marked
1439
1626
  // `complete` so the CLI dispatch recognizes it BEFORE the loud final else — otherwise a
@@ -1609,12 +1796,19 @@ async function readStdin() {
1609
1796
 
1610
1797
  // Malformed hook input crashes the evaluate BEFORE any route/session can be read, so the only
1611
1798
  // stable identity is the project root. Count consecutive crashes there; past MAX_GATE_CRASHES,
1612
- // release LOUD naming the crash and the remedy instead of blocking a corrupt hook forever. Any
1613
- // counter I/O failure fails SAFE to block (still loud, never silent). A successful evaluate
1614
- // resets the count.
1799
+ // release LOUD naming the crash and the remedy instead of blocking a corrupt hook forever. A
1800
+ // successful evaluate resets the count.
1801
+ //
1802
+ // F-3: the counter exists to break the infinite bounce — so losing the count must
1803
+ // never BECOME the infinite bounce. When the counter write fails, fail OPEN (loud
1804
+ // release) instead of emitting an unconditional block that can never reach the cap.
1615
1805
  async function handleEvaluateCrash(projectRoot, error) {
1616
1806
  const detail = `malformed hook input crashed the completion gate (${error?.message || error})`;
1617
1807
  const blockReason = () => `${detail}. This is the UKit completion gate, not the task — re-send the task in a new message so the hook payload is regenerated.`;
1808
+ const loudRelease = (message) => {
1809
+ process.stderr.write(`[ukit-completion] ${message}\n`);
1810
+ process.stdout.write(`${JSON.stringify({ systemMessage: message })}\n`);
1811
+ };
1618
1812
  let count = 0;
1619
1813
  try {
1620
1814
  const current = await readJson(crashCounterPath(projectRoot), null);
@@ -1625,15 +1819,21 @@ async function handleEvaluateCrash(projectRoot, error) {
1625
1819
  const next = count + 1;
1626
1820
  try {
1627
1821
  await writeJsonAtomic(crashCounterPath(projectRoot), { count: next, updatedAt: Date.now() });
1628
- } catch {
1629
- // Counter unwritable: we cannot bound the crash loop, but silence is never an option.
1630
- process.stdout.write(`${JSON.stringify({ decision: 'block', reason: blockReason() })}\n`);
1822
+ } catch (writeError) {
1823
+ // Counter unwritable: the crash loop can no longer be bounded, so blocking again
1824
+ // would bounce this stop forever. Release loudly — never silently.
1825
+ loudRelease(
1826
+ `UKit completion gate: ${detail}. The crash counter could not be persisted `
1827
+ + `(${writeError?.message || writeError}), so the gate cannot bound repeated crashes and is `
1828
+ + 'releasing loudly instead of blocking forever. Run: ukit install to refresh the runtime, '
1829
+ + 'then re-send the task in a new message.',
1830
+ );
1631
1831
  return;
1632
1832
  }
1633
1833
  if (next >= MAX_GATE_CRASHES) {
1634
- const message = `UKit completion gate: ${detail}. This recurred ${next} times and cannot be bounded, so the stop is releasing loudly instead of blocking again. The hook payload is corrupt — run: ukit install to refresh the runtime, then re-send the task in a new message.`;
1635
- process.stderr.write(`[ukit-completion] ${message}\n`);
1636
- process.stdout.write(`${JSON.stringify({ systemMessage: message })}\n`);
1834
+ loudRelease(
1835
+ `UKit completion gate: ${detail}. This recurred ${next} times and cannot be bounded, so the stop is releasing loudly instead of blocking again. The hook payload is corrupt — run: ukit install to refresh the runtime, then re-send the task in a new message.`,
1836
+ );
1637
1837
  return;
1638
1838
  }
1639
1839
  process.stdout.write(`${JSON.stringify({ decision: 'block', reason: blockReason() })}\n`);
@@ -97,6 +97,20 @@ function chainFailureKind(processResult) {
97
97
  }
98
98
  }
99
99
 
100
+ // B1 (TASK-001, SPEC §FR-001): a child that emits a `hookSpecificOutput`
101
+ // permission decision owns the verdict. `hookSpecificOutput` alone is NOT
102
+ // enough — a child killed mid-write can carry the marker in a truncated
103
+ // capture, and calling that a decision would exempt a genuinely-skipped gate
104
+ // from the fail-closed verdict. A real decision is a clean exit-0 verdict.
105
+ function emitsPermissionDecision(entry) {
106
+ return Boolean(entry)
107
+ && !entry.killed
108
+ && entry.code === 0
109
+ && entry.failureKind === 'ok'
110
+ && typeof entry.stdout === 'string'
111
+ && entry.stdout.includes('"hookSpecificOutput"');
112
+ }
113
+
100
114
  function recordTiming(projectRoot, payload, timing) {
101
115
  // Timing telemetry is advisory and must never delay or block a tool call;
102
116
  // appendTelemetryRow carries the same posture (and the per-session cap).
@@ -109,12 +123,37 @@ function recordTiming(projectRoot, payload, timing) {
109
123
  // A `:0` or negative suffix is rejected (falls back) — zero would mean "no
110
124
  // budget", which silently disables the deadline; that is never a valid hook
111
125
  // contract.
126
+ // TASK-001 (C8 / SPEC §FR-006): "rejected silently" was the bug. A `:0`/`: -1`
127
+ // typo used to leave the suffix on the path, so the child never spawned and the
128
+ // hook silently vanished from the chain. It now warns (never hard-fails — a
129
+ // config typo must not block every tool call in a fail-closed chain) and runs
130
+ // the script under the DEFAULT child budget.
131
+ const TIMEOUT_SUFFIX_RE = /^(.*):(-?\d+(?:\.\d+)?)$/;
132
+
133
+ // The ONE suffix stripper. Identity checks that need a script's declared name
134
+ // (fail-closed lookups, skipped-gate reporting) go through it too: a `:-1` arg
135
+ // must not look like an advisory script there while `parseScriptArg` already
136
+ // normalized it to a gate for execution.
137
+ function stripTimeoutSuffix(arg) {
138
+ const match = TIMEOUT_SUFFIX_RE.exec(String(arg ?? ''));
139
+ return match ? match[1] : String(arg ?? '');
140
+ }
141
+
142
+ function failClosedName(arg) {
143
+ return path.basename(stripTimeoutSuffix(arg));
144
+ }
145
+
112
146
  function parseScriptArg(arg) {
113
- const match = /^(.*):(\d+(?:\.\d+)?)$/.exec(arg || '');
114
- if (!match) return { scriptPath: arg, timeoutMs: null };
115
- const seconds = Number(match[2]);
116
- if (!Number.isFinite(seconds) || seconds <= 0) return { scriptPath: arg, timeoutMs: null };
117
- return { scriptPath: match[1], timeoutMs: Math.round(seconds * 1000) };
147
+ const suffix = TIMEOUT_SUFFIX_RE.exec(String(arg ?? ''));
148
+ if (!suffix) return { scriptPath: arg, timeoutMs: null };
149
+ const seconds = Number(suffix[2]);
150
+ if (!Number.isFinite(seconds) || seconds <= 0) {
151
+ process.stderr.write(
152
+ `[ukit] ignoring invalid timeout suffix "${arg}" — using default child budget\n`,
153
+ );
154
+ return { scriptPath: suffix[1], timeoutMs: null };
155
+ }
156
+ return { scriptPath: suffix[1], timeoutMs: Math.round(seconds * 1000) };
118
157
  }
119
158
 
120
159
  // TASK-001 (SPEC §FR-001, §8): a chain arg ending in `.mjs` is an IN-PROC step —
@@ -178,7 +217,7 @@ async function runModuleStep({ scriptPath, payload, payloadText, projectRoot, en
178
217
  }
179
218
  }
180
219
 
181
- async function run(payloadText, scriptArgs, { chainMarker = true } = {}) {
220
+ async function run(payloadText, scriptArgs, { chainMarker = true, decisionShortCircuit = false, stdinStageMs = null } = {}) {
182
221
  const parsedArgs = scriptArgs.map(parseScriptArg);
183
222
  const scriptPaths = parsedArgs.map((a) => a.scriptPath);
184
223
  const payload = JSON.parse(payloadText || '{}');
@@ -224,7 +263,7 @@ async function run(payloadText, scriptArgs, { chainMarker = true } = {}) {
224
263
  const deadline = startedAt + totalBudgetMs;
225
264
  const results = [];
226
265
  let budgetExhausted = false;
227
-
266
+ let decisionEmitted = false;
228
267
  for (let scriptIndex = 0; scriptIndex < scriptPaths.length; scriptIndex++) {
229
268
  const scriptPath = scriptPaths[scriptIndex];
230
269
  const scriptName = path.basename(scriptPath);
@@ -321,7 +360,20 @@ async function run(payloadText, scriptArgs, { chainMarker = true } = {}) {
321
360
  });
322
361
  }
323
362
 
324
- if (code === 2 || killed || (code !== 0 && FAIL_CLOSED_SCRIPTS.has(scriptName))) {
363
+ // B1 (TASK-001, SPEC §FR-001): under `--emit-verdict` a child that emits a
364
+ // `hookSpecificOutput` permission decision owns the verdict even at exit 0
365
+ // (the direct Claude contract: deny + exit 0 still blocks). The chain stops
366
+ // HERE, before the next child spawns — filtering the decision at emit time
367
+ // instead would still let the next script's side effects run (e.g.
368
+ // pre-edit-backup.sh backing up a file whose edit was just denied) and its
369
+ // stdout would be concatenated in front of the decision JSON, which Claude
370
+ // Code rejects as invalid JSON. The predicate is shared with the emit-time
371
+ // lookup so a truncated capture can never count as a decision here.
372
+
373
+ if (decisionShortCircuit && emitsPermissionDecision(results[results.length - 1])) {
374
+ decisionEmitted = true;
375
+ }
376
+ if (decisionEmitted || code === 2 || killed || (code !== 0 && FAIL_CLOSED_SCRIPTS.has(scriptName))) {
325
377
  break;
326
378
  }
327
379
  }
@@ -331,9 +383,12 @@ async function run(payloadText, scriptArgs, { chainMarker = true } = {}) {
331
383
  // of those unrun scripts is fail-closed, the chain must fail CLOSED — the old
332
384
  // per-script path ran every hook independently, so a timed-out advisory never
333
385
  // skipped a gate. `skippedFailClosed` carries that signal to the verdict.
334
- const skippedFailClosed = scriptPaths
386
+ // ... unless this break WAS the decision: a decision owns the verdict, so the
387
+ // unrun gates are not a fail-closed gap (SPEC §FR-001 — the verdict must stay
388
+ // exit 0 with the decision JSON).
389
+ const skippedFailClosed = !decisionEmitted && scriptPaths
335
390
  .slice(results.length)
336
- .some((p) => FAIL_CLOSED_SCRIPTS.has(path.basename(p)));
391
+ .some((p) => FAIL_CLOSED_SCRIPTS.has(failClosedName(p)));
337
392
 
338
393
  const elapsedMs = Date.now() - startedAt;
339
394
  // TASK-019: versioned rows shared with direct hooks. `outcome` reuses this
@@ -350,6 +405,18 @@ async function run(payloadText, scriptArgs, { chainMarker = true } = {}) {
350
405
  toolName: payload?.tool_name || null,
351
406
  toolUseId: payload?.tool_use_id || null,
352
407
  elapsedMs,
408
+ // O3 (SPEC §FR-009): the row's wall time split into the stdin stage — the
409
+ // bounded read this process waits on while the producer writes — and the
410
+ // chain's own execution. A producer holding the pipe open is then
411
+ // attributable instead of looking like a slow hook (the reported artifact:
412
+ // project-important.sh p95 = 1539ms with no row doing real work ≥ 1s).
413
+ // Clamped to `elapsedMs` because the split describes the row it sits on
414
+ // (the two are measured from different origins). Absent — not 0 and not
415
+ // null — when this invocation never read stdin, so a v1 reader sees the
416
+ // row it always saw.
417
+ ...(stdinStageMs === null
418
+ ? {}
419
+ : { stdinStageMs: Math.min(Math.max(0, Math.round(stdinStageMs)), elapsedMs) }),
353
420
  budgetMs: totalBudgetMs,
354
421
  budgetExhausted,
355
422
  scripts: results.map(({ scriptName, code, killed, failureKind, elapsedMs: scriptElapsedMs }) => ({
@@ -358,12 +425,48 @@ async function run(payloadText, scriptArgs, { chainMarker = true } = {}) {
358
425
  killed,
359
426
  failureKind,
360
427
  elapsedMs: scriptElapsedMs,
428
+ // TASK-007 (SPEC §FR-012): which steps the runner treated as gates. The
429
+ // doctor reads this flag instead of importing FAIL_CLOSED_SCRIPTS — the
430
+ // runner is the only component that knows which paths it gated, and a
431
+ // second copy of a security-relevant list is a drift hazard. Additive and
432
+ // optional: v1 readers that read the five original keys keep working.
433
+ failClosed: FAIL_CLOSED_SCRIPTS.has(scriptName),
361
434
  })),
362
435
  });
363
436
 
364
437
  return { results, elapsedMs, budgetMs: totalBudgetMs, budgetExhausted, skippedFailClosed };
365
438
  }
366
439
 
440
+ // TASK-006 (SPEC §S6, F-RC-2): a timed-out in-proc step abandons its
441
+ // Promise.race loser — when that loser holds ref'd libuv handles (timers,
442
+ // watchers, sockets) the event loop stays pinned and the host waits the full
443
+ // registered 30–73s timeout for a process that already produced its verdict.
444
+ // Every emit path ends here: flush stdout/stderr, then exit explicitly. The
445
+ // flush is bounded — a reader that never drains must not turn the fix into a
446
+ // new stall.
447
+ const FLUSH_EXIT_MS = Number(process.env.UKIT_HOOK_FLUSH_EXIT_MS || 2000);
448
+
449
+ function flushStream(stream) {
450
+ return new Promise((resolve) => {
451
+ try {
452
+ // end() flushes every queued chunk before invoking the callback —
453
+ // including when the stream was already ended/destroyed (the callback
454
+ // still fires, with an error we deliberately ignore).
455
+ stream.end(() => resolve());
456
+ } catch {
457
+ resolve();
458
+ }
459
+ });
460
+ }
461
+
462
+ async function flushAndExit(code) {
463
+ await Promise.race([
464
+ Promise.all([flushStream(process.stdout), flushStream(process.stderr)]),
465
+ new Promise((resolve) => setTimeout(resolve, FLUSH_EXIT_MS).unref()),
466
+ ]);
467
+ process.exit(code);
468
+ }
469
+
367
470
  try {
368
471
  let argv = process.argv.slice(2);
369
472
  // TASK-234: `--emit-verdict` adapts the runner for Claude Code settings.json
@@ -384,8 +487,14 @@ try {
384
487
  // '-' reads the payload from stdin — the form Claude Code hook commands use.
385
488
  let payloadText = payloadArg;
386
489
  let stdinTruncated = false;
490
+ // O3: the stage window is measured HERE, around the bounded read, and not
491
+ // inside run() — run() starts after the payload is already in hand, and a
492
+ // truncated read exits before run() is ever called.
493
+ let stdinStageMs = null;
387
494
  if (payloadArg === '-') {
495
+ const stageStartedAt = Date.now();
388
496
  const staged = await readStdinBounded();
497
+ stdinStageMs = Date.now() - stageStartedAt;
389
498
  payloadText = staged.text;
390
499
  stdinTruncated = staged.truncated;
391
500
  } else if (payloadArg.startsWith('@')) {
@@ -402,8 +511,7 @@ try {
402
511
  // chain carries a fail-closed gate, emit deny now instead of letting gates
403
512
  // pass on a payload they never fully received.
404
513
  if (stdinTruncated) {
405
- const hasFailClosed = scriptPaths.some((p) =>
406
- FAIL_CLOSED_SCRIPTS.has(path.basename(String(p).replace(/:\d+(?:\.\d+)?$/, ''))));
514
+ const hasFailClosed = scriptPaths.some((p) => FAIL_CLOSED_SCRIPTS.has(failClosedName(p)));
407
515
  if (hasFailClosed) {
408
516
  const deny = JSON.stringify({
409
517
  hookSpecificOutput: {
@@ -415,19 +523,25 @@ try {
415
523
  });
416
524
  if (emitVerdict) {
417
525
  process.stdout.write(deny);
418
- process.exitCode = 0;
419
526
  } else {
420
527
  process.stdout.write(JSON.stringify({
421
528
  results: [],
422
529
  wrapperError: 'stdin staging truncated — fail-closed chain refused',
423
530
  stdinTruncated: true,
424
531
  }));
425
- process.exitCode = 2;
426
532
  }
427
- process.exit(emitVerdict ? 0 : 2);
533
+ // TASK-006: flush before exit — a bare process.exit can truncate a
534
+ // verdict still queued on the pipe.
535
+ await flushAndExit(emitVerdict ? 0 : 2);
428
536
  }
429
537
  }
430
- const chain = await run(payloadText, scriptPaths, { chainMarker: !emitVerdict });
538
+ const chain = await run(payloadText, scriptPaths, {
539
+ chainMarker: !emitVerdict,
540
+ // B1: only the --emit-verdict contract short-circuits on a decision; the
541
+ // omp bridge parses the JSON envelope and needs every step's context stdout.
542
+ decisionShortCircuit: emitVerdict,
543
+ stdinStageMs,
544
+ });
431
545
  if (!emitVerdict) {
432
546
  process.stdout.write(JSON.stringify(chain));
433
547
  } else {
@@ -435,10 +549,21 @@ try {
435
549
  // verdict even when it exits 0 (the direct Claude contract: deny + exit 0
436
550
  // still blocks). The first decision wins, matching per-script semantics
437
551
  // where each hook's output is its own verdict and a deny short-circuits.
552
+ // The loose lookup decides which entry is REPORTED as the verdict owner (a
553
+ // non-zero decision emitter stays routed through the code-2/fail-closed
554
+ // paths below, per SPEC §FR-001); the strict predicate decides whether that
555
+ // entry may claim stdout outright.
438
556
  const decisionResult = chain.results.find((r) =>
439
557
  typeof r.stdout === 'string' && r.stdout.includes('"hookSpecificOutput"'));
558
+ const decisionOwnsVerdict = emitsPermissionDecision(decisionResult);
440
559
  const last = decisionResult ?? chain.results[chain.results.length - 1];
441
560
 
561
+ // B1 (SPEC §FR-001): a clean decision owns the ENTIRE stdout — no context
562
+ // text before or after it. Claude Code parses stdout as one JSON document,
563
+ // so leading context turns a valid deny into "Hook JSON output validation
564
+ // failed"; the chain already stopped at the decision, so no later script's
565
+ // output can be concatenated in.
566
+
442
567
  // TASK-234 review fix (critical): context stdout must be REPLAYED, not
443
568
  // dropped. SessionStart/UserPromptSubmit hooks emit plain-text context
444
569
  // (PROJECT_IMPORTANT mandate, skill-router guidance) — replaying only the
@@ -449,15 +574,18 @@ try {
449
574
  .filter((r) => r !== decisionResult && r !== last && typeof r.stdout === 'string' && r.stdout.length > 0)
450
575
  .map((r) => r.stdout)
451
576
  .join('');
452
-
453
- // TASK-234 review fix (critical): a mid-chain break that skipped a
454
- // fail-closed gate must fail CLOSED — the old per-script path ran every
455
- // hook independently, so a killed advisory never skipped a gate.
456
- if (chain.skippedFailClosed) {
577
+ if (decisionOwnsVerdict) {
578
+ process.stdout.write(decisionResult.stdout);
579
+ if (decisionResult.stderr) process.stderr.write(decisionResult.stderr);
580
+ process.exitCode = 0;
581
+ } else if (chain.skippedFailClosed) {
582
+ // TASK-234 review fix (critical): a mid-chain break that skipped a
583
+ // fail-closed gate must fail CLOSED — the old per-script path ran every
584
+ // hook independently, so a killed advisory never skipped a gate.
457
585
  if (contextStdout) process.stdout.write(contextStdout);
458
586
  const skipped = scriptPaths
459
587
  .slice(chain.results.length)
460
- .map((p) => path.basename(String(p).replace(/:\d+(?:\.\d+)?$/, '')))
588
+ .map((p) => failClosedName(p))
461
589
  .filter((name) => FAIL_CLOSED_SCRIPTS.has(name))
462
590
  .join(', ');
463
591
  process.stderr.write(`UKit hook chain broke before fail-closed gate(s) ran: ${skipped}\n`);
@@ -466,7 +594,7 @@ try {
466
594
  // No script ran at all (empty chain or budget spent before the first
467
595
  // child). With fail-closed scripts declared in the chain this must not
468
596
  // fail open.
469
- const hasFailClosed = scriptPaths.some((p) => FAIL_CLOSED_SCRIPTS.has(path.basename(String(p).replace(/:\d+(?:\.\d+)?$/, ''))));
597
+ const hasFailClosed = scriptPaths.some((p) => FAIL_CLOSED_SCRIPTS.has(failClosedName(p)));
470
598
  if (hasFailClosed) {
471
599
  process.stderr.write('UKit hook chain produced no verdict — fail-closed gate did not run\n');
472
600
  process.exitCode = 2;
@@ -497,10 +625,14 @@ try {
497
625
  }
498
626
  }
499
627
  }
628
+ // TASK-006 (F-RC-2): the verdict is on the wire — exit now. Waiting for a
629
+ // natural exit lets an abandoned in-proc step's ref'd handles pin the loop
630
+ // until the host's registered timeout kills us.
631
+ await flushAndExit(process.exitCode ?? 0);
500
632
  } catch (error) {
501
633
  process.stdout.write(JSON.stringify({
502
634
  results: [],
503
635
  wrapperError: error?.message || String(error),
504
636
  }));
505
- process.exitCode = 1;
637
+ await flushAndExit(1);
506
638
  }