@ngockhoale/ukit 2.7.7 → 2.7.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/CHANGELOG.md +54 -0
  2. package/package.json +1 -1
  3. package/src/context/detectProjectContext.js +5 -0
  4. package/src/core/codeintel/invalidation.js +4 -0
  5. package/src/core/diffPlan.js +60 -1
  6. package/src/core/fileOps.js +46 -119
  7. package/src/render/buildVariables.js +10 -0
  8. package/templates/.claude/agents/bug-debugger.md +1 -1
  9. package/templates/.claude/agents/feature-implementer.md +2 -2
  10. package/templates/.claude/commands/ukit/handoff-clear.md +11 -0
  11. package/templates/.claude/commands/ukit/handoff-create.md +1 -1
  12. package/templates/.claude/commands/ukit/handoff-fullstack.md +26 -1
  13. package/templates/.claude/commands/ukit/handoff-implement.md +1 -1
  14. package/templates/.claude/commands/ukit/handoff-review.md +1 -1
  15. package/templates/.claude/hooks/context-hardcap-gate.sh +4 -1
  16. package/templates/.claude/hooks/handoff-model-guard.sh +22 -11
  17. package/templates/.claude/hooks/reset-compact-pressure.sh +10 -0
  18. package/templates/.claude/hooks/skill-router.sh +15 -8
  19. package/templates/.claude/hooks/verification-guard.sh +3 -0
  20. package/templates/.claude/ukit/index/route-task.mjs +237 -32
  21. package/templates/.claude/ukit/index/stale-spec-check.mjs +38 -3
  22. package/templates/.claude/ukit/runtime/async-lock.mjs +240 -42
  23. package/templates/.claude/ukit/runtime/compact-threshold.mjs +5 -2
  24. package/templates/.claude/ukit/runtime/execution-ledger.mjs +217 -17
  25. package/templates/.claude/ukit/runtime/hook-chain-runner.mjs +38 -4
  26. package/templates/.claude/ukit/runtime/hook-payload-store.mjs +131 -0
  27. package/templates/.claude/ukit/runtime/stop-coordinator.mjs +35 -20
  28. package/templates/.claude/ukit/runtime/token-utils.mjs +37 -126
  29. package/templates/.codex/settings.json +1 -5
  30. package/templates/.omp/agents/bug-debugger.md +1 -1
  31. package/templates/.omp/agents/feature-implementer.md +2 -2
  32. package/templates/.omp/hooks/pre/ukit-bridge.js +216 -51
  33. package/templates/docs/AI_HANDOFF/INDEX.md +1 -1
  34. package/templates/docs/AI_HANDOFF/RULES.md +6 -6
  35. package/templates/ukit/storage/config.json +2 -2
@@ -16,13 +16,12 @@ import {
16
16
  markNotified,
17
17
  readExecutionLedger,
18
18
  readRouteState,
19
- recordExecutionReceipt,
20
19
  } from '../../../.claude/ukit/runtime/execution-ledger.mjs';
21
20
  import {
22
21
  PAYLOAD_INLINE_MAX_BYTES,
23
- createPayloadReference,
24
- maybeSweepStalePayloads,
25
- probePayloadIntegrity,
22
+ createPayloadReferenceAsync,
23
+ maybeSweepStalePayloadsAsync,
24
+ probePayloadIntegrityAsync,
26
25
  } from '../../../.claude/ukit/runtime/hook-payload-store.mjs';
27
26
  // TASK-018 review fix round 1: the chain budget is resolved by ONE shared module,
28
27
  // so the runner's inner deadline and this bridge's outer pi.exec timeout can never
@@ -30,6 +29,13 @@ import {
30
29
  // for Edit|Write is a fail-closed transport failure = every edit blocked).
31
30
  import { resolveChainExecTimeoutMs } from '../../../.claude/ukit/runtime/hook-chain-budget.mjs';
32
31
 
32
+ // OMP-3 (TASK-002): pi.exec forwards only {cwd, signal, timeout} to its spawn —
33
+ // an `env` option is dropped (verified against the bundled omp exec). The only
34
+ // channel that reaches the chain runner AND its in-proc .mjs steps is the host
35
+ // environment, so the bridge stamps the harness here: record-execution.mjs
36
+ // reads env.UKIT_HARNESS (default 'claude-code') when journaling receipts.
37
+ process.env.UKIT_HARNESS = 'omp';
38
+
33
39
  export const HOOK_EVENT_MAP = {
34
40
  tool_call: {
35
41
  'Read|Grep|Glob': ['sensitive-data-guard.mjs'],
@@ -140,6 +146,17 @@ function classifyFailure(scriptName) {
140
146
  // closed (the .sh thin wrapper is still invocable as a fallback path).
141
147
  const TIMEOUT_STAYS_CLOSED = new Set(['block-dangerous.sh', 'block-dangerous.mjs']);
142
148
 
149
+ // TASK-002 (OMP-4): a hook crash or transport failure whose stderr is a
150
+ // transient filesystem error (EAGAIN/EBUSY/EMFILE/ENFILE/ESTALE — the
151
+ // stalled-external-mount class) produced NO verdict, so it must degrade to a
152
+ // loud "could not verify" warning instead of a block showing raw errno text.
153
+ // The match is deliberately conservative: the errno code must appear in an
154
+ // fs-syscall context (open/read/write/mkdir/stat/rename/unlink/rm), in either
155
+ // order, or the Node "resource temporarily unavailable" phrasing — a genuine
156
+ // block reason containing e.g. "again" in prose never downgrades. ONE pattern
157
+ // shared by translateExecResult and runScriptChain.
158
+ const TRANSIENT_INFRA_STDERR_RE = /\bresource temporarily unavailable\b|\b(?:EAGAIN|EBUSY|EMFILE|ENFILE|ESTALE)\b[\s\S]*?\b(?:open|read|write|mkdir|stat|rename|unlink|rm)\b|\b(?:open|read|write|mkdir|stat|rename|unlink|rm)\b[\s\S]*?\b(?:EAGAIN|EBUSY|EMFILE|ENFILE|ESTALE)\b/i;
159
+
143
160
  // TASK-018: the hook-chain-runner's failure taxonomy. Infrastructure outcomes
144
161
  // (overflow / timeout / signal / budget-exhausted) produced NO verdict, so their
145
162
  // captured output is untrustworthy and never reaches the model context, a block
@@ -184,10 +201,18 @@ function redactDiagnosticText(text, maxChars = 500) {
184
201
 
185
202
  function runtimeMetadata(event = {}, context = {}) {
186
203
  const sessionManager = context?.sessionManager;
204
+ const transcriptPath = event.transcriptPath ?? event.transcript_path ?? sessionManager?.getSessionFile?.();
187
205
  return {
188
- sessionId: event.sessionId ?? event.session_id ?? sessionManager?.getSessionId?.(),
206
+ // F-8: one session key for EVERY event. omp session_start events carry no
207
+ // session id and the context may lack sessionManager — without one,
208
+ // reset-compact-pressure.sh takes the unknown-caller branch and wipes EVERY
209
+ // session's pressure records. Falling back to the transcript path here (not
210
+ // per-caller) keeps the reset key identical to the key the pressure writer
211
+ // used on prompt/tool events that also lack an id; a per-caller fallback
212
+ // let the two diverge and the targeted delete silently missed.
213
+ sessionId: event.sessionId ?? event.session_id ?? sessionManager?.getSessionId?.() ?? transcriptPath,
189
214
  cwd: context?.cwd ?? event.cwd,
190
- transcriptPath: event.transcriptPath ?? event.transcript_path ?? sessionManager?.getSessionFile?.(),
215
+ transcriptPath,
191
216
  };
192
217
  }
193
218
 
@@ -323,6 +348,18 @@ function translateExecResult(scriptName, execResult) {
323
348
  }
324
349
  return { block: true, reason: stderr || `${scriptName} exited 2 (blocked)`, stdout, stderr };
325
350
  }
351
+ // TASK-002 (OMP-4): a crash whose stderr is a transient fs error produced no
352
+ // verdict — it is an infrastructure event, not a safety decision. Fail open
353
+ // loudly instead of blocking on raw errno text. TIMEOUT_STAYS_CLOSED scripts
354
+ // (block-dangerous .sh/.mjs) keep their never-fail-open contract.
355
+ if (TRANSIENT_INFRA_STDERR_RE.test(stderr) && !TIMEOUT_STAYS_CLOSED.has(scriptName)) {
356
+ return {
357
+ block: false,
358
+ warning: `${scriptName} exited ${code} with a transient filesystem error — treated as "could not verify", not as a block (an infrastructure event, not a verdict): ${redactDiagnosticText(stderr) || 'no stderr'}`,
359
+ stdout,
360
+ stderr,
361
+ };
362
+ }
326
363
  if (classifyFailure(scriptName) === 'closed') {
327
364
  return {
328
365
  block: true,
@@ -366,17 +403,18 @@ function hookErrorsDirFor(projectRoot) {
366
403
  }
367
404
 
368
405
  // Bounded work on THIS session's file only — identical keep-newest-half shape
369
- // as hook-telemetry's rotateIfNeeded.
370
- function rotateHookErrorFileIfNeeded(filePath, incomingBytes, maxBytes) {
406
+ // as hook-telemetry's rotateIfNeeded. TASK-002 (OMP-4): async fs only — a
407
+ // stalled mount must never freeze the omp host loop.
408
+ async function rotateHookErrorFileIfNeeded(filePath, incomingBytes, maxBytes) {
371
409
  let size = 0;
372
410
  try {
373
- size = fs.statSync(filePath).size;
411
+ size = (await fs.promises.stat(filePath)).size;
374
412
  } catch {
375
413
  return; // first row for this session
376
414
  }
377
415
  if (size + incomingBytes <= maxBytes) return;
378
416
  try {
379
- const lines = fs.readFileSync(filePath, 'utf8').split('\n');
417
+ const lines = (await fs.promises.readFile(filePath, 'utf8')).split('\n');
380
418
  if (lines.length && lines[lines.length - 1] === '') lines.pop();
381
419
  const keepBudget = Math.floor(maxBytes / 2);
382
420
  const keep = [];
@@ -387,16 +425,17 @@ function rotateHookErrorFileIfNeeded(filePath, incomingBytes, maxBytes) {
387
425
  keep.unshift(lines[i]);
388
426
  kept += lineBytes;
389
427
  }
390
- fs.writeFileSync(filePath, keep.length ? `${keep.join('\n')}\n` : '', 'utf8');
428
+ await fs.promises.writeFile(filePath, keep.length ? `${keep.join('\n')}\n` : '', 'utf8');
391
429
  } catch {
392
430
  // Rotation failed; drop this row rather than grow past the cap.
393
431
  }
394
432
  }
395
433
 
396
434
  // sweepHookErrorsDir(dir, {now, maxAgeMs, maxFiles, maxEntries, maxRemovals}) ->
397
- // { scanned, removed } — bounded: never scans or removes more than the caps,
398
- // so a pre-existing oversized dir is amortized down across sampled sweeps.
399
- function sweepHookErrorsDir(dir, {
435
+ // Promise<{ scanned, removed }> — bounded: never scans or removes more than
436
+ // the caps, so a pre-existing oversized dir is amortized down across sampled
437
+ // sweeps. TASK-002 (OMP-4): async fs only — host loop must never block.
438
+ async function sweepHookErrorsDir(dir, {
400
439
  now = Date.now,
401
440
  maxAgeMs = HOOK_ERROR_MAX_AGE_MS,
402
441
  maxFiles = HOOK_ERROR_MAX_FILES,
@@ -405,7 +444,7 @@ function sweepHookErrorsDir(dir, {
405
444
  } = {}) {
406
445
  let names;
407
446
  try {
408
- names = fs.readdirSync(dir);
447
+ names = await fs.promises.readdir(dir);
409
448
  } catch {
410
449
  return { scanned: 0, removed: 0 };
411
450
  }
@@ -421,7 +460,7 @@ function sweepHookErrorsDir(dir, {
421
460
  if (scanned >= maxEntries) break;
422
461
  scanned += 1;
423
462
  try {
424
- const stats = fs.statSync(path.join(dir, name));
463
+ const stats = await fs.promises.stat(path.join(dir, name));
425
464
  if (stats.isFile()) entries.push({ name, mtimeMs: stats.mtimeMs });
426
465
  } catch { /* raced away — fine */ }
427
466
  }
@@ -435,7 +474,7 @@ function sweepHookErrorsDir(dir, {
435
474
  if (removed >= maxRemovals) break;
436
475
  if (removed < overflow || entry.mtimeMs < cutoff) {
437
476
  try {
438
- fs.rmSync(path.join(dir, entry.name), { force: true });
477
+ await fs.promises.rm(path.join(dir, entry.name), { force: true });
439
478
  removed += 1;
440
479
  } catch { /* raced away — fine */ }
441
480
  }
@@ -449,18 +488,18 @@ function hookErrorsSweepProbabilityFromEnv() {
449
488
  return Math.min(1, Math.max(0, raw));
450
489
  }
451
490
 
452
- function recordHookErrorDiagnostic(projectRoot, sessionId, diagnostic, { maxBytes = HOOK_ERROR_MAX_BYTES } = {}) {
491
+ async function recordHookErrorDiagnostic(projectRoot, sessionId, diagnostic, { maxBytes = HOOK_ERROR_MAX_BYTES } = {}) {
453
492
  try {
454
493
  const dir = hookErrorsDirFor(projectRoot);
455
- fs.mkdirSync(dir, { recursive: true });
494
+ await fs.promises.mkdir(dir, { recursive: true });
456
495
  const safeSession = String(sessionId || 'unknown').replace(/[^a-zA-Z0-9._-]/g, '_').slice(0, 96) || 'unknown';
457
496
  const filePath = path.join(dir, `${safeSession}.jsonl`);
458
497
  const line = `${JSON.stringify(diagnostic)}\n`;
459
- rotateHookErrorFileIfNeeded(filePath, Buffer.byteLength(line, 'utf8'), maxBytes);
460
- fs.appendFileSync(filePath, line, 'utf8');
498
+ await rotateHookErrorFileIfNeeded(filePath, Buffer.byteLength(line, 'utf8'), maxBytes);
499
+ await fs.promises.appendFile(filePath, line, 'utf8');
461
500
  // Sampled bounded dir sweep — amortizes down any pre-existing oversized dir.
462
501
  if (Math.random() < hookErrorsSweepProbabilityFromEnv()) {
463
- sweepHookErrorsDir(dir);
502
+ await sweepHookErrorsDir(dir);
464
503
  }
465
504
  } catch {
466
505
  // Diagnostics are advisory and must never block or throw.
@@ -505,11 +544,11 @@ function payloadsDirFor(projectRoot) {
505
544
  }
506
545
 
507
546
  function schedulePayloadSweep(projectRoot) {
508
- // Deferred: runs after the current turn settles, never inside the chain request.
547
+ // Deferred: runs after the current turn settles, never inside the chain
548
+ // request. Async fs only — a stalled mount must not freeze the host loop.
509
549
  const timer = setTimeout(() => {
510
- try {
511
- maybeSweepStalePayloads(payloadsDirFor(projectRoot));
512
- } catch { /* deferred sweeping is best effort */ }
550
+ maybeSweepStalePayloadsAsync(payloadsDirFor(projectRoot))
551
+ .catch(() => { /* deferred sweeping is best effort */ });
513
552
  }, 0);
514
553
  timer.unref?.();
515
554
  }
@@ -530,7 +569,7 @@ export async function runScriptChain(
530
569
  const scriptPaths = scripts.map((scriptName) => path.join(projectRoot, '.claude', 'hooks', scriptName));
531
570
  const nodeExecutable = resolveNodeExecutable();
532
571
  const startedAt = Date.now();
533
- const payloadReference = createPayloadReference(JSON.stringify(payload), {
572
+ const payloadReference = await createPayloadReferenceAsync(JSON.stringify(payload), {
534
573
  maxBytes: PAYLOAD_INLINE_MAX_BYTES,
535
574
  dir: payloadsDirFor(projectRoot),
536
575
  });
@@ -549,8 +588,8 @@ export async function runScriptChain(
549
588
  // TASK-031: verify the staged payload survived the chain intact BEFORE removing it —
550
589
  // a file that vanished or was truncated mid-flight means the scripts ran against a
551
590
  // different payload than the host captured, so their verdicts are void.
552
- payloadProbe = probePayloadIntegrity(payloadReference);
553
- payloadReference.cleanup();
591
+ payloadProbe = await probePayloadIntegrityAsync(payloadReference);
592
+ await payloadReference.cleanupAsync();
554
593
  if (payloadReference.mode === 'file') schedulePayloadSweep(projectRoot);
555
594
  }
556
595
  const elapsedMs = Date.now() - startedAt;
@@ -595,7 +634,7 @@ export async function runScriptChain(
595
634
  parseError: parseError?.message || null,
596
635
  wrapperError: chainResult?.wrapperError || null,
597
636
  };
598
- recordHookErrorDiagnostic(projectRoot, payload.session_id, diagnostic);
637
+ await recordHookErrorDiagnostic(projectRoot, payload.session_id, diagnostic);
599
638
  // TASK-031: a staged payload that was lost or corrupted mid-chain voids every
600
639
  // verdict below it. Chains that must not fail open on an unverifiable verdict
601
640
  // (Edit|Write transport policy, and block-dangerous's never-fail-open rule)
@@ -626,6 +665,15 @@ export async function runScriptChain(
626
665
  + `(killed=${diagnostic.killed}, code=${diagnostic.code}, elapsedMs=${diagnostic.elapsedMs}, `
627
666
  + `runtime=${diagnostic.nodeExecutable}). No safety-gate verdict was available for [${scripts.join(', ')}]. `
628
667
  + `See .ukit/storage/cache/hook-errors/.${nodePathHint}`;
668
+ // TASK-002 (OMP-4): a transport failure whose stderr is a transient fs
669
+ // error produced no verdict — an infrastructure event, not a safety
670
+ // decision. Even on a fail-closed chain it degrades to a loud warning
671
+ // instead of a block showing raw errno text. A staged-payload integrity
672
+ // failure (payloadProbe !== null) already returned above and stays closed.
673
+ if (failClosedOnTransportError && TRANSIENT_INFRA_STDERR_RE.test(diagnostic.stderrExcerpt)) {
674
+ pi.logger?.warn?.(`[UKit] ${reason} Transient filesystem error — treated as "could not verify", not as a block.`);
675
+ return { block: false, context, invoked };
676
+ }
629
677
  if (failClosedOnTransportError) {
630
678
  return { block: true, reason, context, invoked };
631
679
  }
@@ -767,17 +815,13 @@ export async function runToolResult(pi, event, { projectRoot, context: extension
767
815
  toolUseId: event.toolCallId,
768
816
  ...metadata,
769
817
  });
818
+ // OMP-3 (TASK-002): the receipt is journaled by exactly ONE path — the
819
+ // record-execution.mjs chain step inside runScriptChain (the same step Claude
820
+ // Code runs, carrying the runner's deadline/signal into the ledger lock). The
821
+ // former direct recordExecutionReceipt({harness:'omp'}) call here minted a
822
+ // second receipt per tool result, so one failed test counted as streak 2 and
823
+ // tripped the verification-loop blocker after a single failure.
770
824
  const result = await runScriptChain(pi, scriptsForToolResult(toolName), payload, { projectRoot });
771
- try {
772
- await recordExecutionReceipt({
773
- projectRoot,
774
- payload,
775
- toolName,
776
- harness: 'omp',
777
- });
778
- } catch (error) {
779
- pi.logger?.warn?.(`[UKit] execution receipt failed open: ${error?.message || error}`);
780
- }
781
825
 
782
826
  const hookOutput = result.context.join('\n').trim();
783
827
  if (!hookOutput) return undefined;
@@ -849,19 +893,36 @@ export async function runSessionCompact(pi, event, { projectRoot, context: exten
849
893
  // harnesses; a clean evaluation deletes it, so the count is consecutive crashes.
850
894
  const COMPLETION_CRASH_STREAK_FILE = path.join('.ukit', 'storage', 'cache', 'completion-gate-crash.streak');
851
895
 
852
- async function bumpCompletionCrashStreak(projectRoot) {
896
+ // OMP-1: the bridge is a long-lived in-proc module, so a streak that cannot persist
897
+ // (unwritable cache dir, stalled volume, lock-busy sibling writers) still advances
898
+ // in memory per project root. Pre-fix the catch path returned a constant 1 — the
899
+ // 3-crash loud release was unreachable and every omp Stop blocked forever. The
900
+ // in-memory count is the floor; a persisted count higher than it wins on the next
901
+ // successful read so cross-harness streaks still converge.
902
+ const crashStreakFallback = new Map();
903
+
904
+ async function bumpCompletionCrashStreak(projectRoot, pi = null) {
853
905
  const file = path.join(projectRoot, COMPLETION_CRASH_STREAK_FILE);
854
906
  try {
855
907
  await fs.promises.mkdir(path.dirname(file), { recursive: true });
856
908
  await fs.promises.appendFile(file, '.');
857
909
  const buf = await fs.promises.readFile(file);
858
- return buf.length || 1;
859
- } catch {
860
- return 1;
910
+ const count = Math.max(buf.length || 1, (crashStreakFallback.get(projectRoot) || 0) + 1);
911
+ crashStreakFallback.set(projectRoot, count);
912
+ return count;
913
+ } catch (error) {
914
+ const count = (crashStreakFallback.get(projectRoot) || 0) + 1;
915
+ crashStreakFallback.set(projectRoot, count);
916
+ pi?.logger?.warn?.(
917
+ `[UKit] completion crash streak could not persist (${error?.message || error}); `
918
+ + `counting in memory (streak ${count}) so the breaker still advances.`,
919
+ );
920
+ return count;
861
921
  }
862
922
  }
863
923
 
864
924
  async function resetCompletionCrashStreak(projectRoot) {
925
+ crashStreakFallback.delete(projectRoot);
865
926
  try {
866
927
  await fs.promises.rm(path.join(projectRoot, COMPLETION_CRASH_STREAK_FILE), { force: true });
867
928
  } catch { /* best effort */ }
@@ -879,6 +940,38 @@ async function readStopGateMaxCrashStreaks(projectRoot) {
879
940
  }
880
941
  }
881
942
 
943
+ // ── TASK-007 (F-10/OMP-2): stop-path parity with stop-coordinator.mjs ─────────
944
+ // Claude's Stop runs one coordinator with three protections the omp bridge never
945
+ // had: the handoff-cursor lane (a non-done docs/AI_HANDOFF/RUN.md bounces the
946
+ // stop with the run's own Next: line, bounded by the stopGateMaxStalledBlocks
947
+ // liveness breaker), the 2s double-Stop dedupe window (one logical stop burns
948
+ // exactly one continuation), and a 600ms budget on every ledger/state lock.
949
+ // The evaluators are reused from stop-coordinator.mjs itself so semantics and
950
+ // the shared state file (.ukit/storage/cache/stop-coordinator/state.json) stay
951
+ // byte-identical across both engines.
952
+ const STOP_GATE_LOCK_BUDGET_MS = 600;
953
+ const HANDOFF_PROVENANCE_PREFIX = '[ukit-stop-coordinator] handoff-cursor also requested a block: ';
954
+
955
+ let stopCoordinatorModulePromise = null;
956
+
957
+ // Lazy + env-scrubbed: stop-coordinator.mjs arms a process.exit self-deadline at
958
+ // module top level whenever UKIT_HOOK_DEADLINE_MS is set. That env var is only
959
+ // ever set for spawned hook children, but the bridge is a long-lived in-proc
960
+ // module — a leaked value would kill the whole omp host. Same scrub the
961
+ // chain-runner applies around in-proc .mjs steps (hook-chain-runner.mjs:180).
962
+ function loadStopCoordinatorModule() {
963
+ if (!stopCoordinatorModulePromise) {
964
+ const scrubbed = 'UKIT_HOOK_DEADLINE_MS' in process.env
965
+ ? process.env.UKIT_HOOK_DEADLINE_MS : undefined;
966
+ delete process.env.UKIT_HOOK_DEADLINE_MS;
967
+ stopCoordinatorModulePromise = import('../../../.claude/ukit/runtime/stop-coordinator.mjs')
968
+ .finally(() => {
969
+ if (scrubbed !== undefined) process.env.UKIT_HOOK_DEADLINE_MS = scrubbed;
970
+ });
971
+ }
972
+ return stopCoordinatorModulePromise;
973
+ }
974
+
882
975
  export async function runSessionStop(
883
976
  pi,
884
977
  event,
@@ -887,10 +980,69 @@ export async function runSessionStop(
887
980
  context: extensionContext = {},
888
981
  state: suppliedState,
889
982
  ledger: suppliedLedger,
983
+ now = Date.now(),
984
+ dedupeWindowMs,
985
+ lockBudgetMs = STOP_GATE_LOCK_BUDGET_MS,
890
986
  },
891
987
  ) {
892
- const metadata = runtimeMetadata(event, extensionContext);
988
+ // TASK-007 (OMP-2d): the stop path keys its session identity ONLY from the
989
+ // event itself — never the context's sessionManager fallback. omp emits
990
+ // session_id on every top-level session_stop (verified against the bundled
991
+ // binary: emitSessionStop carries session_id/session_file/stop_hook_active),
992
+ // so an event without one is a nested/foreign stop; attributing it to the
993
+ // parent session would burn the parent's continuation budget and dedupe slot.
994
+ const metadata = {
995
+ sessionId: event?.sessionId ?? event?.session_id,
996
+ cwd: extensionContext?.cwd ?? event?.cwd,
997
+ transcriptPath: event?.transcriptPath ?? event?.transcript_path ?? event?.session_file,
998
+ };
893
999
  const payload = buildHookPayload('Stop', metadata);
1000
+
1001
+ // TASK-007 (OMP-2b): the coordinator's once-per-stop dedupe. omp can deliver
1002
+ // two session_stop events for one logical stop; the second inside the window
1003
+ // is skipped entirely — it must not burn a second continuation.
1004
+ let coordinator = null;
1005
+ try {
1006
+ coordinator = await loadStopCoordinatorModule();
1007
+ } catch (error) {
1008
+ // Advisory lane: a missing/unloadable coordinator module must never wedge a
1009
+ // session — the completion gate below still owns the stop.
1010
+ pi.logger?.warn?.(`[UKit] stop-coordinator module unavailable (dedupe + handoff lanes skipped): ${error?.message || error}`);
1011
+ }
1012
+ if (coordinator) {
1013
+ const sessionKey = typeof payload.session_id === 'string' && payload.session_id
1014
+ ? payload.session_id
1015
+ : (typeof payload.transcript_path === 'string' && payload.transcript_path ? payload.transcript_path : 'no-session');
1016
+ const duplicate = await coordinator.alreadyCoordinatedThisStop({
1017
+ projectRoot,
1018
+ sessionKey,
1019
+ now,
1020
+ dedupeWindowMs: dedupeWindowMs ?? coordinator.DEDUPE_WINDOW_MS,
1021
+ lockBudgetMs,
1022
+ });
1023
+ if (duplicate) return undefined;
1024
+ }
1025
+
1026
+ // TASK-007 (OMP-2c): the handoff-cursor lane. While docs/AI_HANDOFF/RUN.md
1027
+ // reports a phase outside {done, blocked} the run owns this stop — the
1028
+ // cursor's Next: line is the continuation instruction. Advisory on failure,
1029
+ // bounded by the stopGateMaxStalledBlocks breaker (always advances post-S3).
1030
+ let handoff = null;
1031
+ if (coordinator) {
1032
+ try {
1033
+ handoff = await coordinator.evaluateHandoffCursor({ projectRoot, now, lockBudgetMs });
1034
+ } catch (error) {
1035
+ pi.logger?.warn?.(`[UKit] handoff-cursor evaluator failed (advisory lane): ${error?.message || error}`);
1036
+ }
1037
+ }
1038
+ const handoffBlock = handoff?.kind === 'block' ? handoff.reason : null;
1039
+ // A breaker-release advisory (or any handoff systemMessage) is the omp
1040
+ // equivalent of the coordinator's systemMessage channel — surface it once,
1041
+ // visibly, whichever way the stop itself resolves.
1042
+ if (handoff?.kind === 'advisory' && handoff.systemMessage) {
1043
+ sendContext(pi, [handoff.systemMessage], 'nextTurn', { display: true });
1044
+ }
1045
+
894
1046
  const state = suppliedState ?? await readRouteState(projectRoot, payload);
895
1047
  const ledger = suppliedLedger ?? await readExecutionLedger(projectRoot, payload) ?? {};
896
1048
  let evaluation;
@@ -899,8 +1051,7 @@ export async function runSessionStop(
899
1051
  } catch (error) {
900
1052
  // BUG-C22-05: fail-closed like the shell gate (continue the turn), but only up
901
1053
  // to the configured consecutive-crash cap — a persistently broken evaluator
902
- // must release loudly instead of looping every Stop forever.
903
- const streak = await bumpCompletionCrashStreak(projectRoot);
1054
+ const streak = await bumpCompletionCrashStreak(projectRoot, pi);
904
1055
  const maxStreak = await readStopGateMaxCrashStreaks(projectRoot);
905
1056
  if (streak > maxStreak) {
906
1057
  await resetCompletionCrashStreak(projectRoot);
@@ -917,7 +1068,8 @@ export async function runSessionStop(
917
1068
  continue: true,
918
1069
  additionalContext:
919
1070
  `UKit stop coordinator: infrastructure failure while evaluating this stop (crash streak ${streak}/${maxStreak}) — `
920
- + 'the stop is blocked fail-closed (details withheld). Run: ukit install, then re-send the task in a new message.',
1071
+ + 'the stop is blocked fail-closed (details withheld). Run: ukit install, then re-send the task in a new message.'
1072
+ + (handoffBlock ? `\n${HANDOFF_PROVENANCE_PREFIX}${handoffBlock}` : ''),
921
1073
  };
922
1074
  }
923
1075
  // A clean evaluation resets the consecutive-crash streak (shared with the shell hook).
@@ -940,20 +1092,33 @@ export async function runSessionStop(
940
1092
  // message) stays as a second guaranteed channel.
941
1093
  sendContext(pi, [notice], 'nextTurn', { display: true });
942
1094
  }
1095
+ // The completion gate released, but an in-flight handoff run still owns the
1096
+ // stop (coordinator merge order: handoff-cursor block outranks a released
1097
+ // completion). Its own stall streak — not the ledger continuation counter —
1098
+ // bounds this lane, so no continuation bookkeeping runs here.
1099
+ if (handoffBlock) {
1100
+ return { continue: true, additionalContext: handoffBlock };
1101
+ }
943
1102
  return undefined;
944
1103
  }
945
1104
 
946
1105
  if (suppliedLedger === undefined) {
947
1106
  try {
948
- if (evaluation.finalNotice) await markNotified(projectRoot, payload, ledger);
949
- else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null, evidencePromptKey(state));
1107
+ // TASK-007 (OMP-2a): the 600ms lock budget — a contended ledger lock must
1108
+ // never hold the omp host near its hook deadline.
1109
+ if (evaluation.finalNotice) await markNotified(projectRoot, payload, ledger, { deadlineMs: lockBudgetMs });
1110
+ else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null, evidencePromptKey(state), { deadlineMs: lockBudgetMs });
950
1111
  } catch (error) {
951
1112
  pi.logger?.warn?.(`[UKit] continuation bookkeeping failed open: ${error?.message || error}`);
952
1113
  }
953
1114
  }
954
1115
  return {
955
1116
  continue: true,
956
- additionalContext: evaluation.reason,
1117
+ // Coordinator merge order: the completion gate owns the continuation, and a
1118
+ // losing handoff-cursor block travels inside the winning reason.
1119
+ additionalContext: handoffBlock
1120
+ ? `${evaluation.reason}\n${HANDOFF_PROVENANCE_PREFIX}${handoffBlock}`
1121
+ : evaluation.reason,
957
1122
  };
958
1123
  }
959
1124
 
@@ -4,7 +4,7 @@
4
4
  Status values (xem RULES.md §Status state machine):
5
5
  ready | in_progress | pending_review | changes_requested | critical_block | approved | approved_minor | blocked | done
6
6
 
7
- Owner = tool đang giữ task: claude-code | kilo-code | codex | -
7
+ Owner = tool đang giữ task: claude-code | codex | omp | -
8
8
  -->
9
9
 
10
10
  | ID | Title | Priority | Size | Status | Owner | Reviewer | File |
@@ -92,21 +92,21 @@ Next: <bước kế tiếp chính xác>
92
92
 
93
93
  ## Handoff Flow (tool-agnostic, file-based state machine)
94
94
 
95
- UKit handoff hoạt động qua **file state**. Anh tự chọn tool nào cho từng phase — Claude Code / Kilo Code / Codex / tool mới sau này — đều được. UKit chỉ care về **role của model**, không care tool.
95
+ UKit handoff hoạt động qua **file state**. Anh tự chọn tool nào cho từng phase — Claude Code / Codex / omp / tool mới sau này — đều được. UKit chỉ care về **role của model**, không care tool.
96
96
 
97
97
  3 phase × 3 role model:
98
98
 
99
99
  - **Plan** — model mạnh nhất anh có (reasoning model). Có thể chạy ở bất kỳ tool nào hỗ trợ planning tốt.
100
- - **Execute** — model rẻ-mà-vẫn-thông-minh (code model). Có thể là subagent code của Kilo, hay feature-implementer của Claude Code.
101
- - **Review** — **MODEL KHÁC executor** (reasoning model thường tốt hơn). Có thể là tool khác, hoặc cùng tool nhưng subagent khác model (ví dụ Kilo có subagent code và subagent review riêng).
100
+ - **Execute** — model rẻ-mà-vẫn-thông-minh (code model). Có thể là subagent code của omp, hay feature-implementer của Claude Code.
101
+ - **Review** — **MODEL KHÁC executor** (reasoning model thường tốt hơn). Có thể là tool khác, hoặc cùng tool nhưng subagent khác model.
102
102
 
103
103
  Hai mô hình triển khai đều hợp lệ:
104
- - **Cross-tool**: ví dụ Claude (plan) → Kilo (execute) → Claude (review). Bridge qua file.
105
- - **Same-tool different-subagent**: ví dụ Kilo:plan → Kilo:code → Kilo:review, miễn 3 subagent dùng MODEL khác nhau ở role tương ứng.
104
+ - **Cross-tool**: ví dụ Claude (plan) → Codex (execute) → Claude (review). Bridge qua file.
105
+ - **Same-tool different-subagent**: ví dụ omp:plan → omp:task → omp:code-reviewer, miễn 3 subagent dùng MODEL khác nhau ở role tương ứng.
106
106
 
107
107
  Mỗi tool/subagent đọc cùng `INDEX.md` + `tasks/TASK-xxx.md` → chọn task theo `status` → cập nhật status khi xong.
108
108
 
109
- > **Quan trọng — UKit không enforce model:** `handoff.executor.cheapSmartModelHint` và `handoff.reviewer.model` trong `.ukit/storage/config.json` chỉ là **nhãn** để anh biết MUỐN dùng gì. Tool nào dùng model nào là do anh chọn trong settings của tool đó. UKit enforce contract bằng cách bắt executor TỰ KHAI `EXECUTOR_MODEL` trong Executor Report; reviewer so với chính nó và refuse nếu trùng. Vì vậy nếu trong Kilo anh để cả code-subagent và review-subagent đều dùng cùng model → reviewer sẽ tự refuse, không silent-pass.
109
+ > **Quan trọng — UKit không enforce model:** `handoff.executor.cheapSmartModelHint` và `handoff.reviewer.model` trong `.ukit/storage/config.json` chỉ là **nhãn** để anh biết MUỐN dùng gì. Tool nào dùng model nào là do anh chọn trong settings của tool đó. UKit enforce contract bằng cách bắt executor TỰ KHAI `EXECUTOR_MODEL` trong Executor Report; reviewer so với chính nó và refuse nếu trùng. Vì vậy nếu anh để cả executor-subagent và review-subagent đều dùng cùng model → reviewer sẽ tự refuse, không silent-pass.
110
110
 
111
111
  ### Status state machine
112
112
 
@@ -442,7 +442,7 @@
442
442
  "field": "handoff.reviewer.model",
443
443
  "mac_dinh": "unic-smart",
444
444
  "y_nghia": "Model dùng cho reviewer agent ở Phase 3. BẮT BUỘC khác model executor để bắt được lỗi mà executor miss. Có thể dùng claude-opus-5, unic-smart, hoặc bất kỳ model reasoning mạnh nào.",
445
- "vi_du": "Nếu executor là unic-code (Kilo Code), set reviewer.model=unic-smart hoặc claude-opus-5. Nếu executor là claude-sonnet, set reviewer thành claude-opus."
445
+ "vi_du": "Nếu executor là unic-code, set reviewer.model=unic-smart hoặc claude-opus-5. Nếu executor là claude-sonnet, set reviewer thành claude-opus."
446
446
  },
447
447
  "tat_reviewer_phase": {
448
448
  "field": "handoff.reviewer.enabled",
@@ -601,7 +601,7 @@
601
601
  },
602
602
  "handoff": {
603
603
  "enabled": "Bật Quality Gate cho handoff: plan có Test Plan, executor test-first, reviewer model khác. Tắt = quay về flow cũ (dễ lọt lỗi vặt).",
604
- "crossTool": "true nghĩa là handoff truyền qua file (PLAN/INDEX/tasks) chứ không qua in-process subagent — cho phép plan ở Claude Code, execute ở Kilo Code, review ở Claude Code khác model.",
604
+ "crossTool": "true nghĩa là handoff truyền qua file (PLAN/INDEX/tasks) chứ không qua in-process subagent — cho phép plan ở một tool, execute ở tool khác, review ở tool thứ ba khác model.",
605
605
  "maxParallelAgents": "Số agent chạy song song TỐI ĐA trong một wave (mặc định 2 để giữ session chính nhẹ và hạn chế rủi ro compact/worktree rác; nếu máy khỏe và task độc lập nhiều thì có thể nâng dần 3-5, tối đa ~10-15). Một wave có nhiều task hơn số này sẽ được chia thành nhiều batch chạy lần lượt — áp dụng cho cả Phase 3 Implement và Phase 4 Review. Lý do giới hạn vẫn còn: mỗi agent nền có context window riêng, và report của agent khi xong sẽ được inject ngược vào session chính — chạy quá nhiều cùng lúc (vd 20+) vẫn có thể làm session chính vượt context window và bỏ lại worktree rác. Hạ xuống 1-2 nếu task nặng (verification output dài) hoặc thấy compact bị trigger liên tục.",
606
606
  "plan": {
607
607
  "requireTestPlan": "Bắt buộc PLAN.md §4 phải có Test Plan trước khi task chuyển ready.",