@ngockhoale/ukit 2.7.7 → 2.7.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/CHANGELOG.md +42 -0
  2. package/package.json +1 -1
  3. package/src/context/detectProjectContext.js +5 -0
  4. package/src/core/codeintel/invalidation.js +4 -0
  5. package/src/core/fileOps.js +40 -117
  6. package/src/render/buildVariables.js +10 -0
  7. package/templates/.claude/agents/bug-debugger.md +1 -1
  8. package/templates/.claude/agents/feature-implementer.md +2 -2
  9. package/templates/.claude/commands/ukit/handoff-create.md +1 -1
  10. package/templates/.claude/commands/ukit/handoff-fullstack.md +1 -1
  11. package/templates/.claude/commands/ukit/handoff-implement.md +1 -1
  12. package/templates/.claude/commands/ukit/handoff-review.md +1 -1
  13. package/templates/.claude/hooks/context-hardcap-gate.sh +4 -1
  14. package/templates/.claude/hooks/handoff-model-guard.sh +22 -11
  15. package/templates/.claude/hooks/reset-compact-pressure.sh +10 -0
  16. package/templates/.claude/hooks/skill-router.sh +15 -8
  17. package/templates/.claude/hooks/verification-guard.sh +3 -0
  18. package/templates/.claude/ukit/index/route-task.mjs +237 -32
  19. package/templates/.claude/ukit/runtime/async-lock.mjs +144 -10
  20. package/templates/.claude/ukit/runtime/compact-threshold.mjs +5 -2
  21. package/templates/.claude/ukit/runtime/execution-ledger.mjs +217 -17
  22. package/templates/.claude/ukit/runtime/hook-chain-runner.mjs +38 -4
  23. package/templates/.claude/ukit/runtime/hook-payload-store.mjs +57 -0
  24. package/templates/.claude/ukit/runtime/stop-coordinator.mjs +35 -20
  25. package/templates/.claude/ukit/runtime/token-utils.mjs +37 -126
  26. package/templates/.codex/settings.json +1 -5
  27. package/templates/.omp/agents/bug-debugger.md +1 -1
  28. package/templates/.omp/agents/feature-implementer.md +2 -2
  29. package/templates/.omp/hooks/pre/ukit-bridge.js +157 -26
  30. package/templates/docs/AI_HANDOFF/INDEX.md +1 -1
  31. package/templates/docs/AI_HANDOFF/RULES.md +6 -6
  32. package/templates/ukit/storage/config.json +2 -2
@@ -3,6 +3,8 @@ import crypto from 'node:crypto';
3
3
  import fs from 'node:fs/promises';
4
4
  import path from 'node:path';
5
5
 
6
+ import { journalDroppedLockMutation, withAsyncLock } from './async-lock.mjs';
7
+
6
8
  export const DEFAULT_PROMPT_CACHE_MAX_ENTRIES = 20;
7
9
  export const DEFAULT_COMPACT_HISTORY_MAX_ENTRIES = 50;
8
10
  const DEFAULT_MAX_COMPACT_ANCHORS = 3;
@@ -60,7 +62,17 @@ export async function writeJson(filePath, value) {
60
62
  const tempPath = `${filePath}.tmp-${Date.now()}-${Math.random().toString(16).slice(2)}`;
61
63
  try {
62
64
  await fs.writeFile(tempPath, `${JSON.stringify(value, null, 2)}\n`, 'utf8');
63
- await fs.rename(tempPath, filePath);
65
+ try {
66
+ await fs.rename(tempPath, filePath);
67
+ } catch (renameError) {
68
+ // EXDEV: the tmp file and the destination sit on different mounts (union
69
+ // mounts, per-dir bind mounts, tmpfs overlays), so rename cannot link them.
70
+ // The payload is already fully written — copy it over and unlink the tmp.
71
+ // Less atomic than rename, but the update must not be silently lost.
72
+ if (renameError?.code !== 'EXDEV') throw renameError;
73
+ await fs.copyFile(tempPath, filePath);
74
+ await fs.rm(tempPath, { force: true });
75
+ }
64
76
  } catch (error) {
65
77
  try {
66
78
  await fs.rm(tempPath, { force: true });
@@ -74,140 +86,39 @@ export async function writeJson(filePath, value) {
74
86
  const LOCK_STALE_MS = 10_000;
75
87
  const LOCK_MAX_WAIT_MS = 5_000;
76
88
 
77
- function lockBackoffDelayMs() {
78
- return 3 + Math.floor(Math.random() * 9);
79
- }
80
-
81
- function sleep(ms) {
82
- return new Promise((resolve) => setTimeout(resolve, ms));
83
- }
84
-
85
- function isPidAlive(pid) {
86
- try {
87
- process.kill(pid, 0);
88
- return true;
89
- } catch (error) {
90
- // EPERM: the process exists but belongs to another user — still alive.
91
- return error?.code === 'EPERM';
92
- }
93
- }
94
-
95
- async function readLockOwner(lockPath) {
96
- try {
97
- const raw = JSON.parse(await fs.readFile(path.join(lockPath, 'owner'), 'utf8'));
98
- const pid = Number(raw?.pid);
99
- return Number.isInteger(pid) && pid > 0
100
- ? { pid, token: typeof raw?.token === 'string' ? raw.token : null }
101
- : null;
102
- } catch {
103
- return null;
104
- }
105
- }
106
-
107
- // In-process holder registry: same-pid holders are parallel async flows whose liveness a
108
- // pid probe cannot prove, so the module tracks them itself.
109
- const inProcessLockHolders = new Map();
110
-
111
89
  /**
112
90
  * Serialize read-modify-write mutations of a shared state file — across processes
113
91
  * (hook invocations run as separate node processes) and across concurrent async
114
92
  * flows in one process (parallel subagents). The lock is a directory created next
115
93
  * to the target file: `mkdir` is atomic, so exactly one caller can create it.
116
- * Ownership is recorded in an `owner` file inside the lock dir: stale reclaim first
117
- * proves the recorded holder is gone (dead pid, or no in-process holder for our own
118
- * pid — no owner file means a pre-token holder and keeps the legacy mtime-only
119
- * reclaim), so a slow-but-alive holder on a crawling disk is waited out, not stolen.
120
- * Release only removes a dir this acquisition still owns, so a reclaimed-then-
121
- * re-acquired lock is never deleted out from under its successor.
122
- * Liveness wins over strictness: if the lock cannot be acquired within maxWaitMs
123
- * the callback runs anyway (the pre-lock behaviour) — these state files are
124
- * advisory caches, and losing an update beats freezing a hook mid-flight.
125
- * Protocol-compatible with src/core/fileOps.js withFileLock (same `<file>.lock`
126
- * path and owner-file format), so CLI processes and hook processes serialize
127
- * against each other.
94
+ *
95
+ * The entire lock protocol lives in async-lock.mjs (TASK-004 fix round 1): owner
96
+ * stamping (pid + token + pstart), recycled-pid detection via recordedProcessGone,
97
+ * claim+quarantine stale reclaim, and verified release exist exactly once there —
98
+ * this function and the src/core/fileOps.js twin both delegate to it so the two
99
+ * protocol copies can never drift apart again.
100
+ *
101
+ * FAIL-CLOSED (TASK-004, unified with async-lock/ledger policy): if the lock cannot
102
+ * be acquired within maxWaitMs the callback is SKIPPED — never run unlocked — and
103
+ * the drop is journaled to `<file>.lock-drops.jsonl`. These state files are
104
+ * advisory caches: losing an update was already the accepted outcome of the old
105
+ * fail-open race; now it is explicit and journaled instead of a silent torn write.
128
106
  * @param {string} filePath - state file the mutation targets (lock lives beside it)
129
107
  * @param {() => Promise<*>} fn - critical section; its result is returned
130
- * @returns {Promise<*>} whatever fn resolves with
108
+ * @returns {Promise<*|undefined>} whatever fn resolves with, or undefined when the
109
+ * lock wait expired and the mutation was skipped (journaled)
131
110
  */
132
111
  export async function withFileLock(filePath, fn, { staleMs = LOCK_STALE_MS, maxWaitMs = LOCK_MAX_WAIT_MS } = {}) {
133
- const lockPath = `${filePath}.lock`;
134
- const startedAt = Date.now();
135
- const ownerToken = `${process.pid}-${crypto.randomBytes(8).toString('hex')}`;
136
- let locked = false;
137
- let ownerStamped = false;
138
-
139
- while (!locked) {
140
- try {
141
- await fs.mkdir(path.dirname(lockPath), { recursive: true });
142
- await fs.mkdir(lockPath); // atomic acquire — EEXIST means another holder exists
143
- locked = true;
144
- inProcessLockHolders.set(lockPath, ownerToken);
145
- try {
146
- await fs.writeFile(
147
- path.join(lockPath, 'owner'),
148
- `${JSON.stringify({ pid: process.pid, token: ownerToken, ts: Date.now() })}\n`,
149
- 'utf8',
150
- );
151
- ownerStamped = true;
152
- } catch {
153
- ownerStamped = false; // unverifiable release skips removal; stale reclaim cleans up
154
- }
155
- break;
156
- } catch (error) {
157
- if (error?.code !== 'EEXIST') throw error;
158
- }
159
-
160
- // Someone holds the lock. Reclaim it only when the holder is provably gone.
161
- try {
162
- const stat = await fs.stat(lockPath);
163
- if (Date.now() - stat.mtimeMs > staleMs) {
164
- const owner = await readLockOwner(lockPath);
165
- const liveInProcess = inProcessLockHolders.has(lockPath);
166
- // Stealing a live holder reintroduces the exact interleaved-write race this
167
- // lock exists to prevent, and the stolen holder's release then deleted the
168
- // successor's lock. Only a dead pid (or a leaked same-pid dir with no live
169
- // registered flow) may be reclaimed.
170
- const reclaimable = !owner || owner.pid === process.pid
171
- ? !liveInProcess
172
- : !isPidAlive(owner.pid);
173
- if (reclaimable) {
174
- await fs.rm(lockPath, { recursive: true, force: true });
175
- continue; // the slot is free now — retry immediately
176
- }
177
- }
178
- } catch (statError) {
179
- // BUG-C22-16: a persistent stat error (EPERM/ENOTDIR/EIO on a failing
180
- // mount, or ELOOP/ENOENT on a dangling symlink where mkdir still reports
181
- // EEXIST) must not busy-spin — a bare `continue` skipped both the
182
- // maxWait break and the backoff sleep, looping mkdir→stat→throw forever.
183
- if (Date.now() - startedAt >= maxWaitMs) break; // fail open — run unlocked
184
- if (statError?.code === 'ENOENT') continue; // lock vanished — retry immediately
185
- await sleep(lockBackoffDelayMs());
186
- continue;
187
- }
188
-
189
- if (Date.now() - startedAt >= maxWaitMs) break; // fail open — run unlocked
190
- await sleep(lockBackoffDelayMs());
191
- }
192
-
193
- try {
194
- return await fn();
195
- } finally {
196
- if (locked) {
197
- try {
198
- // Remove the lock only if THIS acquisition still owns it: after a stale reclaim
199
- // another holder may already own the dir, and deleting it would unlock their
200
- // critical section for a third waiter.
201
- const current = ownerStamped ? await readLockOwner(lockPath) : null;
202
- if (current && current.token === ownerToken) {
203
- await fs.rm(lockPath, { recursive: true, force: true });
204
- }
205
- if (inProcessLockHolders.get(lockPath) === ownerToken) inProcessLockHolders.delete(lockPath);
206
- } catch {
207
- // best-effort release; a stale lock is reclaimed by the next waiter
208
- }
209
- }
210
- }
112
+ const outcome = await withAsyncLock(filePath, { deadlineMs: maxWaitMs, staleMs }, fn);
113
+ if (outcome?.ok === true) return outcome.value;
114
+ // Fail closed: the mutation is dropped, never run unlocked. The drop is
115
+ // journaled so the lost update is auditable; callers treat undefined as
116
+ // "update skipped" (they already tolerated losing it silently).
117
+ await journalDroppedLockMutation(filePath, {
118
+ reason: 'lock-wait-expired',
119
+ waitedMs: outcome?.waitedMs ?? maxWaitMs,
120
+ });
121
+ return undefined;
211
122
  }
212
123
 
213
124
  export function buildCompactMachineKey(prefix, payload = {}) {
@@ -358,10 +358,6 @@
358
358
  "non-trivial": "full test + lint + typecheck"
359
359
  },
360
360
  "verification": {
361
- "requiredBeforeCompletion": [
362
- "{{runtime.packageManager}} test",
363
- "{{runtime.packageManager}} lint",
364
- "{{runtime.packageManager}} typecheck"
365
- ]
361
+ "requiredBeforeCompletion": {{verification.requiredBeforeCompletion}}
366
362
  }
367
363
  }
@@ -49,7 +49,7 @@ Systematic debugging — understand before fixing.
49
49
 
50
50
  ```
51
51
  STATUS: DONE | BLOCKED | PARTIAL
52
- EXECUTOR_TOOL: [claude-code | kilo-code | codex | other]
52
+ EXECUTOR_TOOL: [claude-code | codex | omp | other]
53
53
  EXECUTOR_MODEL: [exact model name you are running as. "unknown" if you cannot tell.]
54
54
  EXECUTOR_SUBAGENT: [subagent name within your host, if any, else "-"]
55
55
  SUMMARY: [1-2 sentences — root cause and fix]
@@ -71,9 +71,9 @@ and even then, report it, don't ask about it.
71
71
 
72
72
  ```
73
73
  STATUS: DONE | BLOCKED | PARTIAL
74
- EXECUTOR_TOOL: [claude-code | kilo-code | codex | other]
74
+ EXECUTOR_TOOL: [claude-code | codex | omp | other]
75
75
  EXECUTOR_MODEL: [exact model name you are running as — e.g. unic-code, claude-sonnet-4-5, gpt-5-mini. If you truly cannot tell, write "unknown" — reviewer treats unknown as suspicious and asks the human to confirm.]
76
- EXECUTOR_SUBAGENT: [name of the subagent you are, if your host has multiple — e.g. "Kilo:code", "Claude:feature-implementer". Otherwise "-".]
76
+ EXECUTOR_SUBAGENT: [name of the subagent you are, if your host has multiple — e.g. "Claude:feature-implementer", "omp:task". Otherwise "-".]
77
77
  SUMMARY: [1-2 sentences of what was implemented]
78
78
  TEST_PLAN_FOLLOWED: [task §4 / inline / N/A — reason]
79
79
  FILES_CHANGED:
@@ -16,11 +16,10 @@ import {
16
16
  markNotified,
17
17
  readExecutionLedger,
18
18
  readRouteState,
19
- recordExecutionReceipt,
20
19
  } from '../../../.claude/ukit/runtime/execution-ledger.mjs';
21
20
  import {
22
21
  PAYLOAD_INLINE_MAX_BYTES,
23
- createPayloadReference,
22
+ createPayloadReferenceAsync,
24
23
  maybeSweepStalePayloads,
25
24
  probePayloadIntegrity,
26
25
  } from '../../../.claude/ukit/runtime/hook-payload-store.mjs';
@@ -30,6 +29,13 @@ import {
30
29
  // for Edit|Write is a fail-closed transport failure = every edit blocked).
31
30
  import { resolveChainExecTimeoutMs } from '../../../.claude/ukit/runtime/hook-chain-budget.mjs';
32
31
 
32
+ // OMP-3 (TASK-002): pi.exec forwards only {cwd, signal, timeout} to its spawn —
33
+ // an `env` option is dropped (verified against the bundled omp exec). The only
34
+ // channel that reaches the chain runner AND its in-proc .mjs steps is the host
35
+ // environment, so the bridge stamps the harness here: record-execution.mjs
36
+ // reads env.UKIT_HARNESS (default 'claude-code') when journaling receipts.
37
+ process.env.UKIT_HARNESS = 'omp';
38
+
33
39
  export const HOOK_EVENT_MAP = {
34
40
  tool_call: {
35
41
  'Read|Grep|Glob': ['sensitive-data-guard.mjs'],
@@ -184,10 +190,18 @@ function redactDiagnosticText(text, maxChars = 500) {
184
190
 
185
191
  function runtimeMetadata(event = {}, context = {}) {
186
192
  const sessionManager = context?.sessionManager;
193
+ const transcriptPath = event.transcriptPath ?? event.transcript_path ?? sessionManager?.getSessionFile?.();
187
194
  return {
188
- sessionId: event.sessionId ?? event.session_id ?? sessionManager?.getSessionId?.(),
195
+ // F-8: one session key for EVERY event. omp session_start events carry no
196
+ // session id and the context may lack sessionManager — without one,
197
+ // reset-compact-pressure.sh takes the unknown-caller branch and wipes EVERY
198
+ // session's pressure records. Falling back to the transcript path here (not
199
+ // per-caller) keeps the reset key identical to the key the pressure writer
200
+ // used on prompt/tool events that also lack an id; a per-caller fallback
201
+ // let the two diverge and the targeted delete silently missed.
202
+ sessionId: event.sessionId ?? event.session_id ?? sessionManager?.getSessionId?.() ?? transcriptPath,
189
203
  cwd: context?.cwd ?? event.cwd,
190
- transcriptPath: event.transcriptPath ?? event.transcript_path ?? sessionManager?.getSessionFile?.(),
204
+ transcriptPath,
191
205
  };
192
206
  }
193
207
 
@@ -530,7 +544,7 @@ export async function runScriptChain(
530
544
  const scriptPaths = scripts.map((scriptName) => path.join(projectRoot, '.claude', 'hooks', scriptName));
531
545
  const nodeExecutable = resolveNodeExecutable();
532
546
  const startedAt = Date.now();
533
- const payloadReference = createPayloadReference(JSON.stringify(payload), {
547
+ const payloadReference = await createPayloadReferenceAsync(JSON.stringify(payload), {
534
548
  maxBytes: PAYLOAD_INLINE_MAX_BYTES,
535
549
  dir: payloadsDirFor(projectRoot),
536
550
  });
@@ -767,17 +781,13 @@ export async function runToolResult(pi, event, { projectRoot, context: extension
767
781
  toolUseId: event.toolCallId,
768
782
  ...metadata,
769
783
  });
784
+ // OMP-3 (TASK-002): the receipt is journaled by exactly ONE path — the
785
+ // record-execution.mjs chain step inside runScriptChain (the same step Claude
786
+ // Code runs, carrying the runner's deadline/signal into the ledger lock). The
787
+ // former direct recordExecutionReceipt({harness:'omp'}) call here minted a
788
+ // second receipt per tool result, so one failed test counted as streak 2 and
789
+ // tripped the verification-loop blocker after a single failure.
770
790
  const result = await runScriptChain(pi, scriptsForToolResult(toolName), payload, { projectRoot });
771
- try {
772
- await recordExecutionReceipt({
773
- projectRoot,
774
- payload,
775
- toolName,
776
- harness: 'omp',
777
- });
778
- } catch (error) {
779
- pi.logger?.warn?.(`[UKit] execution receipt failed open: ${error?.message || error}`);
780
- }
781
791
 
782
792
  const hookOutput = result.context.join('\n').trim();
783
793
  if (!hookOutput) return undefined;
@@ -849,19 +859,36 @@ export async function runSessionCompact(pi, event, { projectRoot, context: exten
849
859
  // harnesses; a clean evaluation deletes it, so the count is consecutive crashes.
850
860
  const COMPLETION_CRASH_STREAK_FILE = path.join('.ukit', 'storage', 'cache', 'completion-gate-crash.streak');
851
861
 
852
- async function bumpCompletionCrashStreak(projectRoot) {
862
+ // OMP-1: the bridge is a long-lived in-proc module, so a streak that cannot persist
863
+ // (unwritable cache dir, stalled volume, lock-busy sibling writers) still advances
864
+ // in memory per project root. Pre-fix the catch path returned a constant 1 — the
865
+ // 3-crash loud release was unreachable and every omp Stop blocked forever. The
866
+ // in-memory count is the floor; a persisted count higher than it wins on the next
867
+ // successful read so cross-harness streaks still converge.
868
+ const crashStreakFallback = new Map();
869
+
870
+ async function bumpCompletionCrashStreak(projectRoot, pi = null) {
853
871
  const file = path.join(projectRoot, COMPLETION_CRASH_STREAK_FILE);
854
872
  try {
855
873
  await fs.promises.mkdir(path.dirname(file), { recursive: true });
856
874
  await fs.promises.appendFile(file, '.');
857
875
  const buf = await fs.promises.readFile(file);
858
- return buf.length || 1;
859
- } catch {
860
- return 1;
876
+ const count = Math.max(buf.length || 1, (crashStreakFallback.get(projectRoot) || 0) + 1);
877
+ crashStreakFallback.set(projectRoot, count);
878
+ return count;
879
+ } catch (error) {
880
+ const count = (crashStreakFallback.get(projectRoot) || 0) + 1;
881
+ crashStreakFallback.set(projectRoot, count);
882
+ pi?.logger?.warn?.(
883
+ `[UKit] completion crash streak could not persist (${error?.message || error}); `
884
+ + `counting in memory (streak ${count}) so the breaker still advances.`,
885
+ );
886
+ return count;
861
887
  }
862
888
  }
863
889
 
864
890
  async function resetCompletionCrashStreak(projectRoot) {
891
+ crashStreakFallback.delete(projectRoot);
865
892
  try {
866
893
  await fs.promises.rm(path.join(projectRoot, COMPLETION_CRASH_STREAK_FILE), { force: true });
867
894
  } catch { /* best effort */ }
@@ -879,6 +906,38 @@ async function readStopGateMaxCrashStreaks(projectRoot) {
879
906
  }
880
907
  }
881
908
 
909
+ // ── TASK-007 (F-10/OMP-2): stop-path parity with stop-coordinator.mjs ─────────
910
+ // Claude's Stop runs one coordinator with three protections the omp bridge never
911
+ // had: the handoff-cursor lane (a non-done docs/AI_HANDOFF/RUN.md bounces the
912
+ // stop with the run's own Next: line, bounded by the stopGateMaxStalledBlocks
913
+ // liveness breaker), the 2s double-Stop dedupe window (one logical stop burns
914
+ // exactly one continuation), and a 600ms budget on every ledger/state lock.
915
+ // The evaluators are reused from stop-coordinator.mjs itself so semantics and
916
+ // the shared state file (.ukit/storage/cache/stop-coordinator/state.json) stay
917
+ // byte-identical across both engines.
918
+ const STOP_GATE_LOCK_BUDGET_MS = 600;
919
+ const HANDOFF_PROVENANCE_PREFIX = '[ukit-stop-coordinator] handoff-cursor also requested a block: ';
920
+
921
+ let stopCoordinatorModulePromise = null;
922
+
923
+ // Lazy + env-scrubbed: stop-coordinator.mjs arms a process.exit self-deadline at
924
+ // module top level whenever UKIT_HOOK_DEADLINE_MS is set. That env var is only
925
+ // ever set for spawned hook children, but the bridge is a long-lived in-proc
926
+ // module — a leaked value would kill the whole omp host. Same scrub the
927
+ // chain-runner applies around in-proc .mjs steps (hook-chain-runner.mjs:180).
928
+ function loadStopCoordinatorModule() {
929
+ if (!stopCoordinatorModulePromise) {
930
+ const scrubbed = 'UKIT_HOOK_DEADLINE_MS' in process.env
931
+ ? process.env.UKIT_HOOK_DEADLINE_MS : undefined;
932
+ delete process.env.UKIT_HOOK_DEADLINE_MS;
933
+ stopCoordinatorModulePromise = import('../../../.claude/ukit/runtime/stop-coordinator.mjs')
934
+ .finally(() => {
935
+ if (scrubbed !== undefined) process.env.UKIT_HOOK_DEADLINE_MS = scrubbed;
936
+ });
937
+ }
938
+ return stopCoordinatorModulePromise;
939
+ }
940
+
882
941
  export async function runSessionStop(
883
942
  pi,
884
943
  event,
@@ -887,10 +946,69 @@ export async function runSessionStop(
887
946
  context: extensionContext = {},
888
947
  state: suppliedState,
889
948
  ledger: suppliedLedger,
949
+ now = Date.now(),
950
+ dedupeWindowMs,
951
+ lockBudgetMs = STOP_GATE_LOCK_BUDGET_MS,
890
952
  },
891
953
  ) {
892
- const metadata = runtimeMetadata(event, extensionContext);
954
+ // TASK-007 (OMP-2d): the stop path keys its session identity ONLY from the
955
+ // event itself — never the context's sessionManager fallback. omp emits
956
+ // session_id on every top-level session_stop (verified against the bundled
957
+ // binary: emitSessionStop carries session_id/session_file/stop_hook_active),
958
+ // so an event without one is a nested/foreign stop; attributing it to the
959
+ // parent session would burn the parent's continuation budget and dedupe slot.
960
+ const metadata = {
961
+ sessionId: event?.sessionId ?? event?.session_id,
962
+ cwd: extensionContext?.cwd ?? event?.cwd,
963
+ transcriptPath: event?.transcriptPath ?? event?.transcript_path ?? event?.session_file,
964
+ };
893
965
  const payload = buildHookPayload('Stop', metadata);
966
+
967
+ // TASK-007 (OMP-2b): the coordinator's once-per-stop dedupe. omp can deliver
968
+ // two session_stop events for one logical stop; the second inside the window
969
+ // is skipped entirely — it must not burn a second continuation.
970
+ let coordinator = null;
971
+ try {
972
+ coordinator = await loadStopCoordinatorModule();
973
+ } catch (error) {
974
+ // Advisory lane: a missing/unloadable coordinator module must never wedge a
975
+ // session — the completion gate below still owns the stop.
976
+ pi.logger?.warn?.(`[UKit] stop-coordinator module unavailable (dedupe + handoff lanes skipped): ${error?.message || error}`);
977
+ }
978
+ if (coordinator) {
979
+ const sessionKey = typeof payload.session_id === 'string' && payload.session_id
980
+ ? payload.session_id
981
+ : (typeof payload.transcript_path === 'string' && payload.transcript_path ? payload.transcript_path : 'no-session');
982
+ const duplicate = await coordinator.alreadyCoordinatedThisStop({
983
+ projectRoot,
984
+ sessionKey,
985
+ now,
986
+ dedupeWindowMs: dedupeWindowMs ?? coordinator.DEDUPE_WINDOW_MS,
987
+ lockBudgetMs,
988
+ });
989
+ if (duplicate) return undefined;
990
+ }
991
+
992
+ // TASK-007 (OMP-2c): the handoff-cursor lane. While docs/AI_HANDOFF/RUN.md
993
+ // reports a phase outside {done, blocked} the run owns this stop — the
994
+ // cursor's Next: line is the continuation instruction. Advisory on failure,
995
+ // bounded by the stopGateMaxStalledBlocks breaker (always advances post-S3).
996
+ let handoff = null;
997
+ if (coordinator) {
998
+ try {
999
+ handoff = await coordinator.evaluateHandoffCursor({ projectRoot, now, lockBudgetMs });
1000
+ } catch (error) {
1001
+ pi.logger?.warn?.(`[UKit] handoff-cursor evaluator failed (advisory lane): ${error?.message || error}`);
1002
+ }
1003
+ }
1004
+ const handoffBlock = handoff?.kind === 'block' ? handoff.reason : null;
1005
+ // A breaker-release advisory (or any handoff systemMessage) is the omp
1006
+ // equivalent of the coordinator's systemMessage channel — surface it once,
1007
+ // visibly, whichever way the stop itself resolves.
1008
+ if (handoff?.kind === 'advisory' && handoff.systemMessage) {
1009
+ sendContext(pi, [handoff.systemMessage], 'nextTurn', { display: true });
1010
+ }
1011
+
894
1012
  const state = suppliedState ?? await readRouteState(projectRoot, payload);
895
1013
  const ledger = suppliedLedger ?? await readExecutionLedger(projectRoot, payload) ?? {};
896
1014
  let evaluation;
@@ -899,8 +1017,7 @@ export async function runSessionStop(
899
1017
  } catch (error) {
900
1018
  // BUG-C22-05: fail-closed like the shell gate (continue the turn), but only up
901
1019
  // to the configured consecutive-crash cap — a persistently broken evaluator
902
- // must release loudly instead of looping every Stop forever.
903
- const streak = await bumpCompletionCrashStreak(projectRoot);
1020
+ const streak = await bumpCompletionCrashStreak(projectRoot, pi);
904
1021
  const maxStreak = await readStopGateMaxCrashStreaks(projectRoot);
905
1022
  if (streak > maxStreak) {
906
1023
  await resetCompletionCrashStreak(projectRoot);
@@ -917,7 +1034,8 @@ export async function runSessionStop(
917
1034
  continue: true,
918
1035
  additionalContext:
919
1036
  `UKit stop coordinator: infrastructure failure while evaluating this stop (crash streak ${streak}/${maxStreak}) — `
920
- + 'the stop is blocked fail-closed (details withheld). Run: ukit install, then re-send the task in a new message.',
1037
+ + 'the stop is blocked fail-closed (details withheld). Run: ukit install, then re-send the task in a new message.'
1038
+ + (handoffBlock ? `\n${HANDOFF_PROVENANCE_PREFIX}${handoffBlock}` : ''),
921
1039
  };
922
1040
  }
923
1041
  // A clean evaluation resets the consecutive-crash streak (shared with the shell hook).
@@ -940,20 +1058,33 @@ export async function runSessionStop(
940
1058
  // message) stays as a second guaranteed channel.
941
1059
  sendContext(pi, [notice], 'nextTurn', { display: true });
942
1060
  }
1061
+ // The completion gate released, but an in-flight handoff run still owns the
1062
+ // stop (coordinator merge order: handoff-cursor block outranks a released
1063
+ // completion). Its own stall streak — not the ledger continuation counter —
1064
+ // bounds this lane, so no continuation bookkeeping runs here.
1065
+ if (handoffBlock) {
1066
+ return { continue: true, additionalContext: handoffBlock };
1067
+ }
943
1068
  return undefined;
944
1069
  }
945
1070
 
946
1071
  if (suppliedLedger === undefined) {
947
1072
  try {
948
- if (evaluation.finalNotice) await markNotified(projectRoot, payload, ledger);
949
- else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null, evidencePromptKey(state));
1073
+ // TASK-007 (OMP-2a): the 600ms lock budget — a contended ledger lock must
1074
+ // never hold the omp host near its hook deadline.
1075
+ if (evaluation.finalNotice) await markNotified(projectRoot, payload, ledger, { deadlineMs: lockBudgetMs });
1076
+ else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null, evidencePromptKey(state), { deadlineMs: lockBudgetMs });
950
1077
  } catch (error) {
951
1078
  pi.logger?.warn?.(`[UKit] continuation bookkeeping failed open: ${error?.message || error}`);
952
1079
  }
953
1080
  }
954
1081
  return {
955
1082
  continue: true,
956
- additionalContext: evaluation.reason,
1083
+ // Coordinator merge order: the completion gate owns the continuation, and a
1084
+ // losing handoff-cursor block travels inside the winning reason.
1085
+ additionalContext: handoffBlock
1086
+ ? `${evaluation.reason}\n${HANDOFF_PROVENANCE_PREFIX}${handoffBlock}`
1087
+ : evaluation.reason,
957
1088
  };
958
1089
  }
959
1090
 
@@ -4,7 +4,7 @@
4
4
  Status values (xem RULES.md §Status state machine):
5
5
  ready | in_progress | pending_review | changes_requested | critical_block | approved | approved_minor | blocked | done
6
6
 
7
- Owner = tool đang giữ task: claude-code | kilo-code | codex | -
7
+ Owner = tool đang giữ task: claude-code | codex | omp | -
8
8
  -->
9
9
 
10
10
  | ID | Title | Priority | Size | Status | Owner | Reviewer | File |
@@ -92,21 +92,21 @@ Next: <bước kế tiếp chính xác>
92
92
 
93
93
  ## Handoff Flow (tool-agnostic, file-based state machine)
94
94
 
95
- UKit handoff hoạt động qua **file state**. Anh tự chọn tool nào cho từng phase — Claude Code / Kilo Code / Codex / tool mới sau này — đều được. UKit chỉ care về **role của model**, không care tool.
95
+ UKit handoff hoạt động qua **file state**. Anh tự chọn tool nào cho từng phase — Claude Code / Codex / omp / tool mới sau này — đều được. UKit chỉ care về **role của model**, không care tool.
96
96
 
97
97
  3 phase × 3 role model:
98
98
 
99
99
  - **Plan** — model mạnh nhất anh có (reasoning model). Có thể chạy ở bất kỳ tool nào hỗ trợ planning tốt.
100
- - **Execute** — model rẻ-mà-vẫn-thông-minh (code model). Có thể là subagent code của Kilo, hay feature-implementer của Claude Code.
101
- - **Review** — **MODEL KHÁC executor** (reasoning model thường tốt hơn). Có thể là tool khác, hoặc cùng tool nhưng subagent khác model (ví dụ Kilo có subagent code và subagent review riêng).
100
+ - **Execute** — model rẻ-mà-vẫn-thông-minh (code model). Có thể là subagent code của omp, hay feature-implementer của Claude Code.
101
+ - **Review** — **MODEL KHÁC executor** (reasoning model thường tốt hơn). Có thể là tool khác, hoặc cùng tool nhưng subagent khác model.
102
102
 
103
103
  Hai mô hình triển khai đều hợp lệ:
104
- - **Cross-tool**: ví dụ Claude (plan) → Kilo (execute) → Claude (review). Bridge qua file.
105
- - **Same-tool different-subagent**: ví dụ Kilo:plan → Kilo:code → Kilo:review, miễn 3 subagent dùng MODEL khác nhau ở role tương ứng.
104
+ - **Cross-tool**: ví dụ Claude (plan) → Codex (execute) → Claude (review). Bridge qua file.
105
+ - **Same-tool different-subagent**: ví dụ omp:plan → omp:task → omp:code-reviewer, miễn 3 subagent dùng MODEL khác nhau ở role tương ứng.
106
106
 
107
107
  Mỗi tool/subagent đọc cùng `INDEX.md` + `tasks/TASK-xxx.md` → chọn task theo `status` → cập nhật status khi xong.
108
108
 
109
- > **Quan trọng — UKit không enforce model:** `handoff.executor.cheapSmartModelHint` và `handoff.reviewer.model` trong `.ukit/storage/config.json` chỉ là **nhãn** để anh biết MUỐN dùng gì. Tool nào dùng model nào là do anh chọn trong settings của tool đó. UKit enforce contract bằng cách bắt executor TỰ KHAI `EXECUTOR_MODEL` trong Executor Report; reviewer so với chính nó và refuse nếu trùng. Vì vậy nếu trong Kilo anh để cả code-subagent và review-subagent đều dùng cùng model → reviewer sẽ tự refuse, không silent-pass.
109
+ > **Quan trọng — UKit không enforce model:** `handoff.executor.cheapSmartModelHint` và `handoff.reviewer.model` trong `.ukit/storage/config.json` chỉ là **nhãn** để anh biết MUỐN dùng gì. Tool nào dùng model nào là do anh chọn trong settings của tool đó. UKit enforce contract bằng cách bắt executor TỰ KHAI `EXECUTOR_MODEL` trong Executor Report; reviewer so với chính nó và refuse nếu trùng. Vì vậy nếu anh để cả executor-subagent và review-subagent đều dùng cùng model → reviewer sẽ tự refuse, không silent-pass.
110
110
 
111
111
  ### Status state machine
112
112
 
@@ -442,7 +442,7 @@
442
442
  "field": "handoff.reviewer.model",
443
443
  "mac_dinh": "unic-smart",
444
444
  "y_nghia": "Model dùng cho reviewer agent ở Phase 3. BẮT BUỘC khác model executor để bắt được lỗi mà executor miss. Có thể dùng claude-opus-5, unic-smart, hoặc bất kỳ model reasoning mạnh nào.",
445
- "vi_du": "Nếu executor là unic-code (Kilo Code), set reviewer.model=unic-smart hoặc claude-opus-5. Nếu executor là claude-sonnet, set reviewer thành claude-opus."
445
+ "vi_du": "Nếu executor là unic-code, set reviewer.model=unic-smart hoặc claude-opus-5. Nếu executor là claude-sonnet, set reviewer thành claude-opus."
446
446
  },
447
447
  "tat_reviewer_phase": {
448
448
  "field": "handoff.reviewer.enabled",
@@ -601,7 +601,7 @@
601
601
  },
602
602
  "handoff": {
603
603
  "enabled": "Bật Quality Gate cho handoff: plan có Test Plan, executor test-first, reviewer model khác. Tắt = quay về flow cũ (dễ lọt lỗi vặt).",
604
- "crossTool": "true nghĩa là handoff truyền qua file (PLAN/INDEX/tasks) chứ không qua in-process subagent — cho phép plan ở Claude Code, execute ở Kilo Code, review ở Claude Code khác model.",
604
+ "crossTool": "true nghĩa là handoff truyền qua file (PLAN/INDEX/tasks) chứ không qua in-process subagent — cho phép plan ở một tool, execute ở tool khác, review ở tool thứ ba khác model.",
605
605
  "maxParallelAgents": "Số agent chạy song song TỐI ĐA trong một wave (mặc định 2 để giữ session chính nhẹ và hạn chế rủi ro compact/worktree rác; nếu máy khỏe và task độc lập nhiều thì có thể nâng dần 3-5, tối đa ~10-15). Một wave có nhiều task hơn số này sẽ được chia thành nhiều batch chạy lần lượt — áp dụng cho cả Phase 3 Implement và Phase 4 Review. Lý do giới hạn vẫn còn: mỗi agent nền có context window riêng, và report của agent khi xong sẽ được inject ngược vào session chính — chạy quá nhiều cùng lúc (vd 20+) vẫn có thể làm session chính vượt context window và bỏ lại worktree rác. Hạ xuống 1-2 nếu task nặng (verification output dài) hoặc thấy compact bị trigger liên tục.",
606
606
  "plan": {
607
607
  "requireTestPlan": "Bắt buộc PLAN.md §4 phải có Test Plan trước khi task chuyển ready.",