@ngockhoale/ukit 2.3.9 → 2.3.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -13,6 +13,21 @@ const RESUME_INTENT_TTL_MS = 30 * 60 * 1000;
13
13
  const MAX_RECEIPTS = 24;
14
14
  const MAX_SOURCE_FILES = 16;
15
15
  const MAX_CONTINUATIONS = 6;
16
+ // Two adjacent identical failed verifications (no edit attempt between them) are a loop;
17
+ // vibecode routes — which bypass the cap and every reentrant valve — additionally get a
18
+ // cumulative escape after the same verification failed this many times across edits.
19
+ const VERIFICATION_LOOP_STREAK = 2;
20
+ const VIBECODE_VERIFICATION_FAILURE_LIMIT = 4;
21
+ const MAX_TRACKED_VERIFICATIONS = 8;
22
+ // hook-chain-runner.mjs gives every hook script a 4s child budget, and omp blocks Edit|Write
23
+ // when a chain script is killed. Waiting longer than that for a contended lock got the
24
+ // recording hook killed mid-chain (receipt lost, edit blocked). Give up and fail open well
25
+ // inside the budget instead — under extreme contention a receipt may be lost, but the hook
26
+ // is never killed by its own chain.
27
+ const HOOK_SAFE_LOCK_WAIT_MS = 2500;
28
+ function withLedgerLock(target, fn) {
29
+ return withFileLock(target, fn, { maxWaitMs: HOOK_SAFE_LOCK_WAIT_MS });
30
+ }
16
31
  const IMPLEMENT_MODES = new Set([
17
32
  'tiny-fix',
18
33
  'local-fix',
@@ -86,9 +101,14 @@ async function readJson(filePath, fallback = null) {
86
101
  }
87
102
  }
88
103
 
104
+ // Parallel writers in one process (subagent Stop hooks, concurrent receipts) used to share
105
+ // a single `pid`-suffixed temp path: one rename removed the file under the other and the
106
+ // writer crashed with ENOENT mid-hook. Make every write's temp path unique.
107
+ let atomicWriteCounter = 0;
89
108
  async function writeJsonAtomic(filePath, value) {
90
109
  await fs.mkdir(path.dirname(filePath), { recursive: true });
91
- const tempPath = `${filePath}.${process.pid}.tmp`;
110
+ atomicWriteCounter += 1;
111
+ const tempPath = `${filePath}.${process.pid}-${atomicWriteCounter}-${Math.random().toString(16).slice(2)}.tmp`;
92
112
  await fs.writeFile(tempPath, `${JSON.stringify(value, null, 2)}\n`, 'utf8');
93
113
  await fs.rename(tempPath, filePath);
94
114
  }
@@ -276,8 +296,129 @@ function fileMatchesExpected(receiptFile, expectedFile) {
276
296
  return false;
277
297
  }
278
298
 
299
+ // Piping a verification command into a read-only consumer (`yarn test 2>&1 | tail -6`) is
300
+ // the standard idiom for trimming output. The FIRST pipe segment is still the planned
301
+ // verification, so it may mint targeted evidence — but only when every later segment is a
302
+ // benign text consumer. Anything that can reinterpret or execute output (sh, xargs, tee)
303
+ // keeps the receipt broad.
304
+ const BENIGN_OUTPUT_CONSUMERS = new Set([
305
+ 'tail', 'head', 'grep', 'egrep', 'fgrep', 'rg', 'ag', 'sed', 'cat',
306
+ 'awk', 'cut', 'wc', 'sort', 'uniq', 'tr', 'nl',
307
+ ]);
308
+
309
+ function isBenignOutputConsumer(segment) {
310
+ const firstWord = String(segment || '').trim().split(/\s+/)[0] || '';
311
+ return BENIGN_OUTPUT_CONSUMERS.has(firstWord);
312
+ }
313
+
314
+ function terminalShellCommandUnit(command) {
315
+ const text = String(command || '').trim();
316
+ if (!text) return null;
317
+
318
+ let quote = null;
319
+ let escaped = false;
320
+ let pipeBuffer = '';
321
+ let currentPipes = [];
322
+ const segments = [];
323
+
324
+ const closePipe = () => {
325
+ const trimmed = pipeBuffer.trim();
326
+ if (!trimmed) return null;
327
+ currentPipes.push(trimmed);
328
+ pipeBuffer = '';
329
+ return trimmed;
330
+ };
331
+ const closeSegment = () => {
332
+ if (!closePipe()) return false;
333
+ segments.push(currentPipes);
334
+ currentPipes = [];
335
+ return true;
336
+ };
337
+ // Redirections (`> file`, `2>&1`, `>> x`, `&> y`, `< in`) modify the current command's
338
+ // streams; they are consumed so the fd digits/paths do not corrupt the unit text.
339
+ const skipRedirectionTarget = (index) => {
340
+ let next = index;
341
+ while (next < text.length && /\s/.test(text[next])) next += 1;
342
+ while (
343
+ next < text.length
344
+ && !/\s/.test(text[next])
345
+ && !'&|;()<>`\n\r'.includes(text[next])
346
+ ) next += 1;
347
+ return next;
348
+ };
349
+
350
+ for (let index = 0; index < text.length; index += 1) {
351
+ const char = text[index];
352
+ if (escaped) {
353
+ escaped = false;
354
+ pipeBuffer += char;
355
+ continue;
356
+ }
357
+ if (char === '\\') {
358
+ escaped = true;
359
+ pipeBuffer += char;
360
+ continue;
361
+ }
362
+ if (quote) {
363
+ if (char === quote) quote = null;
364
+ pipeBuffer += char;
365
+ continue;
366
+ }
367
+ if (char === "'" || char === '"') {
368
+ quote = char;
369
+ pipeBuffer += char;
370
+ continue;
371
+ }
372
+
373
+ // This deliberately recognizes only simple top-level `&&` chains and pipes into
374
+ // benign consumers. Other shell control/grouping syntax can mask a terminal
375
+ // command's status, so it must remain broad rather than targeted evidence.
376
+ if (char === ';' || char === '`' || char === '(' || char === ')' || char === '\n' || char === '\r') {
377
+ return null;
378
+ }
379
+ if (char === '$' && text[index + 1] === '(') return null;
380
+ if (char === '>' || char === '<') {
381
+ let next = index + 1;
382
+ if (text[next] === '>' && char === '>') next += 1;
383
+ if (text[next] === '&') next += 1;
384
+ index = skipRedirectionTarget(next) - 1;
385
+ continue;
386
+ }
387
+ if (char === '&' && text[index + 1] === '>') {
388
+ let next = index + 2;
389
+ if (text[next] === '>') next += 1;
390
+ index = skipRedirectionTarget(next) - 1;
391
+ continue;
392
+ }
393
+ if (char === '|') {
394
+ if (text[index + 1] === '|') return null;
395
+ if (!closePipe()) return null;
396
+ continue;
397
+ }
398
+ if (char === '&') {
399
+ if (text[index + 1] !== '&') return null;
400
+ if (!closeSegment()) return null;
401
+ index += 1;
402
+ continue;
403
+ }
404
+ pipeBuffer += char;
405
+ }
406
+
407
+ if (quote || escaped) return null;
408
+ if (!closeSegment()) return null;
409
+
410
+ const terminalSegment = segments[segments.length - 1];
411
+ if (terminalSegment.length > 1) {
412
+ const consumers = terminalSegment.slice(1);
413
+ if (!consumers.every((consumer) => isBenignOutputConsumer(consumer))) {
414
+ return null;
415
+ }
416
+ }
417
+ return terminalSegment[0];
418
+ }
419
+
279
420
  function matchesRoutedCommand(command, routedCommand) {
280
- const receipt = String(command || '').trim();
421
+ const receipt = terminalShellCommandUnit(command);
281
422
  const routed = String(routedCommand || '').trim();
282
423
  if (!receipt || !routed) return false;
283
424
  return receipt === routed || receipt.startsWith(routed) || routed.startsWith(receipt);
@@ -317,6 +458,17 @@ function carriedEvidenceLedger(fresh, current) {
317
458
  verificationSucceeded: fresh.verificationSucceeded || current.verificationSucceeded === true,
318
459
  verificationFailed: fresh.verificationFailed || current.verificationFailed === true,
319
460
  receipts: [...(current.receipts || [])].slice(-MAX_RECEIPTS),
461
+ // The continuation budget belongs to the same logical request (promptKey), so a re-key
462
+ // must keep counting toward the cap. Resetting it here made the cap unreachable and the
463
+ // Stop gate loop forever on harnesses without a stop_hook_active valve.
464
+ continuationCount: Math.max(Number(fresh.continuationCount || 0), Number(current.continuationCount || 0)),
465
+ continuationRequestKey: current.continuationRequestKey || fresh.continuationRequestKey || null,
466
+ notified: fresh.notified === true || current.notified === true,
467
+ // Verification-loop tracking and any minted blocker belong to the same logical request
468
+ // too — dropping a blocker on re-key would resume the exact loop it recorded.
469
+ failedVerificationStreak: current.failedVerificationStreak || null,
470
+ verificationFailureCounts: current.verificationFailureCounts || {},
471
+ blocker: current.blocker || null,
320
472
  };
321
473
  }
322
474
 
@@ -331,6 +483,7 @@ function evidencePromptKey(routeState) {
331
483
  if (!promptText) return null;
332
484
  return `prompt-${crypto.createHash('sha256').update(promptText).digest('hex').slice(0, 20)}`;
333
485
  }
486
+ export { evidencePromptKey };
334
487
 
335
488
  function freshLedger(payload, routeState, harness) {
336
489
  return {
@@ -364,61 +517,123 @@ export async function recordExecutionReceipt({
364
517
  harness = 'unknown',
365
518
  } = {}) {
366
519
  if (!projectRoot || !toolName) return null;
367
- const routeState = await readRouteState(projectRoot, payload);
368
- const current = await readExecutionLedger(projectRoot, payload);
369
- const ledger = !current || current.requestKey !== (routeState?.requestKey || null)
370
- ? carriedEvidenceLedger(freshLedger(payload, routeState, harness), current)
371
- : { ...current, harness: current.harness || harness };
372
-
373
- const failed = explicitError(payload);
374
- const exitCode = extractExitCode(payload);
375
- const success = !failed && (exitCode === null || exitCode === 0);
376
- const toolInput = payload.tool_input || {};
377
- const receipt = {
378
- ts: Date.now(),
379
- toolName,
380
- toolUseId: payload.tool_use_id || null,
381
- success,
382
- exitCode,
383
- };
520
+ // The whole read-modify-write is locked: parallel subagents record receipts through the
521
+ // same session ledger file, and an unlocked snapshot rewrite dropped whichever receipts
522
+ // landed between the read and the write.
523
+ return withLedgerLock(ledgerPath(projectRoot, payload), async () => {
524
+ const routeState = await readRouteState(projectRoot, payload);
525
+ const current = await readExecutionLedger(projectRoot, payload);
526
+ const nextRequestKey = routeState?.requestKey || null;
527
+ // A missing/foreign route state must never blank the evidence this session already
528
+ // banked. The shared state slot is single-owner (a parallel subagent re-stamps it), so
529
+ // routeState === null is routine mid-request — rebuilding from fresh there re-demanded
530
+ // every evidence the request had produced and froze the run. Without a live requestKey
531
+ // the only safe move is to keep appending to the current ledger.
532
+ const ledger = current && !nextRequestKey
533
+ ? { ...current, harness: current.harness || harness }
534
+ : (!current || current.requestKey !== nextRequestKey
535
+ ? carriedEvidenceLedger(freshLedger(payload, routeState, harness), current)
536
+ : { ...current, harness: current.harness || harness });
537
+
538
+ const failed = explicitError(payload);
539
+ const exitCode = extractExitCode(payload);
540
+ const success = !failed && (exitCode === null || exitCode === 0);
541
+ const toolInput = payload.tool_input || {};
542
+ const receipt = {
543
+ ts: Date.now(),
544
+ toolName,
545
+ toolUseId: payload.tool_use_id || null,
546
+ success,
547
+ exitCode,
548
+ };
384
549
 
385
- if (toolName === 'Read' || toolName === 'Grep' || toolName === 'Glob') {
386
- receipt.kind = 'source';
387
- receipt.file = toolInput.file_path || toolInput.path || null;
388
- ledger.sourceSucceeded ||= success;
389
- if (success && receipt.file) {
390
- ledger.sourceFiles = [...new Set([...(ledger.sourceFiles || []), receipt.file])].slice(-MAX_SOURCE_FILES);
391
- }
392
- } else if (toolName === 'Edit' || toolName === 'Write') {
393
- receipt.kind = 'write';
394
- receipt.file = toolInput.file_path || toolInput.path || toolInput.paths?.[0] || null;
395
- ledger.writeAttempted = true;
396
- ledger.writeSucceeded ||= success;
397
- } else if (toolName === 'Bash' && isVerificationCommand(toolInput.command)) {
398
- receipt.kind = 'verification';
399
- receipt.command = String(toolInput.command || '').trim();
400
- // WS-C routed-verification receipt: a command counts as "targeted" when it matches the
401
- // routed plan (preferredOrder / primaryCommands). Off-plan verification still records
402
- // as broad — counted only when the route carried no commands at all.
403
- const routedCommands = routedVerificationCommands(routeState?.routeSummary);
404
- if (routedCommands.length > 0) {
405
- const matched = routedCommands.find((cmd) => matchesRoutedCommand(receipt.command, cmd));
406
- receipt.scope = matched ? 'targeted' : 'broad';
407
- if (receipt.scope === 'targeted') {
408
- ledger.targetedVerificationSucceeded ||= success;
550
+ if (toolName === 'Read' || toolName === 'Grep' || toolName === 'Glob') {
551
+ receipt.kind = 'source';
552
+ receipt.file = toolInput.file_path || toolInput.path || null;
553
+ ledger.sourceSucceeded ||= success;
554
+ if (success && receipt.file) {
555
+ ledger.sourceFiles = [...new Set([...(ledger.sourceFiles || []), receipt.file])].slice(-MAX_SOURCE_FILES);
556
+ }
557
+ } else if (toolName === 'Edit' || toolName === 'Write') {
558
+ receipt.kind = 'write';
559
+ receipt.file = toolInput.file_path || toolInput.path || toolInput.paths?.[0] || null;
560
+ ledger.writeAttempted = true;
561
+ ledger.writeSucceeded ||= success;
562
+ // A mutation attempt between failures breaks the "no change in between" loop shape.
563
+ if (ledger.failedVerificationStreak) ledger.failedVerificationStreak = null;
564
+ } else if (toolName === 'Bash' && isVerificationCommand(toolInput.command)) {
565
+ receipt.kind = 'verification';
566
+ receipt.command = String(toolInput.command || '').trim();
567
+ // WS-C routed-verification receipt: a command counts as "targeted" when it matches the
568
+ // routed plan (preferredOrder / primaryCommands). Off-plan verification still records
569
+ // as broad — counted only when the route carried no commands at all.
570
+ const routedCommands = routedVerificationCommands(routeState?.routeSummary);
571
+ if (routedCommands.length > 0) {
572
+ const matched = routedCommands.find((cmd) => matchesRoutedCommand(receipt.command, cmd));
573
+ receipt.scope = matched ? 'targeted' : 'broad';
574
+ if (receipt.scope === 'targeted') {
575
+ ledger.targetedVerificationSucceeded ||= success;
576
+ }
409
577
  }
578
+ ledger.verificationAttempted = true;
579
+ ledger.verificationSucceeded ||= success;
580
+ ledger.verificationFailed ||= !success;
581
+
582
+ // Verification-loop tracking. `terminalShellCommandUnit` gives the loop identity:
583
+ // the same failing check rerun — `setup && yarn test` and `yarn test 2>&1 | tail`
584
+ // both count as the same command, while a different check resets the streak.
585
+ const fingerprint = terminalShellCommandUnit(receipt.command) || receipt.command;
586
+ const counts = { ...(ledger.verificationFailureCounts || {}) };
587
+ if (success) {
588
+ delete counts[fingerprint];
589
+ ledger.verificationFailureCounts = counts;
590
+ if (ledger.failedVerificationStreak?.fingerprint === fingerprint) {
591
+ ledger.failedVerificationStreak = null;
592
+ }
593
+ } else {
594
+ const streakFingerprint = ledger.failedVerificationStreak?.fingerprint;
595
+ const streakCount = streakFingerprint === fingerprint
596
+ ? Number(ledger.failedVerificationStreak.count || 0) + 1
597
+ : 1;
598
+ ledger.failedVerificationStreak = { fingerprint, count: streakCount };
599
+ counts[fingerprint] = Number(counts[fingerprint] || 0) + 1;
600
+ const trackedKeys = Object.keys(counts);
601
+ if (trackedKeys.length > MAX_TRACKED_VERIFICATIONS) {
602
+ for (const key of trackedKeys.slice(0, trackedKeys.length - MAX_TRACKED_VERIFICATIONS)) {
603
+ delete counts[key];
604
+ }
605
+ }
606
+ ledger.verificationFailureCounts = counts;
607
+
608
+ const vibecode = routeState?.routeSummary?.autonomyLevel === 'vibecode';
609
+ const shortCommand = fingerprint.length > 80 ? `${fingerprint.slice(0, 77)}…` : fingerprint;
610
+ if (streakCount >= VERIFICATION_LOOP_STREAK) {
611
+ ledger.blocker = {
612
+ kind: 'verification-loop',
613
+ visible: true,
614
+ detail: `verification "${shortCommand}" failed ${streakCount} times in a row with no edit in between — this is a loop, not progress. Report the failing output as a concrete blocker instead of rerunning it.`,
615
+ command: fingerprint,
616
+ ts: Date.now(),
617
+ };
618
+ } else if (vibecode && counts[fingerprint] >= VIBECODE_VERIFICATION_FAILURE_LIMIT) {
619
+ ledger.blocker = {
620
+ kind: 'verification-loop',
621
+ visible: true,
622
+ detail: `verification "${shortCommand}" failed ${counts[fingerprint]} times in continuous (vibecode) execution — report the failing output as a concrete blocker instead of retrying.`,
623
+ command: fingerprint,
624
+ ts: Date.now(),
625
+ };
626
+ }
627
+ }
628
+ } else {
629
+ return ledger;
410
630
  }
411
- ledger.verificationAttempted = true;
412
- ledger.verificationSucceeded ||= success;
413
- ledger.verificationFailed ||= !success;
414
- } else {
415
- return ledger;
416
- }
417
631
 
418
- ledger.receipts = appendReceipt(ledger.receipts, receipt);
419
- ledger.updatedAt = Date.now();
420
- await writeJsonAtomic(ledgerPath(projectRoot, payload), ledger);
421
- return ledger;
632
+ ledger.receipts = appendReceipt(ledger.receipts, receipt);
633
+ ledger.updatedAt = Date.now();
634
+ await writeJsonAtomic(ledgerPath(projectRoot, payload), ledger);
635
+ return ledger;
636
+ });
422
637
  }
423
638
 
424
639
  function requiredEvidence(state = {}) {
@@ -500,13 +715,30 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
500
715
  const mode = routeSummary.executionMode || routeSummary.approachSelector?.executionMode || null;
501
716
  const evidence = requiredEvidence(state);
502
717
  if (ledger?.blocker) {
718
+ // A blocker minted by the gate itself (verification loop) is the run's exit path — it
719
+ // must be user-visible or the release is indistinguishable from a stall. A blocker
720
+ // minted by a permission decision is already named in the model's own reply and stays
721
+ // silent here.
722
+ if (ledger.blocker.visible === true) {
723
+ return {
724
+ continue: false,
725
+ notify: true,
726
+ missingEvidence: [],
727
+ reason: `UKit stopped automatic recovery: ${ledger.blocker.detail || 'a recorded blocker ended this run.'}`,
728
+ };
729
+ }
503
730
  return { continue: false, notify: false, missingEvidence: [] };
504
731
  }
505
732
  if (evidence.length === 0) {
506
733
  return { continue: false, notify: false, missingEvidence: [] };
507
734
  }
508
735
 
509
- const sameRequest = !ledger?.requestKey || !state?.requestKey || ledger.requestKey === state.requestKey;
736
+ // Request identity is the prompt (see evidencePromptKey): the router re-keys requestKey on
737
+ // every Edit/Write, so a requestKey mismatch alone must not discard the evidence — and the
738
+ // continuation budget — the request already accumulated. A different prompt keeps a clean slate.
739
+ const sameRequest = !ledger?.requestKey || !state?.requestKey
740
+ || ledger.requestKey === state.requestKey
741
+ || (ledger.promptKey && evidencePromptKey(state) === ledger.promptKey);
510
742
  const effectiveLedger = sameRequest ? ledger : {};
511
743
  const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger, routeSummary));
512
744
  if (missingEvidence.length === 0) {
@@ -535,6 +767,28 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
535
767
  };
536
768
  }
537
769
 
770
+ // Same reasoning as find-cause, for the analysis-shaped end of map-impact: an impact map
771
+ // whose honest outcome is "analysis complete, no change warranted" must be able to end
772
+ // visibly. The contract's completionRule only demands impact evidence BEFORE an edit claim
773
+ // — it does not make a mutation mandatory — but the gate treated missing write evidence as
774
+ // an order to edit, and an analysis-only request looped Stop → "make an Edit" → Stop
775
+ // forever. The valve opens only when the analysis half actually finished (impact evidence
776
+ // satisfied), no mutation was attempted, and no verification failed; anything less has
777
+ // concrete unfinished work and keeps recovering.
778
+ if (
779
+ mode === 'map-impact'
780
+ && !missingEvidence.includes('impact-evidence')
781
+ && effectiveLedger.writeAttempted !== true
782
+ && effectiveLedger.verificationFailed !== true
783
+ ) {
784
+ return {
785
+ continue: false,
786
+ notify: true,
787
+ missingEvidence,
788
+ reason: 'UKit impact analysis ended without a mutation. A completed impact map with no warranted change is valid; report the findings and whether a follow-up edit is needed. Do not claim any fix without write and verification evidence.',
789
+ };
790
+ }
791
+
538
792
  const gated = IMPLEMENT_MODES.has(mode);
539
793
  if (!gated) {
540
794
  return {
@@ -547,10 +801,12 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
547
801
 
548
802
  // Continuation attempts only make sense within the request that minted them: a new
549
803
  // routed request in the same session must start with a fresh budget, otherwise a cap
550
- // exhausted on task A suppresses recovery for task B.
804
+ // exhausted on task A suppresses recovery for task B. A requestKey mismatch within the
805
+ // same promptKey is just the router re-keying mid-request — not a new request.
551
806
  const staleContinuations = effectiveLedger?.continuationRequestKey
552
807
  && state?.requestKey
553
- && effectiveLedger.continuationRequestKey !== state.requestKey;
808
+ && effectiveLedger.continuationRequestKey !== state.requestKey
809
+ && !(effectiveLedger.promptKey && evidencePromptKey(state) === effectiveLedger.promptKey);
554
810
  const continuationCount = staleContinuations
555
811
  ? 0
556
812
  : Number(effectiveLedger?.continuationCount || 0);
@@ -588,19 +844,21 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
588
844
  };
589
845
  }
590
846
 
591
- export async function incrementContinuation(projectRoot, payload = {}, ledger = null, requestKey = null) {
847
+ export async function incrementContinuation(projectRoot, payload = {}, ledger = null, requestKey = null, promptKey = null) {
592
848
  const target = ledgerPath(projectRoot, payload);
593
849
  // Locked re-read: parallel subagents firing Stop hooks share one ledger file, and a
594
850
  // stale-snapshot rewrite would reset each other's continuationCount — the budget would
595
851
  // never advance and the gate would keep issuing continuations.
596
- return withFileLock(target, async () => {
852
+ return withLedgerLock(target, async () => {
597
853
  const current = await readExecutionLedger(projectRoot, payload) || ledger || freshLedger(payload, null, 'unknown');
598
854
  // Mirror evaluateCompletion's staleness rule: a count minted by an earlier request must
599
855
  // not be carried into the new request's budget, or the cap fires early (evaluate says 0,
600
- // persist says 7) and the next request inherits a nearly exhausted budget.
856
+ // persist says 7) and the next request inherits a nearly exhausted budget. Same-prompt
857
+ // re-keys are the same request, so their budget must keep counting toward the cap.
601
858
  const stale = requestKey
602
859
  && current?.continuationRequestKey
603
- && current.continuationRequestKey !== requestKey;
860
+ && current.continuationRequestKey !== requestKey
861
+ && !(promptKey && current.promptKey && promptKey === current.promptKey);
604
862
  const next = {
605
863
  ...current,
606
864
  continuationCount: (stale ? 0 : Number(current.continuationCount || 0)) + 1,
@@ -615,7 +873,7 @@ export async function incrementContinuation(projectRoot, payload = {}, ledger =
615
873
 
616
874
  export async function markNotified(projectRoot, payload = {}, ledger = null) {
617
875
  const target = ledgerPath(projectRoot, payload);
618
- return withFileLock(target, async () => {
876
+ return withLedgerLock(target, async () => {
619
877
  const current = await readExecutionLedger(projectRoot, payload) || ledger || freshLedger(payload, null, 'unknown');
620
878
  const next = { ...current, notified: true, updatedAt: Date.now() };
621
879
  await writeJsonAtomic(target, next);
@@ -623,6 +881,25 @@ export async function markNotified(projectRoot, payload = {}, ledger = null) {
623
881
  });
624
882
  }
625
883
 
884
+ // omp's session_stop carries no stop_hook_active marker, so reentrancy must be detected from
885
+ // the ledger itself: a stop whose recovery turn produced no new receipts is the reentrant
886
+ // shape Claude Code flags natively. The fingerprint deliberately excludes continuation
887
+ // bookkeeping fields (count/notified/updatedAt) so this call's own writes stay invisible to
888
+ // the next comparison — only real receipts change it.
889
+ export async function noteStopProgress(projectRoot, payload = {}) {
890
+ const target = ledgerPath(projectRoot, payload);
891
+ return withLedgerLock(target, async () => {
892
+ const current = await readExecutionLedger(projectRoot, payload);
893
+ if (!current) return { reentrant: false };
894
+ const receipts = Array.isArray(current.receipts) ? current.receipts : [];
895
+ const fingerprint = `${receipts.length}:${receipts[receipts.length - 1]?.ts || 0}`;
896
+ const reentrant = typeof current.stopProgressFingerprint === 'string'
897
+ && current.stopProgressFingerprint === fingerprint;
898
+ await writeJsonAtomic(target, { ...current, stopProgressFingerprint: fingerprint, updatedAt: Date.now() });
899
+ return { reentrant };
900
+ });
901
+ }
902
+
626
903
  async function readStdin() {
627
904
  if (process.stdin.isTTY) return '';
628
905
  const chunks = [];
@@ -681,7 +958,7 @@ async function main() {
681
958
 
682
959
  if (result.continue) {
683
960
  if (result.finalNotice) await markNotified(projectRoot, payload, ledger);
684
- else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null);
961
+ else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null, evidencePromptKey(state));
685
962
  process.stdout.write(`${JSON.stringify({ decision: 'block', reason: result.reason })}\n`);
686
963
  } else if (result.capped || result.notify) {
687
964
  // Non-blocking endings (non-gated modes, or cap reached after the final notice) must
@@ -82,17 +82,49 @@ function sleep(ms) {
82
82
  return new Promise((resolve) => setTimeout(resolve, ms));
83
83
  }
84
84
 
85
+ function isPidAlive(pid) {
86
+ try {
87
+ process.kill(pid, 0);
88
+ return true;
89
+ } catch (error) {
90
+ // EPERM: the process exists but belongs to another user — still alive.
91
+ return error?.code === 'EPERM';
92
+ }
93
+ }
94
+
95
+ async function readLockOwner(lockPath) {
96
+ try {
97
+ const raw = JSON.parse(await fs.readFile(path.join(lockPath, 'owner'), 'utf8'));
98
+ const pid = Number(raw?.pid);
99
+ return Number.isInteger(pid) && pid > 0
100
+ ? { pid, token: typeof raw?.token === 'string' ? raw.token : null }
101
+ : null;
102
+ } catch {
103
+ return null;
104
+ }
105
+ }
106
+
107
+ // In-process holder registry: same-pid holders are parallel async flows whose liveness a
108
+ // pid probe cannot prove, so the module tracks them itself.
109
+ const inProcessLockHolders = new Map();
110
+
85
111
  /**
86
112
  * Serialize read-modify-write mutations of a shared state file — across processes
87
113
  * (hook invocations run as separate node processes) and across concurrent async
88
114
  * flows in one process (parallel subagents). The lock is a directory created next
89
115
  * to the target file: `mkdir` is atomic, so exactly one caller can create it.
90
- * A crashed holder is reclaimed once the directory's mtime exceeds staleMs.
116
+ * Ownership is recorded in an `owner` file inside the lock dir: stale reclaim first
117
+ * proves the recorded holder is gone (dead pid, or no in-process holder for our own
118
+ * pid — no owner file means a pre-token holder and keeps the legacy mtime-only
119
+ * reclaim), so a slow-but-alive holder on a crawling disk is waited out, not stolen.
120
+ * Release only removes a dir this acquisition still owns, so a reclaimed-then-
121
+ * re-acquired lock is never deleted out from under its successor.
91
122
  * Liveness wins over strictness: if the lock cannot be acquired within maxWaitMs
92
123
  * the callback runs anyway (the pre-lock behaviour) — these state files are
93
124
  * advisory caches, and losing an update beats freezing a hook mid-flight.
94
125
  * Protocol-compatible with src/core/fileOps.js withFileLock (same `<file>.lock`
95
- * path), so CLI processes and hook processes serialize against each other.
126
+ * path and owner-file format), so CLI processes and hook processes serialize
127
+ * against each other.
96
128
  * @param {string} filePath - state file the mutation targets (lock lives beside it)
97
129
  * @param {() => Promise<*>} fn - critical section; its result is returned
98
130
  * @returns {Promise<*>} whatever fn resolves with
@@ -100,24 +132,48 @@ function sleep(ms) {
100
132
  export async function withFileLock(filePath, fn, { staleMs = LOCK_STALE_MS, maxWaitMs = LOCK_MAX_WAIT_MS } = {}) {
101
133
  const lockPath = `${filePath}.lock`;
102
134
  const startedAt = Date.now();
135
+ const ownerToken = `${process.pid}-${crypto.randomBytes(8).toString('hex')}`;
103
136
  let locked = false;
137
+ let ownerStamped = false;
104
138
 
105
139
  while (!locked) {
106
140
  try {
107
141
  await fs.mkdir(path.dirname(lockPath), { recursive: true });
108
142
  await fs.mkdir(lockPath); // atomic acquire — EEXIST means another holder exists
109
143
  locked = true;
144
+ inProcessLockHolders.set(lockPath, ownerToken);
145
+ try {
146
+ await fs.writeFile(
147
+ path.join(lockPath, 'owner'),
148
+ `${JSON.stringify({ pid: process.pid, token: ownerToken, ts: Date.now() })}\n`,
149
+ 'utf8',
150
+ );
151
+ ownerStamped = true;
152
+ } catch {
153
+ ownerStamped = false; // unverifiable release skips removal; stale reclaim cleans up
154
+ }
110
155
  break;
111
156
  } catch (error) {
112
157
  if (error?.code !== 'EEXIST') throw error;
113
158
  }
114
159
 
115
- // Someone holds the lock. Reclaim it when it looks abandoned; otherwise back off.
160
+ // Someone holds the lock. Reclaim it only when the holder is provably gone.
116
161
  try {
117
162
  const stat = await fs.stat(lockPath);
118
163
  if (Date.now() - stat.mtimeMs > staleMs) {
119
- await fs.rm(lockPath, { recursive: true, force: true });
120
- continue; // the slot is free now — retry immediately
164
+ const owner = await readLockOwner(lockPath);
165
+ const liveInProcess = inProcessLockHolders.has(lockPath);
166
+ // Stealing a live holder reintroduces the exact interleaved-write race this
167
+ // lock exists to prevent, and the stolen holder's release then deleted the
168
+ // successor's lock. Only a dead pid (or a leaked same-pid dir with no live
169
+ // registered flow) may be reclaimed.
170
+ const reclaimable = !owner || owner.pid === process.pid
171
+ ? !liveInProcess
172
+ : !isPidAlive(owner.pid);
173
+ if (reclaimable) {
174
+ await fs.rm(lockPath, { recursive: true, force: true });
175
+ continue; // the slot is free now — retry immediately
176
+ }
121
177
  }
122
178
  } catch {
123
179
  continue; // lock vanished between mkdir and stat — retry immediately
@@ -132,7 +188,14 @@ export async function withFileLock(filePath, fn, { staleMs = LOCK_STALE_MS, maxW
132
188
  } finally {
133
189
  if (locked) {
134
190
  try {
135
- await fs.rm(lockPath, { recursive: true, force: true });
191
+ // Remove the lock only if THIS acquisition still owns it: after a stale reclaim
192
+ // another holder may already own the dir, and deleting it would unlock their
193
+ // critical section for a third waiter.
194
+ const current = ownerStamped ? await readLockOwner(lockPath) : null;
195
+ if (current && current.token === ownerToken) {
196
+ await fs.rm(lockPath, { recursive: true, force: true });
197
+ }
198
+ if (inProcessLockHolders.get(lockPath) === ownerToken) inProcessLockHolders.delete(lockPath);
136
199
  } catch {
137
200
  // best-effort release; a stale lock is reclaimed by the next waiter
138
201
  }