@ngockhoale/ukit 2.3.9 → 2.3.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -499,6 +499,21 @@ export function resetCompactPressureState(rawState = {}, config = {}) {
499
499
  }
500
500
 
501
501
  const BOUNDARY_TAIL_BYTES = 4 * 1024 * 1024;
502
+ // Hook processes are short-lived, so an in-memory cache would not stop every Edit/Write
503
+ // from reopening and parsing a 4MB transcript tail. Persist a very short probe window in
504
+ // the session pressure record instead: a burst of mutations incurs one scan, not one per
505
+ // call. PreCompact is still immediate; this is only the fallback for host auto-compactions
506
+ // that skip PreCompact, where a two-second detection delay is preferable to killing the
507
+ // 4s hook budget on a slow volume.
508
+ const TRANSCRIPT_BOUNDARY_PROBE_TTL_MS = 2_000;
509
+
510
+ function normalizeTranscriptProbe(value) {
511
+ if (!value || typeof value !== 'object' || typeof value.path !== 'string' || !value.path) return null;
512
+ const checkedAt = finiteNumber(value.checkedAt, 0);
513
+ return checkedAt > 0
514
+ ? { path: value.path, checkedAt, boundaryAt: finiteNumber(value.boundaryAt, 0) }
515
+ : null;
516
+ }
502
517
 
503
518
  // Find the newest compaction recorded in the session transcript, as epoch ms (0 if none).
504
519
  // Every real compaction — manual /compact AND harness auto-compact — appends a
@@ -551,18 +566,46 @@ export async function detectLatestCompactBoundaryAt(transcriptPath, { tailBytes
551
566
  // history that was already summarised away and the hard-cap gate bricks a healthy
552
567
  // session. Callers must persist when reset=true, or the reset would re-fire (and
553
568
  // zero the counter) on every call.
554
- export async function syncCompactPressureStateWithTranscript(rawState = null, config = {}, transcriptPath = '') {
555
- const boundaryAt = await detectLatestCompactBoundaryAt(transcriptPath);
569
+ export async function syncCompactPressureStateWithTranscript(
570
+ rawState = null,
571
+ config = {},
572
+ transcriptPath = '',
573
+ { now = Date.now(), probeTtlMs = TRANSCRIPT_BOUNDARY_PROBE_TTL_MS } = {},
574
+ ) {
575
+ const normalizedPath = typeof transcriptPath === 'string' ? transcriptPath.trim() : '';
576
+ const previousProbe = normalizeTranscriptProbe(rawState?.transcriptBoundaryProbe);
577
+ // A process-persistent, session-scoped cache avoids rereading/parsing up to 4MB for every
578
+ // mutation during a tool burst. It keys on the path (not just time) so a resumed/new
579
+ // transcript always scans immediately. The short TTL is deliberately only a performance
580
+ // coalescing window; PreCompact remains the immediate source of truth.
581
+ const probeFresh = previousProbe
582
+ && previousProbe.path === normalizedPath
583
+ && now - previousProbe.checkedAt >= 0
584
+ && now - previousProbe.checkedAt < probeTtlMs;
585
+ const boundaryAt = probeFresh
586
+ ? previousProbe.boundaryAt
587
+ : await detectLatestCompactBoundaryAt(normalizedPath);
588
+ const probe = probeFresh
589
+ ? previousProbe
590
+ : { path: normalizedPath, checkedAt: now, boundaryAt };
556
591
  const accounted = finiteNumber(rawState?.lastCompactBoundaryAt, 0);
557
592
  if (!boundaryAt || boundaryAt <= accounted) {
558
- return { state: rawState, reset: false };
593
+ return {
594
+ state: rawState && typeof rawState === 'object'
595
+ ? { ...rawState, transcriptBoundaryProbe: probe }
596
+ : rawState,
597
+ reset: false,
598
+ probed: !probeFresh,
599
+ };
559
600
  }
560
601
  return {
561
602
  state: {
562
603
  ...resetCompactPressureState(rawState ?? {}, config),
563
604
  lastCompactBoundaryAt: boundaryAt,
605
+ transcriptBoundaryProbe: probe,
564
606
  },
565
607
  reset: true,
608
+ probed: !probeFresh,
566
609
  };
567
610
  }
568
611
 
@@ -795,6 +838,7 @@ export function buildCompactPressureState(rawState = null, config = {}) {
795
838
  taskMode,
796
839
  cooldownUntil: finiteNumber(rawState?.cooldownUntil, 0),
797
840
  lastCompactBoundaryAt: finiteNumber(rawState?.lastCompactBoundaryAt, 0),
841
+ transcriptBoundaryProbe: normalizeTranscriptProbe(rawState?.transcriptBoundaryProbe),
798
842
  routingContext: rawState?.routingContext && typeof rawState.routingContext === 'object'
799
843
  ? rawState.routingContext
800
844
  : {},
@@ -1260,7 +1304,9 @@ async function runCli() {
1260
1304
  sessionConfig,
1261
1305
  payload.transcript_path,
1262
1306
  );
1263
- if (synced.reset) {
1307
+ if (synced.reset || synced.probed) {
1308
+ // Persist a clean probe too: this CLI runs in a fresh process on each hook,
1309
+ // so retaining only resets would make every prompt rescan the same tail.
1264
1310
  await writeCompactPressureState(projectRoot, synced.state, sessionConfig);
1265
1311
  }
1266
1312
  } catch { /* advisory healing only — never block the prompt over it */ }
@@ -13,6 +13,21 @@ const RESUME_INTENT_TTL_MS = 30 * 60 * 1000;
13
13
  const MAX_RECEIPTS = 24;
14
14
  const MAX_SOURCE_FILES = 16;
15
15
  const MAX_CONTINUATIONS = 6;
16
+ // Two adjacent identical failed verifications (no edit attempt between them) are a loop;
17
+ // vibecode routes — which bypass the cap and every reentrant valve — additionally get a
18
+ // cumulative escape after the same verification failed this many times across edits.
19
+ const VERIFICATION_LOOP_STREAK = 2;
20
+ const VIBECODE_VERIFICATION_FAILURE_LIMIT = 4;
21
+ const MAX_TRACKED_VERIFICATIONS = 8;
22
+ // hook-chain-runner.mjs gives every hook script a 4s child budget, and omp blocks Edit|Write
23
+ // when a chain script is killed. Waiting longer than that for a contended lock got the
24
+ // recording hook killed mid-chain (receipt lost, edit blocked). Give up and fail open well
25
+ // inside the budget instead — under extreme contention a receipt may be lost, but the hook
26
+ // is never killed by its own chain.
27
+ const HOOK_SAFE_LOCK_WAIT_MS = 2500;
28
+ function withLedgerLock(target, fn) {
29
+ return withFileLock(target, fn, { maxWaitMs: HOOK_SAFE_LOCK_WAIT_MS });
30
+ }
16
31
  const IMPLEMENT_MODES = new Set([
17
32
  'tiny-fix',
18
33
  'local-fix',
@@ -86,9 +101,14 @@ async function readJson(filePath, fallback = null) {
86
101
  }
87
102
  }
88
103
 
104
+ // Parallel writers in one process (subagent Stop hooks, concurrent receipts) used to share
105
+ // a single `pid`-suffixed temp path: one rename removed the file under the other and the
106
+ // writer crashed with ENOENT mid-hook. Make every write's temp path unique.
107
+ let atomicWriteCounter = 0;
89
108
  async function writeJsonAtomic(filePath, value) {
90
109
  await fs.mkdir(path.dirname(filePath), { recursive: true });
91
- const tempPath = `${filePath}.${process.pid}.tmp`;
110
+ atomicWriteCounter += 1;
111
+ const tempPath = `${filePath}.${process.pid}-${atomicWriteCounter}-${Math.random().toString(16).slice(2)}.tmp`;
92
112
  await fs.writeFile(tempPath, `${JSON.stringify(value, null, 2)}\n`, 'utf8');
93
113
  await fs.rename(tempPath, filePath);
94
114
  }
@@ -276,8 +296,129 @@ function fileMatchesExpected(receiptFile, expectedFile) {
276
296
  return false;
277
297
  }
278
298
 
299
+ // Piping a verification command into a read-only consumer (`yarn test 2>&1 | tail -6`) is
300
+ // the standard idiom for trimming output. The FIRST pipe segment is still the planned
301
+ // verification, so it may mint targeted evidence — but only when every later segment is a
302
+ // benign text consumer. Anything that can reinterpret or execute output (sh, xargs, tee)
303
+ // keeps the receipt broad.
304
+ const BENIGN_OUTPUT_CONSUMERS = new Set([
305
+ 'tail', 'head', 'grep', 'egrep', 'fgrep', 'rg', 'ag', 'sed', 'cat',
306
+ 'awk', 'cut', 'wc', 'sort', 'uniq', 'tr', 'nl',
307
+ ]);
308
+
309
+ function isBenignOutputConsumer(segment) {
310
+ const firstWord = String(segment || '').trim().split(/\s+/)[0] || '';
311
+ return BENIGN_OUTPUT_CONSUMERS.has(firstWord);
312
+ }
313
+
314
+ function terminalShellCommandUnit(command) {
315
+ const text = String(command || '').trim();
316
+ if (!text) return null;
317
+
318
+ let quote = null;
319
+ let escaped = false;
320
+ let pipeBuffer = '';
321
+ let currentPipes = [];
322
+ const segments = [];
323
+
324
+ const closePipe = () => {
325
+ const trimmed = pipeBuffer.trim();
326
+ if (!trimmed) return null;
327
+ currentPipes.push(trimmed);
328
+ pipeBuffer = '';
329
+ return trimmed;
330
+ };
331
+ const closeSegment = () => {
332
+ if (!closePipe()) return false;
333
+ segments.push(currentPipes);
334
+ currentPipes = [];
335
+ return true;
336
+ };
337
+ // Redirections (`> file`, `2>&1`, `>> x`, `&> y`, `< in`) modify the current command's
338
+ // streams; they are consumed so the fd digits/paths do not corrupt the unit text.
339
+ const skipRedirectionTarget = (index) => {
340
+ let next = index;
341
+ while (next < text.length && /\s/.test(text[next])) next += 1;
342
+ while (
343
+ next < text.length
344
+ && !/\s/.test(text[next])
345
+ && !'&|;()<>`\n\r'.includes(text[next])
346
+ ) next += 1;
347
+ return next;
348
+ };
349
+
350
+ for (let index = 0; index < text.length; index += 1) {
351
+ const char = text[index];
352
+ if (escaped) {
353
+ escaped = false;
354
+ pipeBuffer += char;
355
+ continue;
356
+ }
357
+ if (char === '\\') {
358
+ escaped = true;
359
+ pipeBuffer += char;
360
+ continue;
361
+ }
362
+ if (quote) {
363
+ if (char === quote) quote = null;
364
+ pipeBuffer += char;
365
+ continue;
366
+ }
367
+ if (char === "'" || char === '"') {
368
+ quote = char;
369
+ pipeBuffer += char;
370
+ continue;
371
+ }
372
+
373
+ // This deliberately recognizes only simple top-level `&&` chains and pipes into
374
+ // benign consumers. Other shell control/grouping syntax can mask a terminal
375
+ // command's status, so it must remain broad rather than targeted evidence.
376
+ if (char === ';' || char === '`' || char === '(' || char === ')' || char === '\n' || char === '\r') {
377
+ return null;
378
+ }
379
+ if (char === '$' && text[index + 1] === '(') return null;
380
+ if (char === '>' || char === '<') {
381
+ let next = index + 1;
382
+ if (text[next] === '>' && char === '>') next += 1;
383
+ if (text[next] === '&') next += 1;
384
+ index = skipRedirectionTarget(next) - 1;
385
+ continue;
386
+ }
387
+ if (char === '&' && text[index + 1] === '>') {
388
+ let next = index + 2;
389
+ if (text[next] === '>') next += 1;
390
+ index = skipRedirectionTarget(next) - 1;
391
+ continue;
392
+ }
393
+ if (char === '|') {
394
+ if (text[index + 1] === '|') return null;
395
+ if (!closePipe()) return null;
396
+ continue;
397
+ }
398
+ if (char === '&') {
399
+ if (text[index + 1] !== '&') return null;
400
+ if (!closeSegment()) return null;
401
+ index += 1;
402
+ continue;
403
+ }
404
+ pipeBuffer += char;
405
+ }
406
+
407
+ if (quote || escaped) return null;
408
+ if (!closeSegment()) return null;
409
+
410
+ const terminalSegment = segments[segments.length - 1];
411
+ if (terminalSegment.length > 1) {
412
+ const consumers = terminalSegment.slice(1);
413
+ if (!consumers.every((consumer) => isBenignOutputConsumer(consumer))) {
414
+ return null;
415
+ }
416
+ }
417
+ return terminalSegment[0];
418
+ }
419
+
279
420
  function matchesRoutedCommand(command, routedCommand) {
280
- const receipt = String(command || '').trim();
421
+ const receipt = terminalShellCommandUnit(command);
281
422
  const routed = String(routedCommand || '').trim();
282
423
  if (!receipt || !routed) return false;
283
424
  return receipt === routed || receipt.startsWith(routed) || routed.startsWith(receipt);
@@ -317,6 +458,17 @@ function carriedEvidenceLedger(fresh, current) {
317
458
  verificationSucceeded: fresh.verificationSucceeded || current.verificationSucceeded === true,
318
459
  verificationFailed: fresh.verificationFailed || current.verificationFailed === true,
319
460
  receipts: [...(current.receipts || [])].slice(-MAX_RECEIPTS),
461
+ // The continuation budget belongs to the same logical request (promptKey), so a re-key
462
+ // must keep counting toward the cap. Resetting it here made the cap unreachable and the
463
+ // Stop gate loop forever on harnesses without a stop_hook_active valve.
464
+ continuationCount: Math.max(Number(fresh.continuationCount || 0), Number(current.continuationCount || 0)),
465
+ continuationRequestKey: current.continuationRequestKey || fresh.continuationRequestKey || null,
466
+ notified: fresh.notified === true || current.notified === true,
467
+ // Verification-loop tracking and any minted blocker belong to the same logical request
468
+ // too — dropping a blocker on re-key would resume the exact loop it recorded.
469
+ failedVerificationStreak: current.failedVerificationStreak || null,
470
+ verificationFailureCounts: current.verificationFailureCounts || {},
471
+ blocker: current.blocker || null,
320
472
  };
321
473
  }
322
474
 
@@ -331,6 +483,7 @@ function evidencePromptKey(routeState) {
331
483
  if (!promptText) return null;
332
484
  return `prompt-${crypto.createHash('sha256').update(promptText).digest('hex').slice(0, 20)}`;
333
485
  }
486
+ export { evidencePromptKey };
334
487
 
335
488
  function freshLedger(payload, routeState, harness) {
336
489
  return {
@@ -364,61 +517,123 @@ export async function recordExecutionReceipt({
364
517
  harness = 'unknown',
365
518
  } = {}) {
366
519
  if (!projectRoot || !toolName) return null;
367
- const routeState = await readRouteState(projectRoot, payload);
368
- const current = await readExecutionLedger(projectRoot, payload);
369
- const ledger = !current || current.requestKey !== (routeState?.requestKey || null)
370
- ? carriedEvidenceLedger(freshLedger(payload, routeState, harness), current)
371
- : { ...current, harness: current.harness || harness };
372
-
373
- const failed = explicitError(payload);
374
- const exitCode = extractExitCode(payload);
375
- const success = !failed && (exitCode === null || exitCode === 0);
376
- const toolInput = payload.tool_input || {};
377
- const receipt = {
378
- ts: Date.now(),
379
- toolName,
380
- toolUseId: payload.tool_use_id || null,
381
- success,
382
- exitCode,
383
- };
520
+ // The whole read-modify-write is locked: parallel subagents record receipts through the
521
+ // same session ledger file, and an unlocked snapshot rewrite dropped whichever receipts
522
+ // landed between the read and the write.
523
+ return withLedgerLock(ledgerPath(projectRoot, payload), async () => {
524
+ const routeState = await readRouteState(projectRoot, payload);
525
+ const current = await readExecutionLedger(projectRoot, payload);
526
+ const nextRequestKey = routeState?.requestKey || null;
527
+ // A missing/foreign route state must never blank the evidence this session already
528
+ // banked. The shared state slot is single-owner (a parallel subagent re-stamps it), so
529
+ // routeState === null is routine mid-request — rebuilding from fresh there re-demanded
530
+ // every evidence the request had produced and froze the run. Without a live requestKey
531
+ // the only safe move is to keep appending to the current ledger.
532
+ const ledger = current && !nextRequestKey
533
+ ? { ...current, harness: current.harness || harness }
534
+ : (!current || current.requestKey !== nextRequestKey
535
+ ? carriedEvidenceLedger(freshLedger(payload, routeState, harness), current)
536
+ : { ...current, harness: current.harness || harness });
537
+
538
+ const failed = explicitError(payload);
539
+ const exitCode = extractExitCode(payload);
540
+ const success = !failed && (exitCode === null || exitCode === 0);
541
+ const toolInput = payload.tool_input || {};
542
+ const receipt = {
543
+ ts: Date.now(),
544
+ toolName,
545
+ toolUseId: payload.tool_use_id || null,
546
+ success,
547
+ exitCode,
548
+ };
384
549
 
385
- if (toolName === 'Read' || toolName === 'Grep' || toolName === 'Glob') {
386
- receipt.kind = 'source';
387
- receipt.file = toolInput.file_path || toolInput.path || null;
388
- ledger.sourceSucceeded ||= success;
389
- if (success && receipt.file) {
390
- ledger.sourceFiles = [...new Set([...(ledger.sourceFiles || []), receipt.file])].slice(-MAX_SOURCE_FILES);
391
- }
392
- } else if (toolName === 'Edit' || toolName === 'Write') {
393
- receipt.kind = 'write';
394
- receipt.file = toolInput.file_path || toolInput.path || toolInput.paths?.[0] || null;
395
- ledger.writeAttempted = true;
396
- ledger.writeSucceeded ||= success;
397
- } else if (toolName === 'Bash' && isVerificationCommand(toolInput.command)) {
398
- receipt.kind = 'verification';
399
- receipt.command = String(toolInput.command || '').trim();
400
- // WS-C routed-verification receipt: a command counts as "targeted" when it matches the
401
- // routed plan (preferredOrder / primaryCommands). Off-plan verification still records
402
- // as broad — counted only when the route carried no commands at all.
403
- const routedCommands = routedVerificationCommands(routeState?.routeSummary);
404
- if (routedCommands.length > 0) {
405
- const matched = routedCommands.find((cmd) => matchesRoutedCommand(receipt.command, cmd));
406
- receipt.scope = matched ? 'targeted' : 'broad';
407
- if (receipt.scope === 'targeted') {
408
- ledger.targetedVerificationSucceeded ||= success;
550
+ if (toolName === 'Read' || toolName === 'Grep' || toolName === 'Glob') {
551
+ receipt.kind = 'source';
552
+ receipt.file = toolInput.file_path || toolInput.path || null;
553
+ ledger.sourceSucceeded ||= success;
554
+ if (success && receipt.file) {
555
+ ledger.sourceFiles = [...new Set([...(ledger.sourceFiles || []), receipt.file])].slice(-MAX_SOURCE_FILES);
556
+ }
557
+ } else if (toolName === 'Edit' || toolName === 'Write') {
558
+ receipt.kind = 'write';
559
+ receipt.file = toolInput.file_path || toolInput.path || toolInput.paths?.[0] || null;
560
+ ledger.writeAttempted = true;
561
+ ledger.writeSucceeded ||= success;
562
+ // A mutation attempt between failures breaks the "no change in between" loop shape.
563
+ if (ledger.failedVerificationStreak) ledger.failedVerificationStreak = null;
564
+ } else if (toolName === 'Bash' && isVerificationCommand(toolInput.command)) {
565
+ receipt.kind = 'verification';
566
+ receipt.command = String(toolInput.command || '').trim();
567
+ // WS-C routed-verification receipt: a command counts as "targeted" when it matches the
568
+ // routed plan (preferredOrder / primaryCommands). Off-plan verification still records
569
+ // as broad — counted only when the route carried no commands at all.
570
+ const routedCommands = routedVerificationCommands(routeState?.routeSummary);
571
+ if (routedCommands.length > 0) {
572
+ const matched = routedCommands.find((cmd) => matchesRoutedCommand(receipt.command, cmd));
573
+ receipt.scope = matched ? 'targeted' : 'broad';
574
+ if (receipt.scope === 'targeted') {
575
+ ledger.targetedVerificationSucceeded ||= success;
576
+ }
409
577
  }
578
+ ledger.verificationAttempted = true;
579
+ ledger.verificationSucceeded ||= success;
580
+ ledger.verificationFailed ||= !success;
581
+
582
+ // Verification-loop tracking. `terminalShellCommandUnit` gives the loop identity:
583
+ // the same failing check rerun — `setup && yarn test` and `yarn test 2>&1 | tail`
584
+ // both count as the same command, while a different check resets the streak.
585
+ const fingerprint = terminalShellCommandUnit(receipt.command) || receipt.command;
586
+ const counts = { ...(ledger.verificationFailureCounts || {}) };
587
+ if (success) {
588
+ delete counts[fingerprint];
589
+ ledger.verificationFailureCounts = counts;
590
+ if (ledger.failedVerificationStreak?.fingerprint === fingerprint) {
591
+ ledger.failedVerificationStreak = null;
592
+ }
593
+ } else {
594
+ const streakFingerprint = ledger.failedVerificationStreak?.fingerprint;
595
+ const streakCount = streakFingerprint === fingerprint
596
+ ? Number(ledger.failedVerificationStreak.count || 0) + 1
597
+ : 1;
598
+ ledger.failedVerificationStreak = { fingerprint, count: streakCount };
599
+ counts[fingerprint] = Number(counts[fingerprint] || 0) + 1;
600
+ const trackedKeys = Object.keys(counts);
601
+ if (trackedKeys.length > MAX_TRACKED_VERIFICATIONS) {
602
+ for (const key of trackedKeys.slice(0, trackedKeys.length - MAX_TRACKED_VERIFICATIONS)) {
603
+ delete counts[key];
604
+ }
605
+ }
606
+ ledger.verificationFailureCounts = counts;
607
+
608
+ const vibecode = routeState?.routeSummary?.autonomyLevel === 'vibecode';
609
+ const shortCommand = fingerprint.length > 80 ? `${fingerprint.slice(0, 77)}…` : fingerprint;
610
+ if (streakCount >= VERIFICATION_LOOP_STREAK) {
611
+ ledger.blocker = {
612
+ kind: 'verification-loop',
613
+ visible: true,
614
+ detail: `verification "${shortCommand}" failed ${streakCount} times in a row with no edit in between — this is a loop, not progress. Report the failing output as a concrete blocker instead of rerunning it.`,
615
+ command: fingerprint,
616
+ ts: Date.now(),
617
+ };
618
+ } else if (vibecode && counts[fingerprint] >= VIBECODE_VERIFICATION_FAILURE_LIMIT) {
619
+ ledger.blocker = {
620
+ kind: 'verification-loop',
621
+ visible: true,
622
+ detail: `verification "${shortCommand}" failed ${counts[fingerprint]} times in continuous (vibecode) execution — report the failing output as a concrete blocker instead of retrying.`,
623
+ command: fingerprint,
624
+ ts: Date.now(),
625
+ };
626
+ }
627
+ }
628
+ } else {
629
+ return ledger;
410
630
  }
411
- ledger.verificationAttempted = true;
412
- ledger.verificationSucceeded ||= success;
413
- ledger.verificationFailed ||= !success;
414
- } else {
415
- return ledger;
416
- }
417
631
 
418
- ledger.receipts = appendReceipt(ledger.receipts, receipt);
419
- ledger.updatedAt = Date.now();
420
- await writeJsonAtomic(ledgerPath(projectRoot, payload), ledger);
421
- return ledger;
632
+ ledger.receipts = appendReceipt(ledger.receipts, receipt);
633
+ ledger.updatedAt = Date.now();
634
+ await writeJsonAtomic(ledgerPath(projectRoot, payload), ledger);
635
+ return ledger;
636
+ });
422
637
  }
423
638
 
424
639
  function requiredEvidence(state = {}) {
@@ -500,13 +715,30 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
500
715
  const mode = routeSummary.executionMode || routeSummary.approachSelector?.executionMode || null;
501
716
  const evidence = requiredEvidence(state);
502
717
  if (ledger?.blocker) {
718
+ // A blocker minted by the gate itself (verification loop) is the run's exit path — it
719
+ // must be user-visible or the release is indistinguishable from a stall. A blocker
720
+ // minted by a permission decision is already named in the model's own reply and stays
721
+ // silent here.
722
+ if (ledger.blocker.visible === true) {
723
+ return {
724
+ continue: false,
725
+ notify: true,
726
+ missingEvidence: [],
727
+ reason: `UKit stopped automatic recovery: ${ledger.blocker.detail || 'a recorded blocker ended this run.'}`,
728
+ };
729
+ }
503
730
  return { continue: false, notify: false, missingEvidence: [] };
504
731
  }
505
732
  if (evidence.length === 0) {
506
733
  return { continue: false, notify: false, missingEvidence: [] };
507
734
  }
508
735
 
509
- const sameRequest = !ledger?.requestKey || !state?.requestKey || ledger.requestKey === state.requestKey;
736
+ // Request identity is the prompt (see evidencePromptKey): the router re-keys requestKey on
737
+ // every Edit/Write, so a requestKey mismatch alone must not discard the evidence — and the
738
+ // continuation budget — the request already accumulated. A different prompt keeps a clean slate.
739
+ const sameRequest = !ledger?.requestKey || !state?.requestKey
740
+ || ledger.requestKey === state.requestKey
741
+ || (ledger.promptKey && evidencePromptKey(state) === ledger.promptKey);
510
742
  const effectiveLedger = sameRequest ? ledger : {};
511
743
  const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger, routeSummary));
512
744
  if (missingEvidence.length === 0) {
@@ -535,6 +767,28 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
535
767
  };
536
768
  }
537
769
 
770
+ // Same reasoning as find-cause, for the analysis-shaped end of map-impact: an impact map
771
+ // whose honest outcome is "analysis complete, no change warranted" must be able to end
772
+ // visibly. The contract's completionRule only demands impact evidence BEFORE an edit claim
773
+ // — it does not make a mutation mandatory — but the gate treated missing write evidence as
774
+ // an order to edit, and an analysis-only request looped Stop → "make an Edit" → Stop
775
+ // forever. The valve opens only when the analysis half actually finished (impact evidence
776
+ // satisfied), no mutation was attempted, and no verification failed; anything less has
777
+ // concrete unfinished work and keeps recovering.
778
+ if (
779
+ mode === 'map-impact'
780
+ && !missingEvidence.includes('impact-evidence')
781
+ && effectiveLedger.writeAttempted !== true
782
+ && effectiveLedger.verificationFailed !== true
783
+ ) {
784
+ return {
785
+ continue: false,
786
+ notify: true,
787
+ missingEvidence,
788
+ reason: 'UKit impact analysis ended without a mutation. A completed impact map with no warranted change is valid; report the findings and whether a follow-up edit is needed. Do not claim any fix without write and verification evidence.',
789
+ };
790
+ }
791
+
538
792
  const gated = IMPLEMENT_MODES.has(mode);
539
793
  if (!gated) {
540
794
  return {
@@ -547,10 +801,12 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
547
801
 
548
802
  // Continuation attempts only make sense within the request that minted them: a new
549
803
  // routed request in the same session must start with a fresh budget, otherwise a cap
550
- // exhausted on task A suppresses recovery for task B.
804
+ // exhausted on task A suppresses recovery for task B. A requestKey mismatch within the
805
+ // same promptKey is just the router re-keying mid-request — not a new request.
551
806
  const staleContinuations = effectiveLedger?.continuationRequestKey
552
807
  && state?.requestKey
553
- && effectiveLedger.continuationRequestKey !== state.requestKey;
808
+ && effectiveLedger.continuationRequestKey !== state.requestKey
809
+ && !(effectiveLedger.promptKey && evidencePromptKey(state) === effectiveLedger.promptKey);
554
810
  const continuationCount = staleContinuations
555
811
  ? 0
556
812
  : Number(effectiveLedger?.continuationCount || 0);
@@ -588,19 +844,21 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
588
844
  };
589
845
  }
590
846
 
591
- export async function incrementContinuation(projectRoot, payload = {}, ledger = null, requestKey = null) {
847
+ export async function incrementContinuation(projectRoot, payload = {}, ledger = null, requestKey = null, promptKey = null) {
592
848
  const target = ledgerPath(projectRoot, payload);
593
849
  // Locked re-read: parallel subagents firing Stop hooks share one ledger file, and a
594
850
  // stale-snapshot rewrite would reset each other's continuationCount — the budget would
595
851
  // never advance and the gate would keep issuing continuations.
596
- return withFileLock(target, async () => {
852
+ return withLedgerLock(target, async () => {
597
853
  const current = await readExecutionLedger(projectRoot, payload) || ledger || freshLedger(payload, null, 'unknown');
598
854
  // Mirror evaluateCompletion's staleness rule: a count minted by an earlier request must
599
855
  // not be carried into the new request's budget, or the cap fires early (evaluate says 0,
600
- // persist says 7) and the next request inherits a nearly exhausted budget.
856
+ // persist says 7) and the next request inherits a nearly exhausted budget. Same-prompt
857
+ // re-keys are the same request, so their budget must keep counting toward the cap.
601
858
  const stale = requestKey
602
859
  && current?.continuationRequestKey
603
- && current.continuationRequestKey !== requestKey;
860
+ && current.continuationRequestKey !== requestKey
861
+ && !(promptKey && current.promptKey && promptKey === current.promptKey);
604
862
  const next = {
605
863
  ...current,
606
864
  continuationCount: (stale ? 0 : Number(current.continuationCount || 0)) + 1,
@@ -615,7 +873,7 @@ export async function incrementContinuation(projectRoot, payload = {}, ledger =
615
873
 
616
874
  export async function markNotified(projectRoot, payload = {}, ledger = null) {
617
875
  const target = ledgerPath(projectRoot, payload);
618
- return withFileLock(target, async () => {
876
+ return withLedgerLock(target, async () => {
619
877
  const current = await readExecutionLedger(projectRoot, payload) || ledger || freshLedger(payload, null, 'unknown');
620
878
  const next = { ...current, notified: true, updatedAt: Date.now() };
621
879
  await writeJsonAtomic(target, next);
@@ -623,6 +881,25 @@ export async function markNotified(projectRoot, payload = {}, ledger = null) {
623
881
  });
624
882
  }
625
883
 
884
+ // omp's session_stop carries no stop_hook_active marker, so reentrancy must be detected from
885
+ // the ledger itself: a stop whose recovery turn produced no new receipts is the reentrant
886
+ // shape Claude Code flags natively. The fingerprint deliberately excludes continuation
887
+ // bookkeeping fields (count/notified/updatedAt) so this call's own writes stay invisible to
888
+ // the next comparison — only real receipts change it.
889
+ export async function noteStopProgress(projectRoot, payload = {}) {
890
+ const target = ledgerPath(projectRoot, payload);
891
+ return withLedgerLock(target, async () => {
892
+ const current = await readExecutionLedger(projectRoot, payload);
893
+ if (!current) return { reentrant: false };
894
+ const receipts = Array.isArray(current.receipts) ? current.receipts : [];
895
+ const fingerprint = `${receipts.length}:${receipts[receipts.length - 1]?.ts || 0}`;
896
+ const reentrant = typeof current.stopProgressFingerprint === 'string'
897
+ && current.stopProgressFingerprint === fingerprint;
898
+ await writeJsonAtomic(target, { ...current, stopProgressFingerprint: fingerprint, updatedAt: Date.now() });
899
+ return { reentrant };
900
+ });
901
+ }
902
+
626
903
  async function readStdin() {
627
904
  if (process.stdin.isTTY) return '';
628
905
  const chunks = [];
@@ -681,7 +958,7 @@ async function main() {
681
958
 
682
959
  if (result.continue) {
683
960
  if (result.finalNotice) await markNotified(projectRoot, payload, ledger);
684
- else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null);
961
+ else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null, evidencePromptKey(state));
685
962
  process.stdout.write(`${JSON.stringify({ decision: 'block', reason: result.reason })}\n`);
686
963
  } else if (result.capped || result.notify) {
687
964
  // Non-blocking endings (non-gated modes, or cap reached after the final notice) must