@ngockhoale/ukit 2.3.8 → 2.3.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +90 -0
- package/package.json +1 -1
- package/src/core/compact/threshold.js +13 -0
- package/src/core/fileOps.js +61 -7
- package/src/index/taskRouting.js +35 -2
- package/templates/.claude/agents/bug-debugger.md +2 -2
- package/templates/.claude/agents/feature-implementer.md +5 -3
- package/templates/.claude/hooks/auto-allow-bash.sh +44 -7
- package/templates/.claude/hooks/auto-prune-bash.sh +44 -7
- package/templates/.claude/hooks/context-hardcap-gate.sh +33 -8
- package/templates/.claude/hooks/context-window-guard.sh +1 -1
- package/templates/.claude/hooks/reset-compact-pressure.sh +47 -7
- package/templates/.claude/hooks/skill-router.sh +80 -7
- package/templates/.claude/hooks/verification-guard.sh +55 -20
- package/templates/.claude/ukit/index/route-task.mjs +36 -3
- package/templates/.claude/ukit/runtime/compact-threshold.mjs +60 -5
- package/templates/.claude/ukit/runtime/execution-ledger.mjs +361 -62
- package/templates/.claude/ukit/runtime/token-utils.mjs +69 -6
- package/templates/.omp/agents/bug-debugger.md +2 -2
- package/templates/.omp/agents/feature-implementer.md +5 -3
- package/templates/.omp/hooks/pre/ukit-bridge.js +68 -16
|
@@ -13,6 +13,21 @@ const RESUME_INTENT_TTL_MS = 30 * 60 * 1000;
|
|
|
13
13
|
const MAX_RECEIPTS = 24;
|
|
14
14
|
const MAX_SOURCE_FILES = 16;
|
|
15
15
|
const MAX_CONTINUATIONS = 6;
|
|
16
|
+
// Two adjacent identical failed verifications (no edit attempt between them) are a loop;
|
|
17
|
+
// vibecode routes — which bypass the cap and every reentrant valve — additionally get a
|
|
18
|
+
// cumulative escape after the same verification failed this many times across edits.
|
|
19
|
+
const VERIFICATION_LOOP_STREAK = 2;
|
|
20
|
+
const VIBECODE_VERIFICATION_FAILURE_LIMIT = 4;
|
|
21
|
+
const MAX_TRACKED_VERIFICATIONS = 8;
|
|
22
|
+
// hook-chain-runner.mjs gives every hook script a 4s child budget, and omp blocks Edit|Write
|
|
23
|
+
// when a chain script is killed. Waiting longer than that for a contended lock got the
|
|
24
|
+
// recording hook killed mid-chain (receipt lost, edit blocked). Give up and fail open well
|
|
25
|
+
// inside the budget instead — under extreme contention a receipt may be lost, but the hook
|
|
26
|
+
// is never killed by its own chain.
|
|
27
|
+
const HOOK_SAFE_LOCK_WAIT_MS = 2500;
|
|
28
|
+
function withLedgerLock(target, fn) {
|
|
29
|
+
return withFileLock(target, fn, { maxWaitMs: HOOK_SAFE_LOCK_WAIT_MS });
|
|
30
|
+
}
|
|
16
31
|
const IMPLEMENT_MODES = new Set([
|
|
17
32
|
'tiny-fix',
|
|
18
33
|
'local-fix',
|
|
@@ -86,9 +101,14 @@ async function readJson(filePath, fallback = null) {
|
|
|
86
101
|
}
|
|
87
102
|
}
|
|
88
103
|
|
|
104
|
+
// Parallel writers in one process (subagent Stop hooks, concurrent receipts) used to share
|
|
105
|
+
// a single `pid`-suffixed temp path: one rename removed the file under the other and the
|
|
106
|
+
// writer crashed with ENOENT mid-hook. Make every write's temp path unique.
|
|
107
|
+
let atomicWriteCounter = 0;
|
|
89
108
|
async function writeJsonAtomic(filePath, value) {
|
|
90
109
|
await fs.mkdir(path.dirname(filePath), { recursive: true });
|
|
91
|
-
|
|
110
|
+
atomicWriteCounter += 1;
|
|
111
|
+
const tempPath = `${filePath}.${process.pid}-${atomicWriteCounter}-${Math.random().toString(16).slice(2)}.tmp`;
|
|
92
112
|
await fs.writeFile(tempPath, `${JSON.stringify(value, null, 2)}\n`, 'utf8');
|
|
93
113
|
await fs.rename(tempPath, filePath);
|
|
94
114
|
}
|
|
@@ -276,8 +296,129 @@ function fileMatchesExpected(receiptFile, expectedFile) {
|
|
|
276
296
|
return false;
|
|
277
297
|
}
|
|
278
298
|
|
|
299
|
+
// Piping a verification command into a read-only consumer (`yarn test 2>&1 | tail -6`) is
|
|
300
|
+
// the standard idiom for trimming output. The FIRST pipe segment is still the planned
|
|
301
|
+
// verification, so it may mint targeted evidence — but only when every later segment is a
|
|
302
|
+
// benign text consumer. Anything that can reinterpret or execute output (sh, xargs, tee)
|
|
303
|
+
// keeps the receipt broad.
|
|
304
|
+
const BENIGN_OUTPUT_CONSUMERS = new Set([
|
|
305
|
+
'tail', 'head', 'grep', 'egrep', 'fgrep', 'rg', 'ag', 'sed', 'cat',
|
|
306
|
+
'awk', 'cut', 'wc', 'sort', 'uniq', 'tr', 'nl',
|
|
307
|
+
]);
|
|
308
|
+
|
|
309
|
+
function isBenignOutputConsumer(segment) {
|
|
310
|
+
const firstWord = String(segment || '').trim().split(/\s+/)[0] || '';
|
|
311
|
+
return BENIGN_OUTPUT_CONSUMERS.has(firstWord);
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
function terminalShellCommandUnit(command) {
|
|
315
|
+
const text = String(command || '').trim();
|
|
316
|
+
if (!text) return null;
|
|
317
|
+
|
|
318
|
+
let quote = null;
|
|
319
|
+
let escaped = false;
|
|
320
|
+
let pipeBuffer = '';
|
|
321
|
+
let currentPipes = [];
|
|
322
|
+
const segments = [];
|
|
323
|
+
|
|
324
|
+
const closePipe = () => {
|
|
325
|
+
const trimmed = pipeBuffer.trim();
|
|
326
|
+
if (!trimmed) return null;
|
|
327
|
+
currentPipes.push(trimmed);
|
|
328
|
+
pipeBuffer = '';
|
|
329
|
+
return trimmed;
|
|
330
|
+
};
|
|
331
|
+
const closeSegment = () => {
|
|
332
|
+
if (!closePipe()) return false;
|
|
333
|
+
segments.push(currentPipes);
|
|
334
|
+
currentPipes = [];
|
|
335
|
+
return true;
|
|
336
|
+
};
|
|
337
|
+
// Redirections (`> file`, `2>&1`, `>> x`, `&> y`, `< in`) modify the current command's
|
|
338
|
+
// streams; they are consumed so the fd digits/paths do not corrupt the unit text.
|
|
339
|
+
const skipRedirectionTarget = (index) => {
|
|
340
|
+
let next = index;
|
|
341
|
+
while (next < text.length && /\s/.test(text[next])) next += 1;
|
|
342
|
+
while (
|
|
343
|
+
next < text.length
|
|
344
|
+
&& !/\s/.test(text[next])
|
|
345
|
+
&& !'&|;()<>`\n\r'.includes(text[next])
|
|
346
|
+
) next += 1;
|
|
347
|
+
return next;
|
|
348
|
+
};
|
|
349
|
+
|
|
350
|
+
for (let index = 0; index < text.length; index += 1) {
|
|
351
|
+
const char = text[index];
|
|
352
|
+
if (escaped) {
|
|
353
|
+
escaped = false;
|
|
354
|
+
pipeBuffer += char;
|
|
355
|
+
continue;
|
|
356
|
+
}
|
|
357
|
+
if (char === '\\') {
|
|
358
|
+
escaped = true;
|
|
359
|
+
pipeBuffer += char;
|
|
360
|
+
continue;
|
|
361
|
+
}
|
|
362
|
+
if (quote) {
|
|
363
|
+
if (char === quote) quote = null;
|
|
364
|
+
pipeBuffer += char;
|
|
365
|
+
continue;
|
|
366
|
+
}
|
|
367
|
+
if (char === "'" || char === '"') {
|
|
368
|
+
quote = char;
|
|
369
|
+
pipeBuffer += char;
|
|
370
|
+
continue;
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
// This deliberately recognizes only simple top-level `&&` chains and pipes into
|
|
374
|
+
// benign consumers. Other shell control/grouping syntax can mask a terminal
|
|
375
|
+
// command's status, so it must remain broad rather than targeted evidence.
|
|
376
|
+
if (char === ';' || char === '`' || char === '(' || char === ')' || char === '\n' || char === '\r') {
|
|
377
|
+
return null;
|
|
378
|
+
}
|
|
379
|
+
if (char === '$' && text[index + 1] === '(') return null;
|
|
380
|
+
if (char === '>' || char === '<') {
|
|
381
|
+
let next = index + 1;
|
|
382
|
+
if (text[next] === '>' && char === '>') next += 1;
|
|
383
|
+
if (text[next] === '&') next += 1;
|
|
384
|
+
index = skipRedirectionTarget(next) - 1;
|
|
385
|
+
continue;
|
|
386
|
+
}
|
|
387
|
+
if (char === '&' && text[index + 1] === '>') {
|
|
388
|
+
let next = index + 2;
|
|
389
|
+
if (text[next] === '>') next += 1;
|
|
390
|
+
index = skipRedirectionTarget(next) - 1;
|
|
391
|
+
continue;
|
|
392
|
+
}
|
|
393
|
+
if (char === '|') {
|
|
394
|
+
if (text[index + 1] === '|') return null;
|
|
395
|
+
if (!closePipe()) return null;
|
|
396
|
+
continue;
|
|
397
|
+
}
|
|
398
|
+
if (char === '&') {
|
|
399
|
+
if (text[index + 1] !== '&') return null;
|
|
400
|
+
if (!closeSegment()) return null;
|
|
401
|
+
index += 1;
|
|
402
|
+
continue;
|
|
403
|
+
}
|
|
404
|
+
pipeBuffer += char;
|
|
405
|
+
}
|
|
406
|
+
|
|
407
|
+
if (quote || escaped) return null;
|
|
408
|
+
if (!closeSegment()) return null;
|
|
409
|
+
|
|
410
|
+
const terminalSegment = segments[segments.length - 1];
|
|
411
|
+
if (terminalSegment.length > 1) {
|
|
412
|
+
const consumers = terminalSegment.slice(1);
|
|
413
|
+
if (!consumers.every((consumer) => isBenignOutputConsumer(consumer))) {
|
|
414
|
+
return null;
|
|
415
|
+
}
|
|
416
|
+
}
|
|
417
|
+
return terminalSegment[0];
|
|
418
|
+
}
|
|
419
|
+
|
|
279
420
|
function matchesRoutedCommand(command, routedCommand) {
|
|
280
|
-
const receipt =
|
|
421
|
+
const receipt = terminalShellCommandUnit(command);
|
|
281
422
|
const routed = String(routedCommand || '').trim();
|
|
282
423
|
if (!receipt || !routed) return false;
|
|
283
424
|
return receipt === routed || receipt.startsWith(routed) || routed.startsWith(receipt);
|
|
@@ -317,6 +458,17 @@ function carriedEvidenceLedger(fresh, current) {
|
|
|
317
458
|
verificationSucceeded: fresh.verificationSucceeded || current.verificationSucceeded === true,
|
|
318
459
|
verificationFailed: fresh.verificationFailed || current.verificationFailed === true,
|
|
319
460
|
receipts: [...(current.receipts || [])].slice(-MAX_RECEIPTS),
|
|
461
|
+
// The continuation budget belongs to the same logical request (promptKey), so a re-key
|
|
462
|
+
// must keep counting toward the cap. Resetting it here made the cap unreachable and the
|
|
463
|
+
// Stop gate loop forever on harnesses without a stop_hook_active valve.
|
|
464
|
+
continuationCount: Math.max(Number(fresh.continuationCount || 0), Number(current.continuationCount || 0)),
|
|
465
|
+
continuationRequestKey: current.continuationRequestKey || fresh.continuationRequestKey || null,
|
|
466
|
+
notified: fresh.notified === true || current.notified === true,
|
|
467
|
+
// Verification-loop tracking and any minted blocker belong to the same logical request
|
|
468
|
+
// too — dropping a blocker on re-key would resume the exact loop it recorded.
|
|
469
|
+
failedVerificationStreak: current.failedVerificationStreak || null,
|
|
470
|
+
verificationFailureCounts: current.verificationFailureCounts || {},
|
|
471
|
+
blocker: current.blocker || null,
|
|
320
472
|
};
|
|
321
473
|
}
|
|
322
474
|
|
|
@@ -331,6 +483,7 @@ function evidencePromptKey(routeState) {
|
|
|
331
483
|
if (!promptText) return null;
|
|
332
484
|
return `prompt-${crypto.createHash('sha256').update(promptText).digest('hex').slice(0, 20)}`;
|
|
333
485
|
}
|
|
486
|
+
export { evidencePromptKey };
|
|
334
487
|
|
|
335
488
|
function freshLedger(payload, routeState, harness) {
|
|
336
489
|
return {
|
|
@@ -364,61 +517,123 @@ export async function recordExecutionReceipt({
|
|
|
364
517
|
harness = 'unknown',
|
|
365
518
|
} = {}) {
|
|
366
519
|
if (!projectRoot || !toolName) return null;
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
520
|
+
// The whole read-modify-write is locked: parallel subagents record receipts through the
|
|
521
|
+
// same session ledger file, and an unlocked snapshot rewrite dropped whichever receipts
|
|
522
|
+
// landed between the read and the write.
|
|
523
|
+
return withLedgerLock(ledgerPath(projectRoot, payload), async () => {
|
|
524
|
+
const routeState = await readRouteState(projectRoot, payload);
|
|
525
|
+
const current = await readExecutionLedger(projectRoot, payload);
|
|
526
|
+
const nextRequestKey = routeState?.requestKey || null;
|
|
527
|
+
// A missing/foreign route state must never blank the evidence this session already
|
|
528
|
+
// banked. The shared state slot is single-owner (a parallel subagent re-stamps it), so
|
|
529
|
+
// routeState === null is routine mid-request — rebuilding from fresh there re-demanded
|
|
530
|
+
// every evidence the request had produced and froze the run. Without a live requestKey
|
|
531
|
+
// the only safe move is to keep appending to the current ledger.
|
|
532
|
+
const ledger = current && !nextRequestKey
|
|
533
|
+
? { ...current, harness: current.harness || harness }
|
|
534
|
+
: (!current || current.requestKey !== nextRequestKey
|
|
535
|
+
? carriedEvidenceLedger(freshLedger(payload, routeState, harness), current)
|
|
536
|
+
: { ...current, harness: current.harness || harness });
|
|
537
|
+
|
|
538
|
+
const failed = explicitError(payload);
|
|
539
|
+
const exitCode = extractExitCode(payload);
|
|
540
|
+
const success = !failed && (exitCode === null || exitCode === 0);
|
|
541
|
+
const toolInput = payload.tool_input || {};
|
|
542
|
+
const receipt = {
|
|
543
|
+
ts: Date.now(),
|
|
544
|
+
toolName,
|
|
545
|
+
toolUseId: payload.tool_use_id || null,
|
|
546
|
+
success,
|
|
547
|
+
exitCode,
|
|
548
|
+
};
|
|
384
549
|
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
}
|
|
392
|
-
} else if (toolName === 'Edit' || toolName === 'Write') {
|
|
393
|
-
receipt.kind = 'write';
|
|
394
|
-
receipt.file = toolInput.file_path || toolInput.path || toolInput.paths?.[0] || null;
|
|
395
|
-
ledger.writeAttempted = true;
|
|
396
|
-
ledger.writeSucceeded ||= success;
|
|
397
|
-
} else if (toolName === 'Bash' && isVerificationCommand(toolInput.command)) {
|
|
398
|
-
receipt.kind = 'verification';
|
|
399
|
-
receipt.command = String(toolInput.command || '').trim();
|
|
400
|
-
// WS-C routed-verification receipt: a command counts as "targeted" when it matches the
|
|
401
|
-
// routed plan (preferredOrder / primaryCommands). Off-plan verification still records
|
|
402
|
-
// as broad — counted only when the route carried no commands at all.
|
|
403
|
-
const routedCommands = routedVerificationCommands(routeState?.routeSummary);
|
|
404
|
-
if (routedCommands.length > 0) {
|
|
405
|
-
const matched = routedCommands.find((cmd) => matchesRoutedCommand(receipt.command, cmd));
|
|
406
|
-
receipt.scope = matched ? 'targeted' : 'broad';
|
|
407
|
-
if (receipt.scope === 'targeted') {
|
|
408
|
-
ledger.targetedVerificationSucceeded ||= success;
|
|
550
|
+
if (toolName === 'Read' || toolName === 'Grep' || toolName === 'Glob') {
|
|
551
|
+
receipt.kind = 'source';
|
|
552
|
+
receipt.file = toolInput.file_path || toolInput.path || null;
|
|
553
|
+
ledger.sourceSucceeded ||= success;
|
|
554
|
+
if (success && receipt.file) {
|
|
555
|
+
ledger.sourceFiles = [...new Set([...(ledger.sourceFiles || []), receipt.file])].slice(-MAX_SOURCE_FILES);
|
|
409
556
|
}
|
|
557
|
+
} else if (toolName === 'Edit' || toolName === 'Write') {
|
|
558
|
+
receipt.kind = 'write';
|
|
559
|
+
receipt.file = toolInput.file_path || toolInput.path || toolInput.paths?.[0] || null;
|
|
560
|
+
ledger.writeAttempted = true;
|
|
561
|
+
ledger.writeSucceeded ||= success;
|
|
562
|
+
// A mutation attempt between failures breaks the "no change in between" loop shape.
|
|
563
|
+
if (ledger.failedVerificationStreak) ledger.failedVerificationStreak = null;
|
|
564
|
+
} else if (toolName === 'Bash' && isVerificationCommand(toolInput.command)) {
|
|
565
|
+
receipt.kind = 'verification';
|
|
566
|
+
receipt.command = String(toolInput.command || '').trim();
|
|
567
|
+
// WS-C routed-verification receipt: a command counts as "targeted" when it matches the
|
|
568
|
+
// routed plan (preferredOrder / primaryCommands). Off-plan verification still records
|
|
569
|
+
// as broad — counted only when the route carried no commands at all.
|
|
570
|
+
const routedCommands = routedVerificationCommands(routeState?.routeSummary);
|
|
571
|
+
if (routedCommands.length > 0) {
|
|
572
|
+
const matched = routedCommands.find((cmd) => matchesRoutedCommand(receipt.command, cmd));
|
|
573
|
+
receipt.scope = matched ? 'targeted' : 'broad';
|
|
574
|
+
if (receipt.scope === 'targeted') {
|
|
575
|
+
ledger.targetedVerificationSucceeded ||= success;
|
|
576
|
+
}
|
|
577
|
+
}
|
|
578
|
+
ledger.verificationAttempted = true;
|
|
579
|
+
ledger.verificationSucceeded ||= success;
|
|
580
|
+
ledger.verificationFailed ||= !success;
|
|
581
|
+
|
|
582
|
+
// Verification-loop tracking. `terminalShellCommandUnit` gives the loop identity:
|
|
583
|
+
// the same failing check rerun — `setup && yarn test` and `yarn test 2>&1 | tail`
|
|
584
|
+
// both count as the same command, while a different check resets the streak.
|
|
585
|
+
const fingerprint = terminalShellCommandUnit(receipt.command) || receipt.command;
|
|
586
|
+
const counts = { ...(ledger.verificationFailureCounts || {}) };
|
|
587
|
+
if (success) {
|
|
588
|
+
delete counts[fingerprint];
|
|
589
|
+
ledger.verificationFailureCounts = counts;
|
|
590
|
+
if (ledger.failedVerificationStreak?.fingerprint === fingerprint) {
|
|
591
|
+
ledger.failedVerificationStreak = null;
|
|
592
|
+
}
|
|
593
|
+
} else {
|
|
594
|
+
const streakFingerprint = ledger.failedVerificationStreak?.fingerprint;
|
|
595
|
+
const streakCount = streakFingerprint === fingerprint
|
|
596
|
+
? Number(ledger.failedVerificationStreak.count || 0) + 1
|
|
597
|
+
: 1;
|
|
598
|
+
ledger.failedVerificationStreak = { fingerprint, count: streakCount };
|
|
599
|
+
counts[fingerprint] = Number(counts[fingerprint] || 0) + 1;
|
|
600
|
+
const trackedKeys = Object.keys(counts);
|
|
601
|
+
if (trackedKeys.length > MAX_TRACKED_VERIFICATIONS) {
|
|
602
|
+
for (const key of trackedKeys.slice(0, trackedKeys.length - MAX_TRACKED_VERIFICATIONS)) {
|
|
603
|
+
delete counts[key];
|
|
604
|
+
}
|
|
605
|
+
}
|
|
606
|
+
ledger.verificationFailureCounts = counts;
|
|
607
|
+
|
|
608
|
+
const vibecode = routeState?.routeSummary?.autonomyLevel === 'vibecode';
|
|
609
|
+
const shortCommand = fingerprint.length > 80 ? `${fingerprint.slice(0, 77)}…` : fingerprint;
|
|
610
|
+
if (streakCount >= VERIFICATION_LOOP_STREAK) {
|
|
611
|
+
ledger.blocker = {
|
|
612
|
+
kind: 'verification-loop',
|
|
613
|
+
visible: true,
|
|
614
|
+
detail: `verification "${shortCommand}" failed ${streakCount} times in a row with no edit in between — this is a loop, not progress. Report the failing output as a concrete blocker instead of rerunning it.`,
|
|
615
|
+
command: fingerprint,
|
|
616
|
+
ts: Date.now(),
|
|
617
|
+
};
|
|
618
|
+
} else if (vibecode && counts[fingerprint] >= VIBECODE_VERIFICATION_FAILURE_LIMIT) {
|
|
619
|
+
ledger.blocker = {
|
|
620
|
+
kind: 'verification-loop',
|
|
621
|
+
visible: true,
|
|
622
|
+
detail: `verification "${shortCommand}" failed ${counts[fingerprint]} times in continuous (vibecode) execution — report the failing output as a concrete blocker instead of retrying.`,
|
|
623
|
+
command: fingerprint,
|
|
624
|
+
ts: Date.now(),
|
|
625
|
+
};
|
|
626
|
+
}
|
|
627
|
+
}
|
|
628
|
+
} else {
|
|
629
|
+
return ledger;
|
|
410
630
|
}
|
|
411
|
-
ledger.verificationAttempted = true;
|
|
412
|
-
ledger.verificationSucceeded ||= success;
|
|
413
|
-
ledger.verificationFailed ||= !success;
|
|
414
|
-
} else {
|
|
415
|
-
return ledger;
|
|
416
|
-
}
|
|
417
631
|
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
632
|
+
ledger.receipts = appendReceipt(ledger.receipts, receipt);
|
|
633
|
+
ledger.updatedAt = Date.now();
|
|
634
|
+
await writeJsonAtomic(ledgerPath(projectRoot, payload), ledger);
|
|
635
|
+
return ledger;
|
|
636
|
+
});
|
|
422
637
|
}
|
|
423
638
|
|
|
424
639
|
function requiredEvidence(state = {}) {
|
|
@@ -500,19 +715,80 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
|
500
715
|
const mode = routeSummary.executionMode || routeSummary.approachSelector?.executionMode || null;
|
|
501
716
|
const evidence = requiredEvidence(state);
|
|
502
717
|
if (ledger?.blocker) {
|
|
718
|
+
// A blocker minted by the gate itself (verification loop) is the run's exit path — it
|
|
719
|
+
// must be user-visible or the release is indistinguishable from a stall. A blocker
|
|
720
|
+
// minted by a permission decision is already named in the model's own reply and stays
|
|
721
|
+
// silent here.
|
|
722
|
+
if (ledger.blocker.visible === true) {
|
|
723
|
+
return {
|
|
724
|
+
continue: false,
|
|
725
|
+
notify: true,
|
|
726
|
+
missingEvidence: [],
|
|
727
|
+
reason: `UKit stopped automatic recovery: ${ledger.blocker.detail || 'a recorded blocker ended this run.'}`,
|
|
728
|
+
};
|
|
729
|
+
}
|
|
503
730
|
return { continue: false, notify: false, missingEvidence: [] };
|
|
504
731
|
}
|
|
505
732
|
if (evidence.length === 0) {
|
|
506
733
|
return { continue: false, notify: false, missingEvidence: [] };
|
|
507
734
|
}
|
|
508
735
|
|
|
509
|
-
|
|
736
|
+
// Request identity is the prompt (see evidencePromptKey): the router re-keys requestKey on
|
|
737
|
+
// every Edit/Write, so a requestKey mismatch alone must not discard the evidence — and the
|
|
738
|
+
// continuation budget — the request already accumulated. A different prompt keeps a clean slate.
|
|
739
|
+
const sameRequest = !ledger?.requestKey || !state?.requestKey
|
|
740
|
+
|| ledger.requestKey === state.requestKey
|
|
741
|
+
|| (ledger.promptKey && evidencePromptKey(state) === ledger.promptKey);
|
|
510
742
|
const effectiveLedger = sameRequest ? ledger : {};
|
|
511
743
|
const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger, routeSummary));
|
|
512
744
|
if (missingEvidence.length === 0) {
|
|
513
745
|
return { continue: false, notify: false, missingEvidence: [] };
|
|
514
746
|
}
|
|
515
747
|
|
|
748
|
+
// `find-cause` can validly end clean: an investigation may establish that no actionable
|
|
749
|
+
// defect exists. Its contract says a FIX cannot be claimed without write + verification;
|
|
750
|
+
// it does not make a mutation mandatory. Treating missing evidence as an unconditional
|
|
751
|
+
// command to edit made a clean audit self-block forever (Stop -> "make an Edit" -> no
|
|
752
|
+
// honest edit exists -> Stop again). Once a mutation was attempted, retain the normal
|
|
753
|
+
// recovery gate — only a no-mutation, recommend-only investigation that has not already
|
|
754
|
+
// observed a failed verification gets this release valve. A route that named an actionable
|
|
755
|
+
// command, or a failing check, has concrete unfinished work and must keep recovering.
|
|
756
|
+
if (
|
|
757
|
+
mode === 'find-cause'
|
|
758
|
+
&& routeSummary.policyMode === 'recommend-only'
|
|
759
|
+
&& !effectiveLedger.writeAttempted
|
|
760
|
+
&& !effectiveLedger.verificationFailed
|
|
761
|
+
) {
|
|
762
|
+
return {
|
|
763
|
+
continue: false,
|
|
764
|
+
notify: true,
|
|
765
|
+
missingEvidence,
|
|
766
|
+
reason: 'UKit investigation ended without a mutation. A clean audit is valid; report whether no actionable defect was found or a concrete blocker remains. Do not claim a bug was fixed without write and verification evidence.',
|
|
767
|
+
};
|
|
768
|
+
}
|
|
769
|
+
|
|
770
|
+
// Same reasoning as find-cause, for the analysis-shaped end of map-impact: an impact map
|
|
771
|
+
// whose honest outcome is "analysis complete, no change warranted" must be able to end
|
|
772
|
+
// visibly. The contract's completionRule only demands impact evidence BEFORE an edit claim
|
|
773
|
+
// — it does not make a mutation mandatory — but the gate treated missing write evidence as
|
|
774
|
+
// an order to edit, and an analysis-only request looped Stop → "make an Edit" → Stop
|
|
775
|
+
// forever. The valve opens only when the analysis half actually finished (impact evidence
|
|
776
|
+
// satisfied), no mutation was attempted, and no verification failed; anything less has
|
|
777
|
+
// concrete unfinished work and keeps recovering.
|
|
778
|
+
if (
|
|
779
|
+
mode === 'map-impact'
|
|
780
|
+
&& !missingEvidence.includes('impact-evidence')
|
|
781
|
+
&& effectiveLedger.writeAttempted !== true
|
|
782
|
+
&& effectiveLedger.verificationFailed !== true
|
|
783
|
+
) {
|
|
784
|
+
return {
|
|
785
|
+
continue: false,
|
|
786
|
+
notify: true,
|
|
787
|
+
missingEvidence,
|
|
788
|
+
reason: 'UKit impact analysis ended without a mutation. A completed impact map with no warranted change is valid; report the findings and whether a follow-up edit is needed. Do not claim any fix without write and verification evidence.',
|
|
789
|
+
};
|
|
790
|
+
}
|
|
791
|
+
|
|
516
792
|
const gated = IMPLEMENT_MODES.has(mode);
|
|
517
793
|
if (!gated) {
|
|
518
794
|
return {
|
|
@@ -525,10 +801,12 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
|
525
801
|
|
|
526
802
|
// Continuation attempts only make sense within the request that minted them: a new
|
|
527
803
|
// routed request in the same session must start with a fresh budget, otherwise a cap
|
|
528
|
-
// exhausted on task A suppresses recovery for task B.
|
|
804
|
+
// exhausted on task A suppresses recovery for task B. A requestKey mismatch within the
|
|
805
|
+
// same promptKey is just the router re-keying mid-request — not a new request.
|
|
529
806
|
const staleContinuations = effectiveLedger?.continuationRequestKey
|
|
530
807
|
&& state?.requestKey
|
|
531
|
-
&& effectiveLedger.continuationRequestKey !== state.requestKey
|
|
808
|
+
&& effectiveLedger.continuationRequestKey !== state.requestKey
|
|
809
|
+
&& !(effectiveLedger.promptKey && evidencePromptKey(state) === effectiveLedger.promptKey);
|
|
532
810
|
const continuationCount = staleContinuations
|
|
533
811
|
? 0
|
|
534
812
|
: Number(effectiveLedger?.continuationCount || 0);
|
|
@@ -566,19 +844,21 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
|
566
844
|
};
|
|
567
845
|
}
|
|
568
846
|
|
|
569
|
-
export async function incrementContinuation(projectRoot, payload = {}, ledger = null, requestKey = null) {
|
|
847
|
+
export async function incrementContinuation(projectRoot, payload = {}, ledger = null, requestKey = null, promptKey = null) {
|
|
570
848
|
const target = ledgerPath(projectRoot, payload);
|
|
571
849
|
// Locked re-read: parallel subagents firing Stop hooks share one ledger file, and a
|
|
572
850
|
// stale-snapshot rewrite would reset each other's continuationCount — the budget would
|
|
573
851
|
// never advance and the gate would keep issuing continuations.
|
|
574
|
-
return
|
|
852
|
+
return withLedgerLock(target, async () => {
|
|
575
853
|
const current = await readExecutionLedger(projectRoot, payload) || ledger || freshLedger(payload, null, 'unknown');
|
|
576
854
|
// Mirror evaluateCompletion's staleness rule: a count minted by an earlier request must
|
|
577
855
|
// not be carried into the new request's budget, or the cap fires early (evaluate says 0,
|
|
578
|
-
// persist says 7) and the next request inherits a nearly exhausted budget.
|
|
856
|
+
// persist says 7) and the next request inherits a nearly exhausted budget. Same-prompt
|
|
857
|
+
// re-keys are the same request, so their budget must keep counting toward the cap.
|
|
579
858
|
const stale = requestKey
|
|
580
859
|
&& current?.continuationRequestKey
|
|
581
|
-
&& current.continuationRequestKey !== requestKey
|
|
860
|
+
&& current.continuationRequestKey !== requestKey
|
|
861
|
+
&& !(promptKey && current.promptKey && promptKey === current.promptKey);
|
|
582
862
|
const next = {
|
|
583
863
|
...current,
|
|
584
864
|
continuationCount: (stale ? 0 : Number(current.continuationCount || 0)) + 1,
|
|
@@ -593,7 +873,7 @@ export async function incrementContinuation(projectRoot, payload = {}, ledger =
|
|
|
593
873
|
|
|
594
874
|
export async function markNotified(projectRoot, payload = {}, ledger = null) {
|
|
595
875
|
const target = ledgerPath(projectRoot, payload);
|
|
596
|
-
return
|
|
876
|
+
return withLedgerLock(target, async () => {
|
|
597
877
|
const current = await readExecutionLedger(projectRoot, payload) || ledger || freshLedger(payload, null, 'unknown');
|
|
598
878
|
const next = { ...current, notified: true, updatedAt: Date.now() };
|
|
599
879
|
await writeJsonAtomic(target, next);
|
|
@@ -601,6 +881,25 @@ export async function markNotified(projectRoot, payload = {}, ledger = null) {
|
|
|
601
881
|
});
|
|
602
882
|
}
|
|
603
883
|
|
|
884
|
+
// omp's session_stop carries no stop_hook_active marker, so reentrancy must be detected from
|
|
885
|
+
// the ledger itself: a stop whose recovery turn produced no new receipts is the reentrant
|
|
886
|
+
// shape Claude Code flags natively. The fingerprint deliberately excludes continuation
|
|
887
|
+
// bookkeeping fields (count/notified/updatedAt) so this call's own writes stay invisible to
|
|
888
|
+
// the next comparison — only real receipts change it.
|
|
889
|
+
export async function noteStopProgress(projectRoot, payload = {}) {
|
|
890
|
+
const target = ledgerPath(projectRoot, payload);
|
|
891
|
+
return withLedgerLock(target, async () => {
|
|
892
|
+
const current = await readExecutionLedger(projectRoot, payload);
|
|
893
|
+
if (!current) return { reentrant: false };
|
|
894
|
+
const receipts = Array.isArray(current.receipts) ? current.receipts : [];
|
|
895
|
+
const fingerprint = `${receipts.length}:${receipts[receipts.length - 1]?.ts || 0}`;
|
|
896
|
+
const reentrant = typeof current.stopProgressFingerprint === 'string'
|
|
897
|
+
&& current.stopProgressFingerprint === fingerprint;
|
|
898
|
+
await writeJsonAtomic(target, { ...current, stopProgressFingerprint: fingerprint, updatedAt: Date.now() });
|
|
899
|
+
return { reentrant };
|
|
900
|
+
});
|
|
901
|
+
}
|
|
902
|
+
|
|
604
903
|
async function readStdin() {
|
|
605
904
|
if (process.stdin.isTTY) return '';
|
|
606
905
|
const chunks = [];
|
|
@@ -659,7 +958,7 @@ async function main() {
|
|
|
659
958
|
|
|
660
959
|
if (result.continue) {
|
|
661
960
|
if (result.finalNotice) await markNotified(projectRoot, payload, ledger);
|
|
662
|
-
else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null);
|
|
961
|
+
else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null, evidencePromptKey(state));
|
|
663
962
|
process.stdout.write(`${JSON.stringify({ decision: 'block', reason: result.reason })}\n`);
|
|
664
963
|
} else if (result.capped || result.notify) {
|
|
665
964
|
// Non-blocking endings (non-gated modes, or cap reached after the final notice) must
|
|
@@ -82,17 +82,49 @@ function sleep(ms) {
|
|
|
82
82
|
return new Promise((resolve) => setTimeout(resolve, ms));
|
|
83
83
|
}
|
|
84
84
|
|
|
85
|
+
function isPidAlive(pid) {
|
|
86
|
+
try {
|
|
87
|
+
process.kill(pid, 0);
|
|
88
|
+
return true;
|
|
89
|
+
} catch (error) {
|
|
90
|
+
// EPERM: the process exists but belongs to another user — still alive.
|
|
91
|
+
return error?.code === 'EPERM';
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
async function readLockOwner(lockPath) {
|
|
96
|
+
try {
|
|
97
|
+
const raw = JSON.parse(await fs.readFile(path.join(lockPath, 'owner'), 'utf8'));
|
|
98
|
+
const pid = Number(raw?.pid);
|
|
99
|
+
return Number.isInteger(pid) && pid > 0
|
|
100
|
+
? { pid, token: typeof raw?.token === 'string' ? raw.token : null }
|
|
101
|
+
: null;
|
|
102
|
+
} catch {
|
|
103
|
+
return null;
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
// In-process holder registry: same-pid holders are parallel async flows whose liveness a
|
|
108
|
+
// pid probe cannot prove, so the module tracks them itself.
|
|
109
|
+
const inProcessLockHolders = new Map();
|
|
110
|
+
|
|
85
111
|
/**
|
|
86
112
|
* Serialize read-modify-write mutations of a shared state file — across processes
|
|
87
113
|
* (hook invocations run as separate node processes) and across concurrent async
|
|
88
114
|
* flows in one process (parallel subagents). The lock is a directory created next
|
|
89
115
|
* to the target file: `mkdir` is atomic, so exactly one caller can create it.
|
|
90
|
-
*
|
|
116
|
+
* Ownership is recorded in an `owner` file inside the lock dir: stale reclaim first
|
|
117
|
+
* proves the recorded holder is gone (dead pid, or no in-process holder for our own
|
|
118
|
+
* pid — no owner file means a pre-token holder and keeps the legacy mtime-only
|
|
119
|
+
* reclaim), so a slow-but-alive holder on a crawling disk is waited out, not stolen.
|
|
120
|
+
* Release only removes a dir this acquisition still owns, so a reclaimed-then-
|
|
121
|
+
* re-acquired lock is never deleted out from under its successor.
|
|
91
122
|
* Liveness wins over strictness: if the lock cannot be acquired within maxWaitMs
|
|
92
123
|
* the callback runs anyway (the pre-lock behaviour) — these state files are
|
|
93
124
|
* advisory caches, and losing an update beats freezing a hook mid-flight.
|
|
94
125
|
* Protocol-compatible with src/core/fileOps.js withFileLock (same `<file>.lock`
|
|
95
|
-
* path), so CLI processes and hook processes serialize
|
|
126
|
+
* path and owner-file format), so CLI processes and hook processes serialize
|
|
127
|
+
* against each other.
|
|
96
128
|
* @param {string} filePath - state file the mutation targets (lock lives beside it)
|
|
97
129
|
* @param {() => Promise<*>} fn - critical section; its result is returned
|
|
98
130
|
* @returns {Promise<*>} whatever fn resolves with
|
|
@@ -100,24 +132,48 @@ function sleep(ms) {
|
|
|
100
132
|
export async function withFileLock(filePath, fn, { staleMs = LOCK_STALE_MS, maxWaitMs = LOCK_MAX_WAIT_MS } = {}) {
|
|
101
133
|
const lockPath = `${filePath}.lock`;
|
|
102
134
|
const startedAt = Date.now();
|
|
135
|
+
const ownerToken = `${process.pid}-${crypto.randomBytes(8).toString('hex')}`;
|
|
103
136
|
let locked = false;
|
|
137
|
+
let ownerStamped = false;
|
|
104
138
|
|
|
105
139
|
while (!locked) {
|
|
106
140
|
try {
|
|
107
141
|
await fs.mkdir(path.dirname(lockPath), { recursive: true });
|
|
108
142
|
await fs.mkdir(lockPath); // atomic acquire — EEXIST means another holder exists
|
|
109
143
|
locked = true;
|
|
144
|
+
inProcessLockHolders.set(lockPath, ownerToken);
|
|
145
|
+
try {
|
|
146
|
+
await fs.writeFile(
|
|
147
|
+
path.join(lockPath, 'owner'),
|
|
148
|
+
`${JSON.stringify({ pid: process.pid, token: ownerToken, ts: Date.now() })}\n`,
|
|
149
|
+
'utf8',
|
|
150
|
+
);
|
|
151
|
+
ownerStamped = true;
|
|
152
|
+
} catch {
|
|
153
|
+
ownerStamped = false; // unverifiable release skips removal; stale reclaim cleans up
|
|
154
|
+
}
|
|
110
155
|
break;
|
|
111
156
|
} catch (error) {
|
|
112
157
|
if (error?.code !== 'EEXIST') throw error;
|
|
113
158
|
}
|
|
114
159
|
|
|
115
|
-
// Someone holds the lock. Reclaim it when
|
|
160
|
+
// Someone holds the lock. Reclaim it only when the holder is provably gone.
|
|
116
161
|
try {
|
|
117
162
|
const stat = await fs.stat(lockPath);
|
|
118
163
|
if (Date.now() - stat.mtimeMs > staleMs) {
|
|
119
|
-
await
|
|
120
|
-
|
|
164
|
+
const owner = await readLockOwner(lockPath);
|
|
165
|
+
const liveInProcess = inProcessLockHolders.has(lockPath);
|
|
166
|
+
// Stealing a live holder reintroduces the exact interleaved-write race this
|
|
167
|
+
// lock exists to prevent, and the stolen holder's release then deleted the
|
|
168
|
+
// successor's lock. Only a dead pid (or a leaked same-pid dir with no live
|
|
169
|
+
// registered flow) may be reclaimed.
|
|
170
|
+
const reclaimable = !owner || owner.pid === process.pid
|
|
171
|
+
? !liveInProcess
|
|
172
|
+
: !isPidAlive(owner.pid);
|
|
173
|
+
if (reclaimable) {
|
|
174
|
+
await fs.rm(lockPath, { recursive: true, force: true });
|
|
175
|
+
continue; // the slot is free now — retry immediately
|
|
176
|
+
}
|
|
121
177
|
}
|
|
122
178
|
} catch {
|
|
123
179
|
continue; // lock vanished between mkdir and stat — retry immediately
|
|
@@ -132,7 +188,14 @@ export async function withFileLock(filePath, fn, { staleMs = LOCK_STALE_MS, maxW
|
|
|
132
188
|
} finally {
|
|
133
189
|
if (locked) {
|
|
134
190
|
try {
|
|
135
|
-
|
|
191
|
+
// Remove the lock only if THIS acquisition still owns it: after a stale reclaim
|
|
192
|
+
// another holder may already own the dir, and deleting it would unlock their
|
|
193
|
+
// critical section for a third waiter.
|
|
194
|
+
const current = ownerStamped ? await readLockOwner(lockPath) : null;
|
|
195
|
+
if (current && current.token === ownerToken) {
|
|
196
|
+
await fs.rm(lockPath, { recursive: true, force: true });
|
|
197
|
+
}
|
|
198
|
+
if (inProcessLockHolders.get(lockPath) === ownerToken) inProcessLockHolders.delete(lockPath);
|
|
136
199
|
} catch {
|
|
137
200
|
// best-effort release; a stale lock is reclaimed by the next waiter
|
|
138
201
|
}
|
|
@@ -17,7 +17,7 @@ Systematic debugging — understand before fixing.
|
|
|
17
17
|
|
|
18
18
|
- Run the failing command/action.
|
|
19
19
|
- Capture exact error message and stack trace.
|
|
20
|
-
- If not reproducible → document conditions
|
|
20
|
+
- If not reproducible → document conditions, run the next most-discriminating bounded repro, then report any exact missing user-only artifact/permission to the parent agent. Never wait for or ask a user directly — you are a worker.
|
|
21
21
|
|
|
22
22
|
### 2. Trace Root Cause
|
|
23
23
|
|
|
@@ -82,4 +82,4 @@ Daily mode: skip. Handoff mode: set task status `pending_review` in `INDEX.md`;
|
|
|
82
82
|
- non-trivial bug: `docs/MEMORY.md` + `docs/PROJECT.md` + `docs/CODE_MAP.md`
|
|
83
83
|
- read `docs/WORKLOG.md` only recent relevant entries
|
|
84
84
|
- Keep fix scope minimal — no drive-by refactors.
|
|
85
|
-
- If root cause is unclear after 5 minutes of tracing → ask user
|
|
85
|
+
- If root cause is unclear after 5 minutes of tracing → try one bounded alternative hypothesis/repro, then report precise evidence plus the smallest needed missing context to the parent agent. Never wait for or ask a user directly — the parent owns user communication.
|