@ngockhoale/ukit 2.7.7 → 2.7.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +54 -0
- package/package.json +1 -1
- package/src/context/detectProjectContext.js +5 -0
- package/src/core/codeintel/invalidation.js +4 -0
- package/src/core/diffPlan.js +60 -1
- package/src/core/fileOps.js +46 -119
- package/src/render/buildVariables.js +10 -0
- package/templates/.claude/agents/bug-debugger.md +1 -1
- package/templates/.claude/agents/feature-implementer.md +2 -2
- package/templates/.claude/commands/ukit/handoff-clear.md +11 -0
- package/templates/.claude/commands/ukit/handoff-create.md +1 -1
- package/templates/.claude/commands/ukit/handoff-fullstack.md +26 -1
- package/templates/.claude/commands/ukit/handoff-implement.md +1 -1
- package/templates/.claude/commands/ukit/handoff-review.md +1 -1
- package/templates/.claude/hooks/context-hardcap-gate.sh +4 -1
- package/templates/.claude/hooks/handoff-model-guard.sh +22 -11
- package/templates/.claude/hooks/reset-compact-pressure.sh +10 -0
- package/templates/.claude/hooks/skill-router.sh +15 -8
- package/templates/.claude/hooks/verification-guard.sh +3 -0
- package/templates/.claude/ukit/index/route-task.mjs +237 -32
- package/templates/.claude/ukit/index/stale-spec-check.mjs +38 -3
- package/templates/.claude/ukit/runtime/async-lock.mjs +240 -42
- package/templates/.claude/ukit/runtime/compact-threshold.mjs +5 -2
- package/templates/.claude/ukit/runtime/execution-ledger.mjs +217 -17
- package/templates/.claude/ukit/runtime/hook-chain-runner.mjs +38 -4
- package/templates/.claude/ukit/runtime/hook-payload-store.mjs +131 -0
- package/templates/.claude/ukit/runtime/stop-coordinator.mjs +35 -20
- package/templates/.claude/ukit/runtime/token-utils.mjs +37 -126
- package/templates/.codex/settings.json +1 -5
- package/templates/.omp/agents/bug-debugger.md +1 -1
- package/templates/.omp/agents/feature-implementer.md +2 -2
- package/templates/.omp/hooks/pre/ukit-bridge.js +216 -51
- package/templates/docs/AI_HANDOFF/INDEX.md +1 -1
- package/templates/docs/AI_HANDOFF/RULES.md +6 -6
- package/templates/ukit/storage/config.json +2 -2
|
@@ -16,13 +16,12 @@ import {
|
|
|
16
16
|
markNotified,
|
|
17
17
|
readExecutionLedger,
|
|
18
18
|
readRouteState,
|
|
19
|
-
recordExecutionReceipt,
|
|
20
19
|
} from '../../../.claude/ukit/runtime/execution-ledger.mjs';
|
|
21
20
|
import {
|
|
22
21
|
PAYLOAD_INLINE_MAX_BYTES,
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
22
|
+
createPayloadReferenceAsync,
|
|
23
|
+
maybeSweepStalePayloadsAsync,
|
|
24
|
+
probePayloadIntegrityAsync,
|
|
26
25
|
} from '../../../.claude/ukit/runtime/hook-payload-store.mjs';
|
|
27
26
|
// TASK-018 review fix round 1: the chain budget is resolved by ONE shared module,
|
|
28
27
|
// so the runner's inner deadline and this bridge's outer pi.exec timeout can never
|
|
@@ -30,6 +29,13 @@ import {
|
|
|
30
29
|
// for Edit|Write is a fail-closed transport failure = every edit blocked).
|
|
31
30
|
import { resolveChainExecTimeoutMs } from '../../../.claude/ukit/runtime/hook-chain-budget.mjs';
|
|
32
31
|
|
|
32
|
+
// OMP-3 (TASK-002): pi.exec forwards only {cwd, signal, timeout} to its spawn —
|
|
33
|
+
// an `env` option is dropped (verified against the bundled omp exec). The only
|
|
34
|
+
// channel that reaches the chain runner AND its in-proc .mjs steps is the host
|
|
35
|
+
// environment, so the bridge stamps the harness here: record-execution.mjs
|
|
36
|
+
// reads env.UKIT_HARNESS (default 'claude-code') when journaling receipts.
|
|
37
|
+
process.env.UKIT_HARNESS = 'omp';
|
|
38
|
+
|
|
33
39
|
export const HOOK_EVENT_MAP = {
|
|
34
40
|
tool_call: {
|
|
35
41
|
'Read|Grep|Glob': ['sensitive-data-guard.mjs'],
|
|
@@ -140,6 +146,17 @@ function classifyFailure(scriptName) {
|
|
|
140
146
|
// closed (the .sh thin wrapper is still invocable as a fallback path).
|
|
141
147
|
const TIMEOUT_STAYS_CLOSED = new Set(['block-dangerous.sh', 'block-dangerous.mjs']);
|
|
142
148
|
|
|
149
|
+
// TASK-002 (OMP-4): a hook crash or transport failure whose stderr is a
|
|
150
|
+
// transient filesystem error (EAGAIN/EBUSY/EMFILE/ENFILE/ESTALE — the
|
|
151
|
+
// stalled-external-mount class) produced NO verdict, so it must degrade to a
|
|
152
|
+
// loud "could not verify" warning instead of a block showing raw errno text.
|
|
153
|
+
// The match is deliberately conservative: the errno code must appear in an
|
|
154
|
+
// fs-syscall context (open/read/write/mkdir/stat/rename/unlink/rm), in either
|
|
155
|
+
// order, or the Node "resource temporarily unavailable" phrasing — a genuine
|
|
156
|
+
// block reason containing e.g. "again" in prose never downgrades. ONE pattern
|
|
157
|
+
// shared by translateExecResult and runScriptChain.
|
|
158
|
+
const TRANSIENT_INFRA_STDERR_RE = /\bresource temporarily unavailable\b|\b(?:EAGAIN|EBUSY|EMFILE|ENFILE|ESTALE)\b[\s\S]*?\b(?:open|read|write|mkdir|stat|rename|unlink|rm)\b|\b(?:open|read|write|mkdir|stat|rename|unlink|rm)\b[\s\S]*?\b(?:EAGAIN|EBUSY|EMFILE|ENFILE|ESTALE)\b/i;
|
|
159
|
+
|
|
143
160
|
// TASK-018: the hook-chain-runner's failure taxonomy. Infrastructure outcomes
|
|
144
161
|
// (overflow / timeout / signal / budget-exhausted) produced NO verdict, so their
|
|
145
162
|
// captured output is untrustworthy and never reaches the model context, a block
|
|
@@ -184,10 +201,18 @@ function redactDiagnosticText(text, maxChars = 500) {
|
|
|
184
201
|
|
|
185
202
|
function runtimeMetadata(event = {}, context = {}) {
|
|
186
203
|
const sessionManager = context?.sessionManager;
|
|
204
|
+
const transcriptPath = event.transcriptPath ?? event.transcript_path ?? sessionManager?.getSessionFile?.();
|
|
187
205
|
return {
|
|
188
|
-
|
|
206
|
+
// F-8: one session key for EVERY event. omp session_start events carry no
|
|
207
|
+
// session id and the context may lack sessionManager — without one,
|
|
208
|
+
// reset-compact-pressure.sh takes the unknown-caller branch and wipes EVERY
|
|
209
|
+
// session's pressure records. Falling back to the transcript path here (not
|
|
210
|
+
// per-caller) keeps the reset key identical to the key the pressure writer
|
|
211
|
+
// used on prompt/tool events that also lack an id; a per-caller fallback
|
|
212
|
+
// let the two diverge and the targeted delete silently missed.
|
|
213
|
+
sessionId: event.sessionId ?? event.session_id ?? sessionManager?.getSessionId?.() ?? transcriptPath,
|
|
189
214
|
cwd: context?.cwd ?? event.cwd,
|
|
190
|
-
transcriptPath
|
|
215
|
+
transcriptPath,
|
|
191
216
|
};
|
|
192
217
|
}
|
|
193
218
|
|
|
@@ -323,6 +348,18 @@ function translateExecResult(scriptName, execResult) {
|
|
|
323
348
|
}
|
|
324
349
|
return { block: true, reason: stderr || `${scriptName} exited 2 (blocked)`, stdout, stderr };
|
|
325
350
|
}
|
|
351
|
+
// TASK-002 (OMP-4): a crash whose stderr is a transient fs error produced no
|
|
352
|
+
// verdict — it is an infrastructure event, not a safety decision. Fail open
|
|
353
|
+
// loudly instead of blocking on raw errno text. TIMEOUT_STAYS_CLOSED scripts
|
|
354
|
+
// (block-dangerous .sh/.mjs) keep their never-fail-open contract.
|
|
355
|
+
if (TRANSIENT_INFRA_STDERR_RE.test(stderr) && !TIMEOUT_STAYS_CLOSED.has(scriptName)) {
|
|
356
|
+
return {
|
|
357
|
+
block: false,
|
|
358
|
+
warning: `${scriptName} exited ${code} with a transient filesystem error — treated as "could not verify", not as a block (an infrastructure event, not a verdict): ${redactDiagnosticText(stderr) || 'no stderr'}`,
|
|
359
|
+
stdout,
|
|
360
|
+
stderr,
|
|
361
|
+
};
|
|
362
|
+
}
|
|
326
363
|
if (classifyFailure(scriptName) === 'closed') {
|
|
327
364
|
return {
|
|
328
365
|
block: true,
|
|
@@ -366,17 +403,18 @@ function hookErrorsDirFor(projectRoot) {
|
|
|
366
403
|
}
|
|
367
404
|
|
|
368
405
|
// Bounded work on THIS session's file only — identical keep-newest-half shape
|
|
369
|
-
// as hook-telemetry's rotateIfNeeded.
|
|
370
|
-
|
|
406
|
+
// as hook-telemetry's rotateIfNeeded. TASK-002 (OMP-4): async fs only — a
|
|
407
|
+
// stalled mount must never freeze the omp host loop.
|
|
408
|
+
async function rotateHookErrorFileIfNeeded(filePath, incomingBytes, maxBytes) {
|
|
371
409
|
let size = 0;
|
|
372
410
|
try {
|
|
373
|
-
size = fs.
|
|
411
|
+
size = (await fs.promises.stat(filePath)).size;
|
|
374
412
|
} catch {
|
|
375
413
|
return; // first row for this session
|
|
376
414
|
}
|
|
377
415
|
if (size + incomingBytes <= maxBytes) return;
|
|
378
416
|
try {
|
|
379
|
-
const lines = fs.
|
|
417
|
+
const lines = (await fs.promises.readFile(filePath, 'utf8')).split('\n');
|
|
380
418
|
if (lines.length && lines[lines.length - 1] === '') lines.pop();
|
|
381
419
|
const keepBudget = Math.floor(maxBytes / 2);
|
|
382
420
|
const keep = [];
|
|
@@ -387,16 +425,17 @@ function rotateHookErrorFileIfNeeded(filePath, incomingBytes, maxBytes) {
|
|
|
387
425
|
keep.unshift(lines[i]);
|
|
388
426
|
kept += lineBytes;
|
|
389
427
|
}
|
|
390
|
-
fs.
|
|
428
|
+
await fs.promises.writeFile(filePath, keep.length ? `${keep.join('\n')}\n` : '', 'utf8');
|
|
391
429
|
} catch {
|
|
392
430
|
// Rotation failed; drop this row rather than grow past the cap.
|
|
393
431
|
}
|
|
394
432
|
}
|
|
395
433
|
|
|
396
434
|
// sweepHookErrorsDir(dir, {now, maxAgeMs, maxFiles, maxEntries, maxRemovals}) ->
|
|
397
|
-
// { scanned, removed } — bounded: never scans or removes more than
|
|
398
|
-
// so a pre-existing oversized dir is amortized down across sampled
|
|
399
|
-
|
|
435
|
+
// Promise<{ scanned, removed }> — bounded: never scans or removes more than
|
|
436
|
+
// the caps, so a pre-existing oversized dir is amortized down across sampled
|
|
437
|
+
// sweeps. TASK-002 (OMP-4): async fs only — host loop must never block.
|
|
438
|
+
async function sweepHookErrorsDir(dir, {
|
|
400
439
|
now = Date.now,
|
|
401
440
|
maxAgeMs = HOOK_ERROR_MAX_AGE_MS,
|
|
402
441
|
maxFiles = HOOK_ERROR_MAX_FILES,
|
|
@@ -405,7 +444,7 @@ function sweepHookErrorsDir(dir, {
|
|
|
405
444
|
} = {}) {
|
|
406
445
|
let names;
|
|
407
446
|
try {
|
|
408
|
-
names = fs.
|
|
447
|
+
names = await fs.promises.readdir(dir);
|
|
409
448
|
} catch {
|
|
410
449
|
return { scanned: 0, removed: 0 };
|
|
411
450
|
}
|
|
@@ -421,7 +460,7 @@ function sweepHookErrorsDir(dir, {
|
|
|
421
460
|
if (scanned >= maxEntries) break;
|
|
422
461
|
scanned += 1;
|
|
423
462
|
try {
|
|
424
|
-
const stats = fs.
|
|
463
|
+
const stats = await fs.promises.stat(path.join(dir, name));
|
|
425
464
|
if (stats.isFile()) entries.push({ name, mtimeMs: stats.mtimeMs });
|
|
426
465
|
} catch { /* raced away — fine */ }
|
|
427
466
|
}
|
|
@@ -435,7 +474,7 @@ function sweepHookErrorsDir(dir, {
|
|
|
435
474
|
if (removed >= maxRemovals) break;
|
|
436
475
|
if (removed < overflow || entry.mtimeMs < cutoff) {
|
|
437
476
|
try {
|
|
438
|
-
fs.
|
|
477
|
+
await fs.promises.rm(path.join(dir, entry.name), { force: true });
|
|
439
478
|
removed += 1;
|
|
440
479
|
} catch { /* raced away — fine */ }
|
|
441
480
|
}
|
|
@@ -449,18 +488,18 @@ function hookErrorsSweepProbabilityFromEnv() {
|
|
|
449
488
|
return Math.min(1, Math.max(0, raw));
|
|
450
489
|
}
|
|
451
490
|
|
|
452
|
-
function recordHookErrorDiagnostic(projectRoot, sessionId, diagnostic, { maxBytes = HOOK_ERROR_MAX_BYTES } = {}) {
|
|
491
|
+
async function recordHookErrorDiagnostic(projectRoot, sessionId, diagnostic, { maxBytes = HOOK_ERROR_MAX_BYTES } = {}) {
|
|
453
492
|
try {
|
|
454
493
|
const dir = hookErrorsDirFor(projectRoot);
|
|
455
|
-
fs.
|
|
494
|
+
await fs.promises.mkdir(dir, { recursive: true });
|
|
456
495
|
const safeSession = String(sessionId || 'unknown').replace(/[^a-zA-Z0-9._-]/g, '_').slice(0, 96) || 'unknown';
|
|
457
496
|
const filePath = path.join(dir, `${safeSession}.jsonl`);
|
|
458
497
|
const line = `${JSON.stringify(diagnostic)}\n`;
|
|
459
|
-
rotateHookErrorFileIfNeeded(filePath, Buffer.byteLength(line, 'utf8'), maxBytes);
|
|
460
|
-
fs.
|
|
498
|
+
await rotateHookErrorFileIfNeeded(filePath, Buffer.byteLength(line, 'utf8'), maxBytes);
|
|
499
|
+
await fs.promises.appendFile(filePath, line, 'utf8');
|
|
461
500
|
// Sampled bounded dir sweep — amortizes down any pre-existing oversized dir.
|
|
462
501
|
if (Math.random() < hookErrorsSweepProbabilityFromEnv()) {
|
|
463
|
-
sweepHookErrorsDir(dir);
|
|
502
|
+
await sweepHookErrorsDir(dir);
|
|
464
503
|
}
|
|
465
504
|
} catch {
|
|
466
505
|
// Diagnostics are advisory and must never block or throw.
|
|
@@ -505,11 +544,11 @@ function payloadsDirFor(projectRoot) {
|
|
|
505
544
|
}
|
|
506
545
|
|
|
507
546
|
function schedulePayloadSweep(projectRoot) {
|
|
508
|
-
// Deferred: runs after the current turn settles, never inside the chain
|
|
547
|
+
// Deferred: runs after the current turn settles, never inside the chain
|
|
548
|
+
// request. Async fs only — a stalled mount must not freeze the host loop.
|
|
509
549
|
const timer = setTimeout(() => {
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
} catch { /* deferred sweeping is best effort */ }
|
|
550
|
+
maybeSweepStalePayloadsAsync(payloadsDirFor(projectRoot))
|
|
551
|
+
.catch(() => { /* deferred sweeping is best effort */ });
|
|
513
552
|
}, 0);
|
|
514
553
|
timer.unref?.();
|
|
515
554
|
}
|
|
@@ -530,7 +569,7 @@ export async function runScriptChain(
|
|
|
530
569
|
const scriptPaths = scripts.map((scriptName) => path.join(projectRoot, '.claude', 'hooks', scriptName));
|
|
531
570
|
const nodeExecutable = resolveNodeExecutable();
|
|
532
571
|
const startedAt = Date.now();
|
|
533
|
-
const payloadReference =
|
|
572
|
+
const payloadReference = await createPayloadReferenceAsync(JSON.stringify(payload), {
|
|
534
573
|
maxBytes: PAYLOAD_INLINE_MAX_BYTES,
|
|
535
574
|
dir: payloadsDirFor(projectRoot),
|
|
536
575
|
});
|
|
@@ -549,8 +588,8 @@ export async function runScriptChain(
|
|
|
549
588
|
// TASK-031: verify the staged payload survived the chain intact BEFORE removing it —
|
|
550
589
|
// a file that vanished or was truncated mid-flight means the scripts ran against a
|
|
551
590
|
// different payload than the host captured, so their verdicts are void.
|
|
552
|
-
payloadProbe =
|
|
553
|
-
payloadReference.
|
|
591
|
+
payloadProbe = await probePayloadIntegrityAsync(payloadReference);
|
|
592
|
+
await payloadReference.cleanupAsync();
|
|
554
593
|
if (payloadReference.mode === 'file') schedulePayloadSweep(projectRoot);
|
|
555
594
|
}
|
|
556
595
|
const elapsedMs = Date.now() - startedAt;
|
|
@@ -595,7 +634,7 @@ export async function runScriptChain(
|
|
|
595
634
|
parseError: parseError?.message || null,
|
|
596
635
|
wrapperError: chainResult?.wrapperError || null,
|
|
597
636
|
};
|
|
598
|
-
recordHookErrorDiagnostic(projectRoot, payload.session_id, diagnostic);
|
|
637
|
+
await recordHookErrorDiagnostic(projectRoot, payload.session_id, diagnostic);
|
|
599
638
|
// TASK-031: a staged payload that was lost or corrupted mid-chain voids every
|
|
600
639
|
// verdict below it. Chains that must not fail open on an unverifiable verdict
|
|
601
640
|
// (Edit|Write transport policy, and block-dangerous's never-fail-open rule)
|
|
@@ -626,6 +665,15 @@ export async function runScriptChain(
|
|
|
626
665
|
+ `(killed=${diagnostic.killed}, code=${diagnostic.code}, elapsedMs=${diagnostic.elapsedMs}, `
|
|
627
666
|
+ `runtime=${diagnostic.nodeExecutable}). No safety-gate verdict was available for [${scripts.join(', ')}]. `
|
|
628
667
|
+ `See .ukit/storage/cache/hook-errors/.${nodePathHint}`;
|
|
668
|
+
// TASK-002 (OMP-4): a transport failure whose stderr is a transient fs
|
|
669
|
+
// error produced no verdict — an infrastructure event, not a safety
|
|
670
|
+
// decision. Even on a fail-closed chain it degrades to a loud warning
|
|
671
|
+
// instead of a block showing raw errno text. A staged-payload integrity
|
|
672
|
+
// failure (payloadProbe !== null) already returned above and stays closed.
|
|
673
|
+
if (failClosedOnTransportError && TRANSIENT_INFRA_STDERR_RE.test(diagnostic.stderrExcerpt)) {
|
|
674
|
+
pi.logger?.warn?.(`[UKit] ${reason} Transient filesystem error — treated as "could not verify", not as a block.`);
|
|
675
|
+
return { block: false, context, invoked };
|
|
676
|
+
}
|
|
629
677
|
if (failClosedOnTransportError) {
|
|
630
678
|
return { block: true, reason, context, invoked };
|
|
631
679
|
}
|
|
@@ -767,17 +815,13 @@ export async function runToolResult(pi, event, { projectRoot, context: extension
|
|
|
767
815
|
toolUseId: event.toolCallId,
|
|
768
816
|
...metadata,
|
|
769
817
|
});
|
|
818
|
+
// OMP-3 (TASK-002): the receipt is journaled by exactly ONE path — the
|
|
819
|
+
// record-execution.mjs chain step inside runScriptChain (the same step Claude
|
|
820
|
+
// Code runs, carrying the runner's deadline/signal into the ledger lock). The
|
|
821
|
+
// former direct recordExecutionReceipt({harness:'omp'}) call here minted a
|
|
822
|
+
// second receipt per tool result, so one failed test counted as streak 2 and
|
|
823
|
+
// tripped the verification-loop blocker after a single failure.
|
|
770
824
|
const result = await runScriptChain(pi, scriptsForToolResult(toolName), payload, { projectRoot });
|
|
771
|
-
try {
|
|
772
|
-
await recordExecutionReceipt({
|
|
773
|
-
projectRoot,
|
|
774
|
-
payload,
|
|
775
|
-
toolName,
|
|
776
|
-
harness: 'omp',
|
|
777
|
-
});
|
|
778
|
-
} catch (error) {
|
|
779
|
-
pi.logger?.warn?.(`[UKit] execution receipt failed open: ${error?.message || error}`);
|
|
780
|
-
}
|
|
781
825
|
|
|
782
826
|
const hookOutput = result.context.join('\n').trim();
|
|
783
827
|
if (!hookOutput) return undefined;
|
|
@@ -849,19 +893,36 @@ export async function runSessionCompact(pi, event, { projectRoot, context: exten
|
|
|
849
893
|
// harnesses; a clean evaluation deletes it, so the count is consecutive crashes.
|
|
850
894
|
const COMPLETION_CRASH_STREAK_FILE = path.join('.ukit', 'storage', 'cache', 'completion-gate-crash.streak');
|
|
851
895
|
|
|
852
|
-
|
|
896
|
+
// OMP-1: the bridge is a long-lived in-proc module, so a streak that cannot persist
|
|
897
|
+
// (unwritable cache dir, stalled volume, lock-busy sibling writers) still advances
|
|
898
|
+
// in memory per project root. Pre-fix the catch path returned a constant 1 — the
|
|
899
|
+
// 3-crash loud release was unreachable and every omp Stop blocked forever. The
|
|
900
|
+
// in-memory count is the floor; a persisted count higher than it wins on the next
|
|
901
|
+
// successful read so cross-harness streaks still converge.
|
|
902
|
+
const crashStreakFallback = new Map();
|
|
903
|
+
|
|
904
|
+
async function bumpCompletionCrashStreak(projectRoot, pi = null) {
|
|
853
905
|
const file = path.join(projectRoot, COMPLETION_CRASH_STREAK_FILE);
|
|
854
906
|
try {
|
|
855
907
|
await fs.promises.mkdir(path.dirname(file), { recursive: true });
|
|
856
908
|
await fs.promises.appendFile(file, '.');
|
|
857
909
|
const buf = await fs.promises.readFile(file);
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
return
|
|
910
|
+
const count = Math.max(buf.length || 1, (crashStreakFallback.get(projectRoot) || 0) + 1);
|
|
911
|
+
crashStreakFallback.set(projectRoot, count);
|
|
912
|
+
return count;
|
|
913
|
+
} catch (error) {
|
|
914
|
+
const count = (crashStreakFallback.get(projectRoot) || 0) + 1;
|
|
915
|
+
crashStreakFallback.set(projectRoot, count);
|
|
916
|
+
pi?.logger?.warn?.(
|
|
917
|
+
`[UKit] completion crash streak could not persist (${error?.message || error}); `
|
|
918
|
+
+ `counting in memory (streak ${count}) so the breaker still advances.`,
|
|
919
|
+
);
|
|
920
|
+
return count;
|
|
861
921
|
}
|
|
862
922
|
}
|
|
863
923
|
|
|
864
924
|
async function resetCompletionCrashStreak(projectRoot) {
|
|
925
|
+
crashStreakFallback.delete(projectRoot);
|
|
865
926
|
try {
|
|
866
927
|
await fs.promises.rm(path.join(projectRoot, COMPLETION_CRASH_STREAK_FILE), { force: true });
|
|
867
928
|
} catch { /* best effort */ }
|
|
@@ -879,6 +940,38 @@ async function readStopGateMaxCrashStreaks(projectRoot) {
|
|
|
879
940
|
}
|
|
880
941
|
}
|
|
881
942
|
|
|
943
|
+
// ── TASK-007 (F-10/OMP-2): stop-path parity with stop-coordinator.mjs ─────────
|
|
944
|
+
// Claude's Stop runs one coordinator with three protections the omp bridge never
|
|
945
|
+
// had: the handoff-cursor lane (a non-done docs/AI_HANDOFF/RUN.md bounces the
|
|
946
|
+
// stop with the run's own Next: line, bounded by the stopGateMaxStalledBlocks
|
|
947
|
+
// liveness breaker), the 2s double-Stop dedupe window (one logical stop burns
|
|
948
|
+
// exactly one continuation), and a 600ms budget on every ledger/state lock.
|
|
949
|
+
// The evaluators are reused from stop-coordinator.mjs itself so semantics and
|
|
950
|
+
// the shared state file (.ukit/storage/cache/stop-coordinator/state.json) stay
|
|
951
|
+
// byte-identical across both engines.
|
|
952
|
+
const STOP_GATE_LOCK_BUDGET_MS = 600;
|
|
953
|
+
const HANDOFF_PROVENANCE_PREFIX = '[ukit-stop-coordinator] handoff-cursor also requested a block: ';
|
|
954
|
+
|
|
955
|
+
let stopCoordinatorModulePromise = null;
|
|
956
|
+
|
|
957
|
+
// Lazy + env-scrubbed: stop-coordinator.mjs arms a process.exit self-deadline at
|
|
958
|
+
// module top level whenever UKIT_HOOK_DEADLINE_MS is set. That env var is only
|
|
959
|
+
// ever set for spawned hook children, but the bridge is a long-lived in-proc
|
|
960
|
+
// module — a leaked value would kill the whole omp host. Same scrub the
|
|
961
|
+
// chain-runner applies around in-proc .mjs steps (hook-chain-runner.mjs:180).
|
|
962
|
+
function loadStopCoordinatorModule() {
|
|
963
|
+
if (!stopCoordinatorModulePromise) {
|
|
964
|
+
const scrubbed = 'UKIT_HOOK_DEADLINE_MS' in process.env
|
|
965
|
+
? process.env.UKIT_HOOK_DEADLINE_MS : undefined;
|
|
966
|
+
delete process.env.UKIT_HOOK_DEADLINE_MS;
|
|
967
|
+
stopCoordinatorModulePromise = import('../../../.claude/ukit/runtime/stop-coordinator.mjs')
|
|
968
|
+
.finally(() => {
|
|
969
|
+
if (scrubbed !== undefined) process.env.UKIT_HOOK_DEADLINE_MS = scrubbed;
|
|
970
|
+
});
|
|
971
|
+
}
|
|
972
|
+
return stopCoordinatorModulePromise;
|
|
973
|
+
}
|
|
974
|
+
|
|
882
975
|
export async function runSessionStop(
|
|
883
976
|
pi,
|
|
884
977
|
event,
|
|
@@ -887,10 +980,69 @@ export async function runSessionStop(
|
|
|
887
980
|
context: extensionContext = {},
|
|
888
981
|
state: suppliedState,
|
|
889
982
|
ledger: suppliedLedger,
|
|
983
|
+
now = Date.now(),
|
|
984
|
+
dedupeWindowMs,
|
|
985
|
+
lockBudgetMs = STOP_GATE_LOCK_BUDGET_MS,
|
|
890
986
|
},
|
|
891
987
|
) {
|
|
892
|
-
|
|
988
|
+
// TASK-007 (OMP-2d): the stop path keys its session identity ONLY from the
|
|
989
|
+
// event itself — never the context's sessionManager fallback. omp emits
|
|
990
|
+
// session_id on every top-level session_stop (verified against the bundled
|
|
991
|
+
// binary: emitSessionStop carries session_id/session_file/stop_hook_active),
|
|
992
|
+
// so an event without one is a nested/foreign stop; attributing it to the
|
|
993
|
+
// parent session would burn the parent's continuation budget and dedupe slot.
|
|
994
|
+
const metadata = {
|
|
995
|
+
sessionId: event?.sessionId ?? event?.session_id,
|
|
996
|
+
cwd: extensionContext?.cwd ?? event?.cwd,
|
|
997
|
+
transcriptPath: event?.transcriptPath ?? event?.transcript_path ?? event?.session_file,
|
|
998
|
+
};
|
|
893
999
|
const payload = buildHookPayload('Stop', metadata);
|
|
1000
|
+
|
|
1001
|
+
// TASK-007 (OMP-2b): the coordinator's once-per-stop dedupe. omp can deliver
|
|
1002
|
+
// two session_stop events for one logical stop; the second inside the window
|
|
1003
|
+
// is skipped entirely — it must not burn a second continuation.
|
|
1004
|
+
let coordinator = null;
|
|
1005
|
+
try {
|
|
1006
|
+
coordinator = await loadStopCoordinatorModule();
|
|
1007
|
+
} catch (error) {
|
|
1008
|
+
// Advisory lane: a missing/unloadable coordinator module must never wedge a
|
|
1009
|
+
// session — the completion gate below still owns the stop.
|
|
1010
|
+
pi.logger?.warn?.(`[UKit] stop-coordinator module unavailable (dedupe + handoff lanes skipped): ${error?.message || error}`);
|
|
1011
|
+
}
|
|
1012
|
+
if (coordinator) {
|
|
1013
|
+
const sessionKey = typeof payload.session_id === 'string' && payload.session_id
|
|
1014
|
+
? payload.session_id
|
|
1015
|
+
: (typeof payload.transcript_path === 'string' && payload.transcript_path ? payload.transcript_path : 'no-session');
|
|
1016
|
+
const duplicate = await coordinator.alreadyCoordinatedThisStop({
|
|
1017
|
+
projectRoot,
|
|
1018
|
+
sessionKey,
|
|
1019
|
+
now,
|
|
1020
|
+
dedupeWindowMs: dedupeWindowMs ?? coordinator.DEDUPE_WINDOW_MS,
|
|
1021
|
+
lockBudgetMs,
|
|
1022
|
+
});
|
|
1023
|
+
if (duplicate) return undefined;
|
|
1024
|
+
}
|
|
1025
|
+
|
|
1026
|
+
// TASK-007 (OMP-2c): the handoff-cursor lane. While docs/AI_HANDOFF/RUN.md
|
|
1027
|
+
// reports a phase outside {done, blocked} the run owns this stop — the
|
|
1028
|
+
// cursor's Next: line is the continuation instruction. Advisory on failure,
|
|
1029
|
+
// bounded by the stopGateMaxStalledBlocks breaker (always advances post-S3).
|
|
1030
|
+
let handoff = null;
|
|
1031
|
+
if (coordinator) {
|
|
1032
|
+
try {
|
|
1033
|
+
handoff = await coordinator.evaluateHandoffCursor({ projectRoot, now, lockBudgetMs });
|
|
1034
|
+
} catch (error) {
|
|
1035
|
+
pi.logger?.warn?.(`[UKit] handoff-cursor evaluator failed (advisory lane): ${error?.message || error}`);
|
|
1036
|
+
}
|
|
1037
|
+
}
|
|
1038
|
+
const handoffBlock = handoff?.kind === 'block' ? handoff.reason : null;
|
|
1039
|
+
// A breaker-release advisory (or any handoff systemMessage) is the omp
|
|
1040
|
+
// equivalent of the coordinator's systemMessage channel — surface it once,
|
|
1041
|
+
// visibly, whichever way the stop itself resolves.
|
|
1042
|
+
if (handoff?.kind === 'advisory' && handoff.systemMessage) {
|
|
1043
|
+
sendContext(pi, [handoff.systemMessage], 'nextTurn', { display: true });
|
|
1044
|
+
}
|
|
1045
|
+
|
|
894
1046
|
const state = suppliedState ?? await readRouteState(projectRoot, payload);
|
|
895
1047
|
const ledger = suppliedLedger ?? await readExecutionLedger(projectRoot, payload) ?? {};
|
|
896
1048
|
let evaluation;
|
|
@@ -899,8 +1051,7 @@ export async function runSessionStop(
|
|
|
899
1051
|
} catch (error) {
|
|
900
1052
|
// BUG-C22-05: fail-closed like the shell gate (continue the turn), but only up
|
|
901
1053
|
// to the configured consecutive-crash cap — a persistently broken evaluator
|
|
902
|
-
|
|
903
|
-
const streak = await bumpCompletionCrashStreak(projectRoot);
|
|
1054
|
+
const streak = await bumpCompletionCrashStreak(projectRoot, pi);
|
|
904
1055
|
const maxStreak = await readStopGateMaxCrashStreaks(projectRoot);
|
|
905
1056
|
if (streak > maxStreak) {
|
|
906
1057
|
await resetCompletionCrashStreak(projectRoot);
|
|
@@ -917,7 +1068,8 @@ export async function runSessionStop(
|
|
|
917
1068
|
continue: true,
|
|
918
1069
|
additionalContext:
|
|
919
1070
|
`UKit stop coordinator: infrastructure failure while evaluating this stop (crash streak ${streak}/${maxStreak}) — `
|
|
920
|
-
+ 'the stop is blocked fail-closed (details withheld). Run: ukit install, then re-send the task in a new message.'
|
|
1071
|
+
+ 'the stop is blocked fail-closed (details withheld). Run: ukit install, then re-send the task in a new message.'
|
|
1072
|
+
+ (handoffBlock ? `\n${HANDOFF_PROVENANCE_PREFIX}${handoffBlock}` : ''),
|
|
921
1073
|
};
|
|
922
1074
|
}
|
|
923
1075
|
// A clean evaluation resets the consecutive-crash streak (shared with the shell hook).
|
|
@@ -940,20 +1092,33 @@ export async function runSessionStop(
|
|
|
940
1092
|
// message) stays as a second guaranteed channel.
|
|
941
1093
|
sendContext(pi, [notice], 'nextTurn', { display: true });
|
|
942
1094
|
}
|
|
1095
|
+
// The completion gate released, but an in-flight handoff run still owns the
|
|
1096
|
+
// stop (coordinator merge order: handoff-cursor block outranks a released
|
|
1097
|
+
// completion). Its own stall streak — not the ledger continuation counter —
|
|
1098
|
+
// bounds this lane, so no continuation bookkeeping runs here.
|
|
1099
|
+
if (handoffBlock) {
|
|
1100
|
+
return { continue: true, additionalContext: handoffBlock };
|
|
1101
|
+
}
|
|
943
1102
|
return undefined;
|
|
944
1103
|
}
|
|
945
1104
|
|
|
946
1105
|
if (suppliedLedger === undefined) {
|
|
947
1106
|
try {
|
|
948
|
-
|
|
949
|
-
|
|
1107
|
+
// TASK-007 (OMP-2a): the 600ms lock budget — a contended ledger lock must
|
|
1108
|
+
// never hold the omp host near its hook deadline.
|
|
1109
|
+
if (evaluation.finalNotice) await markNotified(projectRoot, payload, ledger, { deadlineMs: lockBudgetMs });
|
|
1110
|
+
else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null, evidencePromptKey(state), { deadlineMs: lockBudgetMs });
|
|
950
1111
|
} catch (error) {
|
|
951
1112
|
pi.logger?.warn?.(`[UKit] continuation bookkeeping failed open: ${error?.message || error}`);
|
|
952
1113
|
}
|
|
953
1114
|
}
|
|
954
1115
|
return {
|
|
955
1116
|
continue: true,
|
|
956
|
-
|
|
1117
|
+
// Coordinator merge order: the completion gate owns the continuation, and a
|
|
1118
|
+
// losing handoff-cursor block travels inside the winning reason.
|
|
1119
|
+
additionalContext: handoffBlock
|
|
1120
|
+
? `${evaluation.reason}\n${HANDOFF_PROVENANCE_PREFIX}${handoffBlock}`
|
|
1121
|
+
: evaluation.reason,
|
|
957
1122
|
};
|
|
958
1123
|
}
|
|
959
1124
|
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
Status values (xem RULES.md §Status state machine):
|
|
5
5
|
ready | in_progress | pending_review | changes_requested | critical_block | approved | approved_minor | blocked | done
|
|
6
6
|
|
|
7
|
-
Owner = tool đang giữ task: claude-code |
|
|
7
|
+
Owner = tool đang giữ task: claude-code | codex | omp | -
|
|
8
8
|
-->
|
|
9
9
|
|
|
10
10
|
| ID | Title | Priority | Size | Status | Owner | Reviewer | File |
|
|
@@ -92,21 +92,21 @@ Next: <bước kế tiếp chính xác>
|
|
|
92
92
|
|
|
93
93
|
## Handoff Flow (tool-agnostic, file-based state machine)
|
|
94
94
|
|
|
95
|
-
UKit handoff hoạt động qua **file state**. Anh tự chọn tool nào cho từng phase — Claude Code /
|
|
95
|
+
UKit handoff hoạt động qua **file state**. Anh tự chọn tool nào cho từng phase — Claude Code / Codex / omp / tool mới sau này — đều được. UKit chỉ care về **role của model**, không care tool.
|
|
96
96
|
|
|
97
97
|
3 phase × 3 role model:
|
|
98
98
|
|
|
99
99
|
- **Plan** — model mạnh nhất anh có (reasoning model). Có thể chạy ở bất kỳ tool nào hỗ trợ planning tốt.
|
|
100
|
-
- **Execute** — model rẻ-mà-vẫn-thông-minh (code model). Có thể là subagent code của
|
|
101
|
-
- **Review** — **MODEL KHÁC executor** (reasoning model thường tốt hơn). Có thể là tool khác, hoặc cùng tool nhưng subagent khác model
|
|
100
|
+
- **Execute** — model rẻ-mà-vẫn-thông-minh (code model). Có thể là subagent code của omp, hay feature-implementer của Claude Code.
|
|
101
|
+
- **Review** — **MODEL KHÁC executor** (reasoning model thường tốt hơn). Có thể là tool khác, hoặc cùng tool nhưng subagent khác model.
|
|
102
102
|
|
|
103
103
|
Hai mô hình triển khai đều hợp lệ:
|
|
104
|
-
- **Cross-tool**: ví dụ Claude (plan) →
|
|
105
|
-
- **Same-tool different-subagent**: ví dụ
|
|
104
|
+
- **Cross-tool**: ví dụ Claude (plan) → Codex (execute) → Claude (review). Bridge qua file.
|
|
105
|
+
- **Same-tool different-subagent**: ví dụ omp:plan → omp:task → omp:code-reviewer, miễn 3 subagent dùng MODEL khác nhau ở role tương ứng.
|
|
106
106
|
|
|
107
107
|
Mỗi tool/subagent đọc cùng `INDEX.md` + `tasks/TASK-xxx.md` → chọn task theo `status` → cập nhật status khi xong.
|
|
108
108
|
|
|
109
|
-
> **Quan trọng — UKit không enforce model:** `handoff.executor.cheapSmartModelHint` và `handoff.reviewer.model` trong `.ukit/storage/config.json` chỉ là **nhãn** để anh biết MUỐN dùng gì. Tool nào dùng model nào là do anh chọn trong settings của tool đó. UKit enforce contract bằng cách bắt executor TỰ KHAI `EXECUTOR_MODEL` trong Executor Report; reviewer so với chính nó và refuse nếu trùng. Vì vậy nếu
|
|
109
|
+
> **Quan trọng — UKit không enforce model:** `handoff.executor.cheapSmartModelHint` và `handoff.reviewer.model` trong `.ukit/storage/config.json` chỉ là **nhãn** để anh biết MUỐN dùng gì. Tool nào dùng model nào là do anh chọn trong settings của tool đó. UKit enforce contract bằng cách bắt executor TỰ KHAI `EXECUTOR_MODEL` trong Executor Report; reviewer so với chính nó và refuse nếu trùng. Vì vậy nếu anh để cả executor-subagent và review-subagent đều dùng cùng model → reviewer sẽ tự refuse, không silent-pass.
|
|
110
110
|
|
|
111
111
|
### Status state machine
|
|
112
112
|
|
|
@@ -442,7 +442,7 @@
|
|
|
442
442
|
"field": "handoff.reviewer.model",
|
|
443
443
|
"mac_dinh": "unic-smart",
|
|
444
444
|
"y_nghia": "Model dùng cho reviewer agent ở Phase 3. BẮT BUỘC khác model executor để bắt được lỗi mà executor miss. Có thể dùng claude-opus-5, unic-smart, hoặc bất kỳ model reasoning mạnh nào.",
|
|
445
|
-
"vi_du": "Nếu executor là unic-code
|
|
445
|
+
"vi_du": "Nếu executor là unic-code, set reviewer.model=unic-smart hoặc claude-opus-5. Nếu executor là claude-sonnet, set reviewer thành claude-opus."
|
|
446
446
|
},
|
|
447
447
|
"tat_reviewer_phase": {
|
|
448
448
|
"field": "handoff.reviewer.enabled",
|
|
@@ -601,7 +601,7 @@
|
|
|
601
601
|
},
|
|
602
602
|
"handoff": {
|
|
603
603
|
"enabled": "Bật Quality Gate cho handoff: plan có Test Plan, executor test-first, reviewer model khác. Tắt = quay về flow cũ (dễ lọt lỗi vặt).",
|
|
604
|
-
"crossTool": "true nghĩa là handoff truyền qua file (PLAN/INDEX/tasks) chứ không qua in-process subagent — cho phép plan ở
|
|
604
|
+
"crossTool": "true nghĩa là handoff truyền qua file (PLAN/INDEX/tasks) chứ không qua in-process subagent — cho phép plan ở một tool, execute ở tool khác, review ở tool thứ ba khác model.",
|
|
605
605
|
"maxParallelAgents": "Số agent chạy song song TỐI ĐA trong một wave (mặc định 2 để giữ session chính nhẹ và hạn chế rủi ro compact/worktree rác; nếu máy khỏe và task độc lập nhiều thì có thể nâng dần 3-5, tối đa ~10-15). Một wave có nhiều task hơn số này sẽ được chia thành nhiều batch chạy lần lượt — áp dụng cho cả Phase 3 Implement và Phase 4 Review. Lý do giới hạn vẫn còn: mỗi agent nền có context window riêng, và report của agent khi xong sẽ được inject ngược vào session chính — chạy quá nhiều cùng lúc (vd 20+) vẫn có thể làm session chính vượt context window và bỏ lại worktree rác. Hạ xuống 1-2 nếu task nặng (verification output dài) hoặc thấy compact bị trigger liên tục.",
|
|
606
606
|
"plan": {
|
|
607
607
|
"requireTestPlan": "Bắt buộc PLAN.md §4 phải có Test Plan trước khi task chuyển ready.",
|