@ngockhoale/ukit 2.7.6 → 2.7.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +105 -0
- package/package.json +1 -1
- package/scripts/install/sync-installed-mirror.mjs +250 -0
- package/scripts/perf/audit-perf.mjs +287 -36
- package/scripts/perf/diff-perf-findings.mjs +136 -0
- package/scripts/perf/perf-findings.json +260 -206
- package/scripts/perf/perf-measure.md +271 -0
- package/src/context/detectProjectContext.js +5 -0
- package/src/core/codeintel/invalidation.js +4 -0
- package/src/core/fileOps.js +40 -117
- package/src/core/hookChainDoctor.js +65 -2
- package/src/core/memory/store.js +22 -1
- package/src/core/taskBudgetValidator.js +9 -6
- package/src/core/unattendedDoctor.js +8 -1
- package/src/render/buildVariables.js +10 -0
- package/templates/.claude/agents/bug-debugger.md +1 -1
- package/templates/.claude/agents/feature-implementer.md +2 -2
- package/templates/.claude/commands/ukit/handoff-create.md +1 -1
- package/templates/.claude/commands/ukit/handoff-fullstack.md +1 -1
- package/templates/.claude/commands/ukit/handoff-implement.md +1 -1
- package/templates/.claude/commands/ukit/handoff-review.md +1 -1
- package/templates/.claude/hooks/auto-prune-bash.sh +19 -0
- package/templates/.claude/hooks/context-hardcap-gate.sh +4 -1
- package/templates/.claude/hooks/handoff-model-guard.sh +22 -11
- package/templates/.claude/hooks/reinject-context.sh +22 -0
- package/templates/.claude/hooks/reset-compact-pressure.sh +29 -0
- package/templates/.claude/hooks/session-episode.sh +20 -0
- package/templates/.claude/hooks/skill-router.sh +15 -8
- package/templates/.claude/hooks/verification-guard.sh +3 -0
- package/templates/.claude/ukit/index/route-task.mjs +237 -32
- package/templates/.claude/ukit/index/task-budget-validator.mjs +6 -2
- package/templates/.claude/ukit/runtime/async-lock.mjs +144 -10
- package/templates/.claude/ukit/runtime/compact-threshold.mjs +5 -2
- package/templates/.claude/ukit/runtime/execution-ledger.mjs +217 -17
- package/templates/.claude/ukit/runtime/hook-chain-runner.mjs +156 -24
- package/templates/.claude/ukit/runtime/hook-payload-store.mjs +57 -0
- package/templates/.claude/ukit/runtime/hook-telemetry.mjs +84 -12
- package/templates/.claude/ukit/runtime/hook-telemetry.sh +50 -0
- package/templates/.claude/ukit/runtime/stop-coordinator.mjs +35 -20
- package/templates/.claude/ukit/runtime/token-utils.mjs +37 -126
- package/templates/.codex/settings.json +1 -5
- package/templates/.omp/agents/bug-debugger.md +1 -1
- package/templates/.omp/agents/feature-implementer.md +2 -2
- package/templates/.omp/hooks/pre/ukit-bridge.js +157 -26
- package/templates/docs/AI_HANDOFF/INDEX.md +1 -1
- package/templates/docs/AI_HANDOFF/RULES.md +6 -6
- package/templates/ukit/storage/config.json +2 -2
|
@@ -294,7 +294,8 @@ export async function readRouteState(projectRoot, payload = {}) {
|
|
|
294
294
|
}
|
|
295
295
|
|
|
296
296
|
export async function readExecutionLedger(projectRoot, payload = {}) {
|
|
297
|
-
|
|
297
|
+
const target = ledgerPath(projectRoot, payload);
|
|
298
|
+
return mergeContinuationSidecar(await readJson(target, null), target);
|
|
298
299
|
}
|
|
299
300
|
|
|
300
301
|
function promptKeyFromText(promptText) {
|
|
@@ -313,7 +314,7 @@ function hasUnfinishedCompletion(state = {}, ledger = {}) {
|
|
|
313
314
|
if (!IMPLEMENT_MODES.has(mode)) return false;
|
|
314
315
|
const required = requiredEvidence(state);
|
|
315
316
|
if (required.length === 0) return false;
|
|
316
|
-
return required.some((item) => !evidenceSatisfied(item, ledger, state
|
|
317
|
+
return required.some((item) => !evidenceSatisfied(item, ledger, state));
|
|
317
318
|
}
|
|
318
319
|
|
|
319
320
|
function resumeSessionHash(sessionId) {
|
|
@@ -400,9 +401,38 @@ function explicitError(payload = {}) {
|
|
|
400
401
|
payload.tool_result?.is_error,
|
|
401
402
|
payload.tool_response?.isError,
|
|
402
403
|
payload.tool_response?.is_error,
|
|
404
|
+
// omp tool details carry their own error flag (eval cells, bash timeouts).
|
|
405
|
+
payload.details?.isError,
|
|
406
|
+
payload.tool_output?.details?.isError,
|
|
407
|
+
payload.tool_result?.details?.isError,
|
|
403
408
|
].some((value) => value === true);
|
|
404
409
|
}
|
|
405
410
|
|
|
411
|
+
// omp tool results do not all spell the exit code the same way: bash details
|
|
412
|
+
// carry `exitCode` (non-zero only), other producers serialize `exit_code`, and
|
|
413
|
+
// eval results report per-cell codes under `details.cells[].exitCode`. A missed
|
|
414
|
+
// shape used to journal a failed verification as exitCode null → success, so
|
|
415
|
+
// evidence never satisfied and the Stop gate bounced to the cap.
|
|
416
|
+
function detailsExitCode(details) {
|
|
417
|
+
if (!details || typeof details !== 'object') return null;
|
|
418
|
+
for (const value of [details.exitCode, details.exit_code]) {
|
|
419
|
+
// Number(null) === 0 — a JSON `"exitCode": null` must fall through to the
|
|
420
|
+
// next probe (exit_code, then cells[]), never journal as a successful exit.
|
|
421
|
+
if (value === undefined || value === null) continue;
|
|
422
|
+
const number = Number(value);
|
|
423
|
+
if (Number.isFinite(number)) return number;
|
|
424
|
+
}
|
|
425
|
+
const cells = Array.isArray(details.cells) ? details.cells : [];
|
|
426
|
+
let lastCode = null;
|
|
427
|
+
for (const cell of cells) {
|
|
428
|
+
const number = Number(cell?.exitCode ?? cell?.exit_code);
|
|
429
|
+
if (!Number.isFinite(number)) continue;
|
|
430
|
+
if (number !== 0) return number;
|
|
431
|
+
lastCode = number;
|
|
432
|
+
}
|
|
433
|
+
return lastCode;
|
|
434
|
+
}
|
|
435
|
+
|
|
406
436
|
function extractExitCode(payload = {}) {
|
|
407
437
|
const candidates = [
|
|
408
438
|
payload.tool_output?.exitCode,
|
|
@@ -415,8 +445,20 @@ function extractExitCode(payload = {}) {
|
|
|
415
445
|
payload.tool_response?.exit_code,
|
|
416
446
|
payload.exitCode,
|
|
417
447
|
payload.exit_code,
|
|
448
|
+
// omp details shapes: the bridge mirrors event.details under BOTH
|
|
449
|
+
// tool_output.details and tool_result.details, and eval results nest the
|
|
450
|
+
// real code under details.cells[].
|
|
451
|
+
detailsExitCode(payload.tool_output?.details),
|
|
452
|
+
detailsExitCode(payload.tool_result?.details),
|
|
453
|
+
detailsExitCode(payload.details),
|
|
454
|
+
detailsExitCode(payload.tool_response?.details),
|
|
455
|
+
detailsExitCode(payload.tool_output),
|
|
456
|
+
detailsExitCode(payload.tool_result),
|
|
418
457
|
];
|
|
419
458
|
for (const value of candidates) {
|
|
459
|
+
// Number(null) === 0 — a JSON `"exitCode": null` (or a details probe that
|
|
460
|
+
// found nothing) must read as "no code", never as a successful exit.
|
|
461
|
+
if (value === undefined || value === null) continue;
|
|
420
462
|
const number = Number(value);
|
|
421
463
|
if (Number.isFinite(number)) return number;
|
|
422
464
|
}
|
|
@@ -801,6 +843,96 @@ function journalQuarantinePathFor(target) {
|
|
|
801
843
|
return `${journalPathFor(target)}.quarantine`;
|
|
802
844
|
}
|
|
803
845
|
|
|
846
|
+
// F-4: the continuation counter is STATE, not a journal row. When the journal is
|
|
847
|
+
// full (or its lock unavailable) a continuation/notified event used to be rejected
|
|
848
|
+
// outright, so MAX_CONTINUATIONS fired several bounces late — or never while the
|
|
849
|
+
// ledger lock stayed contended. Rejected counter events land in this small sidecar
|
|
850
|
+
// instead; readExecutionLedger merges it into the view every evaluator sees, and
|
|
851
|
+
// the next committed ledger write folds it in and removes it.
|
|
852
|
+
function continuationSidecarPathFor(target) {
|
|
853
|
+
return `${target}.continuations.json`;
|
|
854
|
+
}
|
|
855
|
+
|
|
856
|
+
// Overlay sidecar counter state onto the ledger when the sidecar is newer than the
|
|
857
|
+
// ledger's last continuation. A sidecar at or behind the ledger is already folded
|
|
858
|
+
// (or written for a superseded request) and is ignored — never double-counted.
|
|
859
|
+
// When the main file is absent the sidecar IS the counter state: returning null
|
|
860
|
+
// hid every tick from evaluators and the next committed write deleted the sidecar
|
|
861
|
+
// unread, so the cap fired late. A sidecar without counter state still yields null
|
|
862
|
+
// — the no-ledger contract is preserved.
|
|
863
|
+
async function mergeContinuationSidecar(ledger, target) {
|
|
864
|
+
let sidecar = null;
|
|
865
|
+
try {
|
|
866
|
+
sidecar = JSON.parse(await fs.readFile(continuationSidecarPathFor(target), 'utf8'));
|
|
867
|
+
} catch {
|
|
868
|
+
return ledger;
|
|
869
|
+
}
|
|
870
|
+
if (!sidecar || typeof sidecar !== 'object') return ledger;
|
|
871
|
+
const sidecarAt = Number(sidecar.lastContinuationAt) || 0;
|
|
872
|
+
if (!ledger) {
|
|
873
|
+
const hasState = sidecarAt > 0
|
|
874
|
+
|| Number(sidecar.continuationCount) > 0
|
|
875
|
+
|| sidecar.notified === true;
|
|
876
|
+
if (!hasState) return null;
|
|
877
|
+
return {
|
|
878
|
+
requestKey: sidecar.requestKey ?? null,
|
|
879
|
+
promptKey: sidecar.promptKey ?? null,
|
|
880
|
+
continuationRequestKey: sidecar.requestKey ?? null,
|
|
881
|
+
continuationCount: Number(sidecar.continuationCount) || 0,
|
|
882
|
+
noProgressCount: Number(sidecar.noProgressCount) || 0,
|
|
883
|
+
lastProgressDigest: sidecar.lastProgressDigest ?? null,
|
|
884
|
+
notified: sidecar.notified === true,
|
|
885
|
+
lastContinuationAt: sidecarAt,
|
|
886
|
+
updatedAt: Number(sidecar.updatedAt) || sidecarAt,
|
|
887
|
+
};
|
|
888
|
+
}
|
|
889
|
+
const ledgerAt = Number(ledger.lastContinuationAt) || 0;
|
|
890
|
+
if (!(sidecarAt > ledgerAt)) return ledger;
|
|
891
|
+
return {
|
|
892
|
+
...ledger,
|
|
893
|
+
continuationCount: Number(sidecar.continuationCount) || 0,
|
|
894
|
+
noProgressCount: Number(sidecar.noProgressCount) || 0,
|
|
895
|
+
continuationRequestKey: sidecar.requestKey ?? ledger.continuationRequestKey ?? null,
|
|
896
|
+
lastProgressDigest: sidecar.lastProgressDigest ?? ledger.lastProgressDigest ?? null,
|
|
897
|
+
notified: ledger.notified === true || sidecar.notified === true,
|
|
898
|
+
lastContinuationAt: sidecarAt,
|
|
899
|
+
updatedAt: Math.max(Number(ledger.updatedAt) || 0, sidecarAt),
|
|
900
|
+
};
|
|
901
|
+
}
|
|
902
|
+
|
|
903
|
+
// Best-effort counter tick for a continuation/notified event the journal rejected.
|
|
904
|
+
// Reads the merged ledger view (main file + existing sidecar) so consecutive ticks
|
|
905
|
+
// accumulate even while the main lock stays busy. Returns { ok } — a sidecar write
|
|
906
|
+
// failure means the tick is genuinely lost and the caller reports the rejection.
|
|
907
|
+
async function bumpContinuationSidecar(target, event) {
|
|
908
|
+
try {
|
|
909
|
+
const base = (await mergeContinuationSidecar(await readJson(target, null), target)) || {};
|
|
910
|
+
const applied = event.type === 'notified'
|
|
911
|
+
? { ...base, notified: true, updatedAt: Date.now() }
|
|
912
|
+
: applyContinuationToLedger(base, event);
|
|
913
|
+
// The merge gate requires sidecarAt > ledgerAt. A notified tick inherits the
|
|
914
|
+
// base stamp unchanged, so it must be bumped past the base it was computed
|
|
915
|
+
// from — otherwise the tick is invisible to every evaluator and the cap's
|
|
916
|
+
// final notice loops forever while contention persists. The +1 floor also
|
|
917
|
+
// covers a continuation landing in the same millisecond as the base stamp.
|
|
918
|
+
const baseAt = Number(base.lastContinuationAt) || 0;
|
|
919
|
+
const stampedAt = Math.max(Number(applied.lastContinuationAt) || 0, baseAt + 1);
|
|
920
|
+
await writeJsonAtomic(continuationSidecarPathFor(target), {
|
|
921
|
+
requestKey: event?.requestKey ?? applied.requestKey ?? applied.continuationRequestKey ?? null,
|
|
922
|
+
promptKey: event?.promptKey ?? applied.promptKey ?? null,
|
|
923
|
+
continuationCount: Number(applied.continuationCount) || 0,
|
|
924
|
+
noProgressCount: Number(applied.noProgressCount) || 0,
|
|
925
|
+
lastProgressDigest: applied.lastProgressDigest ?? null,
|
|
926
|
+
notified: applied.notified === true,
|
|
927
|
+
lastContinuationAt: stampedAt,
|
|
928
|
+
updatedAt: Date.now(),
|
|
929
|
+
});
|
|
930
|
+
return { ok: true };
|
|
931
|
+
} catch {
|
|
932
|
+
return { ok: false };
|
|
933
|
+
}
|
|
934
|
+
}
|
|
935
|
+
|
|
804
936
|
function newEventId() {
|
|
805
937
|
return `${Date.now().toString(36)}-${crypto.randomBytes(8).toString('hex')}`;
|
|
806
938
|
}
|
|
@@ -1214,6 +1346,11 @@ export async function recordLedgerEvent(event, {
|
|
|
1214
1346
|
// Consume the journal only after the ledger write landed: a crash before this line
|
|
1215
1347
|
// leaves the journal intact and the next drain re-applies idempotently by eventId.
|
|
1216
1348
|
if (drained.commit) await drained.commit();
|
|
1349
|
+
// The continuation sidecar was already merged into `current` (and therefore into
|
|
1350
|
+
// the ledger just written): its ticks are now durable state, so the sidecar is
|
|
1351
|
+
// retired. A tick landing between the read and this rm is lost — same accepted
|
|
1352
|
+
// window as the journal drain — but the breaker still advances.
|
|
1353
|
+
try { await fs.rm(continuationSidecarPathFor(target), { force: true }); } catch {}
|
|
1217
1354
|
return { committed: true, eventId, value, drained: drained.applied, quarantined: drained.quarantined };
|
|
1218
1355
|
});
|
|
1219
1356
|
if (outcome.ok) {
|
|
@@ -1228,9 +1365,16 @@ export async function recordLedgerEvent(event, {
|
|
|
1228
1365
|
// NEVER mutated unlocked — journal the event once; the next acquired lock reconciles it.
|
|
1229
1366
|
const record = buildJournalRecord({ event, eventId, payload, routeState: routeState || null });
|
|
1230
1367
|
const journalResult = await appendJournalRecord(target, record, { signal, deadlineMs });
|
|
1231
|
-
|
|
1232
|
-
|
|
1233
|
-
|
|
1368
|
+
if (journalResult.ok) return { journaled: true, eventId };
|
|
1369
|
+
// F-4: continuation/notified are counter state, not journal rows — a journal-full
|
|
1370
|
+
// (or journal-unavailable) rejection must still advance the counter or
|
|
1371
|
+
// MAX_CONTINUATIONS fires late/never. Land the tick in the sidecar the evaluator
|
|
1372
|
+
// merges; only a sidecar write failure is a real rejection.
|
|
1373
|
+
if (event.type === 'continuation' || event.type === 'notified') {
|
|
1374
|
+
const sidecar = await bumpContinuationSidecar(target, event);
|
|
1375
|
+
if (sidecar.ok) return { counted: true, eventId, reason: journalResult.reason };
|
|
1376
|
+
}
|
|
1377
|
+
return { rejected: true, eventId, reason: journalResult.reason };
|
|
1234
1378
|
}
|
|
1235
1379
|
|
|
1236
1380
|
// TASK-027: the receipt is classified (kind, file, command, targeted/broad scope) from the
|
|
@@ -1291,6 +1435,44 @@ export async function recordExecutionReceipt({
|
|
|
1291
1435
|
);
|
|
1292
1436
|
}
|
|
1293
1437
|
|
|
1438
|
+
// F-15: mirror of isExplicitBroadVerificationRequest in
|
|
1439
|
+
// templates/.claude/hooks/verification-guard.sh — the guard uses this prompt signal to
|
|
1440
|
+
// let a broad suite run under a targeted policy, so the completion gate must honor the
|
|
1441
|
+
// same signal when it classifies the resulting broad receipt. Keep both lists in sync.
|
|
1442
|
+
function isExplicitBroadVerificationRequest(promptText) {
|
|
1443
|
+
const text = String(promptText || '').toLowerCase();
|
|
1444
|
+
return [
|
|
1445
|
+
/run (the )?full test suite/,
|
|
1446
|
+
/run all tests/,
|
|
1447
|
+
/full verification/,
|
|
1448
|
+
/verify everything/,
|
|
1449
|
+
/run the whole suite/,
|
|
1450
|
+
/run broad verification/,
|
|
1451
|
+
/full suite/,
|
|
1452
|
+
/chạy (?:toàn bộ|full) test/,
|
|
1453
|
+
/chạy hết test/,
|
|
1454
|
+
/chạy full test suite/,
|
|
1455
|
+
/verify toàn bộ/,
|
|
1456
|
+
/kiểm tra toàn bộ/,
|
|
1457
|
+
// Plan-driven execution prompts often authorize the plan's final broad
|
|
1458
|
+
// verification without spelling it as "full test suite".
|
|
1459
|
+
/(?:run|execute|implement|do)\b.*\bfull\b.*\b(?:prd\s+)?plan\b/,
|
|
1460
|
+
/\bfull\b.*\b(?:prd\s+)?plan\b.*\b(?:verify|verification|test|tests)\b/,
|
|
1461
|
+
/(?:chạy|lam|làm|thực hiện|triển khai)\b.*\bfull\b.*\b(?:prd|plan|ke hoach|kế hoạch)\b/,
|
|
1462
|
+
/\bfull\b.*\b(?:prd|plan|ke hoach|kế hoạch)\b.*\b(?:verify|verification|test|tests|kiểm tra)\b/,
|
|
1463
|
+
].some((pattern) => pattern.test(text));
|
|
1464
|
+
}
|
|
1465
|
+
|
|
1466
|
+
// The route cache strips prompt text (route-task.mjs *ForCache deletes it), so the live
|
|
1467
|
+
// state is the only place the explicit-broad signal survives for the evaluator.
|
|
1468
|
+
function explicitBroadVerificationRequested(state = {}) {
|
|
1469
|
+
const routingContext = state?.routingContext || {};
|
|
1470
|
+
return [
|
|
1471
|
+
routingContext.lastExplicitUserPromptText,
|
|
1472
|
+
routingContext.promptText,
|
|
1473
|
+
].some((promptText) => isExplicitBroadVerificationRequest(promptText));
|
|
1474
|
+
}
|
|
1475
|
+
|
|
1294
1476
|
function requiredEvidence(state = {}) {
|
|
1295
1477
|
const routeSummary = state?.routeSummary || {};
|
|
1296
1478
|
const contractEvidence = routeSummary.executionContract?.completionEvidence;
|
|
@@ -1300,14 +1482,19 @@ function requiredEvidence(state = {}) {
|
|
|
1300
1482
|
return [...new Set(routeSummary.completionState?.missingEvidence || [])];
|
|
1301
1483
|
}
|
|
1302
1484
|
|
|
1303
|
-
function evidenceSatisfied(evidence, ledger = {},
|
|
1485
|
+
function evidenceSatisfied(evidence, ledger = {}, state = {}) {
|
|
1486
|
+
const routeSummary = state?.routeSummary || {};
|
|
1304
1487
|
if (evidence === 'write-evidence') return ledger.writeSucceeded === true;
|
|
1305
1488
|
if (evidence === 'verification-evidence') {
|
|
1306
1489
|
// WS-C: when the route names concrete verification commands, only a receipt that ran
|
|
1307
1490
|
// one of them counts — an unrelated `yarn test` no longer satisfies the gate.
|
|
1491
|
+
// F-15: unless the user explicitly requested/approved a broad suite — the same
|
|
1492
|
+
// prompt signal verification-guard.sh honors — in which case a passing
|
|
1493
|
+
// scope='broad' receipt (verificationSucceeded) satisfies the gate too.
|
|
1308
1494
|
const routedCommands = routedVerificationCommands(routeSummary);
|
|
1309
1495
|
if (routedCommands.length > 0) {
|
|
1310
|
-
return ledger.targetedVerificationSucceeded === true
|
|
1496
|
+
return ledger.targetedVerificationSucceeded === true
|
|
1497
|
+
|| (explicitBroadVerificationRequested(state) && ledger.verificationSucceeded === true);
|
|
1311
1498
|
}
|
|
1312
1499
|
return ledger.verificationSucceeded === true;
|
|
1313
1500
|
}
|
|
@@ -1433,7 +1620,7 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
|
1433
1620
|
|| ledger.requestKey === state.requestKey
|
|
1434
1621
|
|| (ledger.promptKey && evidencePromptKey(state) === ledger.promptKey);
|
|
1435
1622
|
const effectiveLedger = sameRequest ? ledger : {};
|
|
1436
|
-
const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger,
|
|
1623
|
+
const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger, state));
|
|
1437
1624
|
if (missingEvidence.length === 0) {
|
|
1438
1625
|
// Silent success: the route is present and every required evidence is satisfied. Marked
|
|
1439
1626
|
// `complete` so the CLI dispatch recognizes it BEFORE the loud final else — otherwise a
|
|
@@ -1609,12 +1796,19 @@ async function readStdin() {
|
|
|
1609
1796
|
|
|
1610
1797
|
// Malformed hook input crashes the evaluate BEFORE any route/session can be read, so the only
|
|
1611
1798
|
// stable identity is the project root. Count consecutive crashes there; past MAX_GATE_CRASHES,
|
|
1612
|
-
// release LOUD naming the crash and the remedy instead of blocking a corrupt hook forever.
|
|
1613
|
-
//
|
|
1614
|
-
//
|
|
1799
|
+
// release LOUD naming the crash and the remedy instead of blocking a corrupt hook forever. A
|
|
1800
|
+
// successful evaluate resets the count.
|
|
1801
|
+
//
|
|
1802
|
+
// F-3: the counter exists to break the infinite bounce — so losing the count must
|
|
1803
|
+
// never BECOME the infinite bounce. When the counter write fails, fail OPEN (loud
|
|
1804
|
+
// release) instead of emitting an unconditional block that can never reach the cap.
|
|
1615
1805
|
async function handleEvaluateCrash(projectRoot, error) {
|
|
1616
1806
|
const detail = `malformed hook input crashed the completion gate (${error?.message || error})`;
|
|
1617
1807
|
const blockReason = () => `${detail}. This is the UKit completion gate, not the task — re-send the task in a new message so the hook payload is regenerated.`;
|
|
1808
|
+
const loudRelease = (message) => {
|
|
1809
|
+
process.stderr.write(`[ukit-completion] ${message}\n`);
|
|
1810
|
+
process.stdout.write(`${JSON.stringify({ systemMessage: message })}\n`);
|
|
1811
|
+
};
|
|
1618
1812
|
let count = 0;
|
|
1619
1813
|
try {
|
|
1620
1814
|
const current = await readJson(crashCounterPath(projectRoot), null);
|
|
@@ -1625,15 +1819,21 @@ async function handleEvaluateCrash(projectRoot, error) {
|
|
|
1625
1819
|
const next = count + 1;
|
|
1626
1820
|
try {
|
|
1627
1821
|
await writeJsonAtomic(crashCounterPath(projectRoot), { count: next, updatedAt: Date.now() });
|
|
1628
|
-
} catch {
|
|
1629
|
-
// Counter unwritable:
|
|
1630
|
-
|
|
1822
|
+
} catch (writeError) {
|
|
1823
|
+
// Counter unwritable: the crash loop can no longer be bounded, so blocking again
|
|
1824
|
+
// would bounce this stop forever. Release loudly — never silently.
|
|
1825
|
+
loudRelease(
|
|
1826
|
+
`UKit completion gate: ${detail}. The crash counter could not be persisted `
|
|
1827
|
+
+ `(${writeError?.message || writeError}), so the gate cannot bound repeated crashes and is `
|
|
1828
|
+
+ 'releasing loudly instead of blocking forever. Run: ukit install to refresh the runtime, '
|
|
1829
|
+
+ 'then re-send the task in a new message.',
|
|
1830
|
+
);
|
|
1631
1831
|
return;
|
|
1632
1832
|
}
|
|
1633
1833
|
if (next >= MAX_GATE_CRASHES) {
|
|
1634
|
-
|
|
1635
|
-
|
|
1636
|
-
|
|
1834
|
+
loudRelease(
|
|
1835
|
+
`UKit completion gate: ${detail}. This recurred ${next} times and cannot be bounded, so the stop is releasing loudly instead of blocking again. The hook payload is corrupt — run: ukit install to refresh the runtime, then re-send the task in a new message.`,
|
|
1836
|
+
);
|
|
1637
1837
|
return;
|
|
1638
1838
|
}
|
|
1639
1839
|
process.stdout.write(`${JSON.stringify({ decision: 'block', reason: blockReason() })}\n`);
|
|
@@ -97,6 +97,20 @@ function chainFailureKind(processResult) {
|
|
|
97
97
|
}
|
|
98
98
|
}
|
|
99
99
|
|
|
100
|
+
// B1 (TASK-001, SPEC §FR-001): a child that emits a `hookSpecificOutput`
|
|
101
|
+
// permission decision owns the verdict. `hookSpecificOutput` alone is NOT
|
|
102
|
+
// enough — a child killed mid-write can carry the marker in a truncated
|
|
103
|
+
// capture, and calling that a decision would exempt a genuinely-skipped gate
|
|
104
|
+
// from the fail-closed verdict. A real decision is a clean exit-0 verdict.
|
|
105
|
+
function emitsPermissionDecision(entry) {
|
|
106
|
+
return Boolean(entry)
|
|
107
|
+
&& !entry.killed
|
|
108
|
+
&& entry.code === 0
|
|
109
|
+
&& entry.failureKind === 'ok'
|
|
110
|
+
&& typeof entry.stdout === 'string'
|
|
111
|
+
&& entry.stdout.includes('"hookSpecificOutput"');
|
|
112
|
+
}
|
|
113
|
+
|
|
100
114
|
function recordTiming(projectRoot, payload, timing) {
|
|
101
115
|
// Timing telemetry is advisory and must never delay or block a tool call;
|
|
102
116
|
// appendTelemetryRow carries the same posture (and the per-session cap).
|
|
@@ -109,12 +123,37 @@ function recordTiming(projectRoot, payload, timing) {
|
|
|
109
123
|
// A `:0` or negative suffix is rejected (falls back) — zero would mean "no
|
|
110
124
|
// budget", which silently disables the deadline; that is never a valid hook
|
|
111
125
|
// contract.
|
|
126
|
+
// TASK-001 (C8 / SPEC §FR-006): "rejected silently" was the bug. A `:0`/`: -1`
|
|
127
|
+
// typo used to leave the suffix on the path, so the child never spawned and the
|
|
128
|
+
// hook silently vanished from the chain. It now warns (never hard-fails — a
|
|
129
|
+
// config typo must not block every tool call in a fail-closed chain) and runs
|
|
130
|
+
// the script under the DEFAULT child budget.
|
|
131
|
+
const TIMEOUT_SUFFIX_RE = /^(.*):(-?\d+(?:\.\d+)?)$/;
|
|
132
|
+
|
|
133
|
+
// The ONE suffix stripper. Identity checks that need a script's declared name
|
|
134
|
+
// (fail-closed lookups, skipped-gate reporting) go through it too: a `:-1` arg
|
|
135
|
+
// must not look like an advisory script there while `parseScriptArg` already
|
|
136
|
+
// normalized it to a gate for execution.
|
|
137
|
+
function stripTimeoutSuffix(arg) {
|
|
138
|
+
const match = TIMEOUT_SUFFIX_RE.exec(String(arg ?? ''));
|
|
139
|
+
return match ? match[1] : String(arg ?? '');
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
function failClosedName(arg) {
|
|
143
|
+
return path.basename(stripTimeoutSuffix(arg));
|
|
144
|
+
}
|
|
145
|
+
|
|
112
146
|
function parseScriptArg(arg) {
|
|
113
|
-
const
|
|
114
|
-
if (!
|
|
115
|
-
const seconds = Number(
|
|
116
|
-
if (!Number.isFinite(seconds) || seconds <= 0)
|
|
117
|
-
|
|
147
|
+
const suffix = TIMEOUT_SUFFIX_RE.exec(String(arg ?? ''));
|
|
148
|
+
if (!suffix) return { scriptPath: arg, timeoutMs: null };
|
|
149
|
+
const seconds = Number(suffix[2]);
|
|
150
|
+
if (!Number.isFinite(seconds) || seconds <= 0) {
|
|
151
|
+
process.stderr.write(
|
|
152
|
+
`[ukit] ignoring invalid timeout suffix "${arg}" — using default child budget\n`,
|
|
153
|
+
);
|
|
154
|
+
return { scriptPath: suffix[1], timeoutMs: null };
|
|
155
|
+
}
|
|
156
|
+
return { scriptPath: suffix[1], timeoutMs: Math.round(seconds * 1000) };
|
|
118
157
|
}
|
|
119
158
|
|
|
120
159
|
// TASK-001 (SPEC §FR-001, §8): a chain arg ending in `.mjs` is an IN-PROC step —
|
|
@@ -178,7 +217,7 @@ async function runModuleStep({ scriptPath, payload, payloadText, projectRoot, en
|
|
|
178
217
|
}
|
|
179
218
|
}
|
|
180
219
|
|
|
181
|
-
async function run(payloadText, scriptArgs, { chainMarker = true } = {}) {
|
|
220
|
+
async function run(payloadText, scriptArgs, { chainMarker = true, decisionShortCircuit = false, stdinStageMs = null } = {}) {
|
|
182
221
|
const parsedArgs = scriptArgs.map(parseScriptArg);
|
|
183
222
|
const scriptPaths = parsedArgs.map((a) => a.scriptPath);
|
|
184
223
|
const payload = JSON.parse(payloadText || '{}');
|
|
@@ -224,7 +263,7 @@ async function run(payloadText, scriptArgs, { chainMarker = true } = {}) {
|
|
|
224
263
|
const deadline = startedAt + totalBudgetMs;
|
|
225
264
|
const results = [];
|
|
226
265
|
let budgetExhausted = false;
|
|
227
|
-
|
|
266
|
+
let decisionEmitted = false;
|
|
228
267
|
for (let scriptIndex = 0; scriptIndex < scriptPaths.length; scriptIndex++) {
|
|
229
268
|
const scriptPath = scriptPaths[scriptIndex];
|
|
230
269
|
const scriptName = path.basename(scriptPath);
|
|
@@ -321,7 +360,20 @@ async function run(payloadText, scriptArgs, { chainMarker = true } = {}) {
|
|
|
321
360
|
});
|
|
322
361
|
}
|
|
323
362
|
|
|
324
|
-
|
|
363
|
+
// B1 (TASK-001, SPEC §FR-001): under `--emit-verdict` a child that emits a
|
|
364
|
+
// `hookSpecificOutput` permission decision owns the verdict even at exit 0
|
|
365
|
+
// (the direct Claude contract: deny + exit 0 still blocks). The chain stops
|
|
366
|
+
// HERE, before the next child spawns — filtering the decision at emit time
|
|
367
|
+
// instead would still let the next script's side effects run (e.g.
|
|
368
|
+
// pre-edit-backup.sh backing up a file whose edit was just denied) and its
|
|
369
|
+
// stdout would be concatenated in front of the decision JSON, which Claude
|
|
370
|
+
// Code rejects as invalid JSON. The predicate is shared with the emit-time
|
|
371
|
+
// lookup so a truncated capture can never count as a decision here.
|
|
372
|
+
|
|
373
|
+
if (decisionShortCircuit && emitsPermissionDecision(results[results.length - 1])) {
|
|
374
|
+
decisionEmitted = true;
|
|
375
|
+
}
|
|
376
|
+
if (decisionEmitted || code === 2 || killed || (code !== 0 && FAIL_CLOSED_SCRIPTS.has(scriptName))) {
|
|
325
377
|
break;
|
|
326
378
|
}
|
|
327
379
|
}
|
|
@@ -331,9 +383,12 @@ async function run(payloadText, scriptArgs, { chainMarker = true } = {}) {
|
|
|
331
383
|
// of those unrun scripts is fail-closed, the chain must fail CLOSED — the old
|
|
332
384
|
// per-script path ran every hook independently, so a timed-out advisory never
|
|
333
385
|
// skipped a gate. `skippedFailClosed` carries that signal to the verdict.
|
|
334
|
-
|
|
386
|
+
// ... unless this break WAS the decision: a decision owns the verdict, so the
|
|
387
|
+
// unrun gates are not a fail-closed gap (SPEC §FR-001 — the verdict must stay
|
|
388
|
+
// exit 0 with the decision JSON).
|
|
389
|
+
const skippedFailClosed = !decisionEmitted && scriptPaths
|
|
335
390
|
.slice(results.length)
|
|
336
|
-
.some((p) => FAIL_CLOSED_SCRIPTS.has(
|
|
391
|
+
.some((p) => FAIL_CLOSED_SCRIPTS.has(failClosedName(p)));
|
|
337
392
|
|
|
338
393
|
const elapsedMs = Date.now() - startedAt;
|
|
339
394
|
// TASK-019: versioned rows shared with direct hooks. `outcome` reuses this
|
|
@@ -350,6 +405,18 @@ async function run(payloadText, scriptArgs, { chainMarker = true } = {}) {
|
|
|
350
405
|
toolName: payload?.tool_name || null,
|
|
351
406
|
toolUseId: payload?.tool_use_id || null,
|
|
352
407
|
elapsedMs,
|
|
408
|
+
// O3 (SPEC §FR-009): the row's wall time split into the stdin stage — the
|
|
409
|
+
// bounded read this process waits on while the producer writes — and the
|
|
410
|
+
// chain's own execution. A producer holding the pipe open is then
|
|
411
|
+
// attributable instead of looking like a slow hook (the reported artifact:
|
|
412
|
+
// project-important.sh p95 = 1539ms with no row doing real work ≥ 1s).
|
|
413
|
+
// Clamped to `elapsedMs` because the split describes the row it sits on
|
|
414
|
+
// (the two are measured from different origins). Absent — not 0 and not
|
|
415
|
+
// null — when this invocation never read stdin, so a v1 reader sees the
|
|
416
|
+
// row it always saw.
|
|
417
|
+
...(stdinStageMs === null
|
|
418
|
+
? {}
|
|
419
|
+
: { stdinStageMs: Math.min(Math.max(0, Math.round(stdinStageMs)), elapsedMs) }),
|
|
353
420
|
budgetMs: totalBudgetMs,
|
|
354
421
|
budgetExhausted,
|
|
355
422
|
scripts: results.map(({ scriptName, code, killed, failureKind, elapsedMs: scriptElapsedMs }) => ({
|
|
@@ -358,12 +425,48 @@ async function run(payloadText, scriptArgs, { chainMarker = true } = {}) {
|
|
|
358
425
|
killed,
|
|
359
426
|
failureKind,
|
|
360
427
|
elapsedMs: scriptElapsedMs,
|
|
428
|
+
// TASK-007 (SPEC §FR-012): which steps the runner treated as gates. The
|
|
429
|
+
// doctor reads this flag instead of importing FAIL_CLOSED_SCRIPTS — the
|
|
430
|
+
// runner is the only component that knows which paths it gated, and a
|
|
431
|
+
// second copy of a security-relevant list is a drift hazard. Additive and
|
|
432
|
+
// optional: v1 readers that read the five original keys keep working.
|
|
433
|
+
failClosed: FAIL_CLOSED_SCRIPTS.has(scriptName),
|
|
361
434
|
})),
|
|
362
435
|
});
|
|
363
436
|
|
|
364
437
|
return { results, elapsedMs, budgetMs: totalBudgetMs, budgetExhausted, skippedFailClosed };
|
|
365
438
|
}
|
|
366
439
|
|
|
440
|
+
// TASK-006 (SPEC §S6, F-RC-2): a timed-out in-proc step abandons its
|
|
441
|
+
// Promise.race loser — when that loser holds ref'd libuv handles (timers,
|
|
442
|
+
// watchers, sockets) the event loop stays pinned and the host waits the full
|
|
443
|
+
// registered 30–73s timeout for a process that already produced its verdict.
|
|
444
|
+
// Every emit path ends here: flush stdout/stderr, then exit explicitly. The
|
|
445
|
+
// flush is bounded — a reader that never drains must not turn the fix into a
|
|
446
|
+
// new stall.
|
|
447
|
+
const FLUSH_EXIT_MS = Number(process.env.UKIT_HOOK_FLUSH_EXIT_MS || 2000);
|
|
448
|
+
|
|
449
|
+
function flushStream(stream) {
|
|
450
|
+
return new Promise((resolve) => {
|
|
451
|
+
try {
|
|
452
|
+
// end() flushes every queued chunk before invoking the callback —
|
|
453
|
+
// including when the stream was already ended/destroyed (the callback
|
|
454
|
+
// still fires, with an error we deliberately ignore).
|
|
455
|
+
stream.end(() => resolve());
|
|
456
|
+
} catch {
|
|
457
|
+
resolve();
|
|
458
|
+
}
|
|
459
|
+
});
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
async function flushAndExit(code) {
|
|
463
|
+
await Promise.race([
|
|
464
|
+
Promise.all([flushStream(process.stdout), flushStream(process.stderr)]),
|
|
465
|
+
new Promise((resolve) => setTimeout(resolve, FLUSH_EXIT_MS).unref()),
|
|
466
|
+
]);
|
|
467
|
+
process.exit(code);
|
|
468
|
+
}
|
|
469
|
+
|
|
367
470
|
try {
|
|
368
471
|
let argv = process.argv.slice(2);
|
|
369
472
|
// TASK-234: `--emit-verdict` adapts the runner for Claude Code settings.json
|
|
@@ -384,8 +487,14 @@ try {
|
|
|
384
487
|
// '-' reads the payload from stdin — the form Claude Code hook commands use.
|
|
385
488
|
let payloadText = payloadArg;
|
|
386
489
|
let stdinTruncated = false;
|
|
490
|
+
// O3: the stage window is measured HERE, around the bounded read, and not
|
|
491
|
+
// inside run() — run() starts after the payload is already in hand, and a
|
|
492
|
+
// truncated read exits before run() is ever called.
|
|
493
|
+
let stdinStageMs = null;
|
|
387
494
|
if (payloadArg === '-') {
|
|
495
|
+
const stageStartedAt = Date.now();
|
|
388
496
|
const staged = await readStdinBounded();
|
|
497
|
+
stdinStageMs = Date.now() - stageStartedAt;
|
|
389
498
|
payloadText = staged.text;
|
|
390
499
|
stdinTruncated = staged.truncated;
|
|
391
500
|
} else if (payloadArg.startsWith('@')) {
|
|
@@ -402,8 +511,7 @@ try {
|
|
|
402
511
|
// chain carries a fail-closed gate, emit deny now instead of letting gates
|
|
403
512
|
// pass on a payload they never fully received.
|
|
404
513
|
if (stdinTruncated) {
|
|
405
|
-
const hasFailClosed = scriptPaths.some((p) =>
|
|
406
|
-
FAIL_CLOSED_SCRIPTS.has(path.basename(String(p).replace(/:\d+(?:\.\d+)?$/, ''))));
|
|
514
|
+
const hasFailClosed = scriptPaths.some((p) => FAIL_CLOSED_SCRIPTS.has(failClosedName(p)));
|
|
407
515
|
if (hasFailClosed) {
|
|
408
516
|
const deny = JSON.stringify({
|
|
409
517
|
hookSpecificOutput: {
|
|
@@ -415,19 +523,25 @@ try {
|
|
|
415
523
|
});
|
|
416
524
|
if (emitVerdict) {
|
|
417
525
|
process.stdout.write(deny);
|
|
418
|
-
process.exitCode = 0;
|
|
419
526
|
} else {
|
|
420
527
|
process.stdout.write(JSON.stringify({
|
|
421
528
|
results: [],
|
|
422
529
|
wrapperError: 'stdin staging truncated — fail-closed chain refused',
|
|
423
530
|
stdinTruncated: true,
|
|
424
531
|
}));
|
|
425
|
-
process.exitCode = 2;
|
|
426
532
|
}
|
|
427
|
-
process.exit
|
|
533
|
+
// TASK-006: flush before exit — a bare process.exit can truncate a
|
|
534
|
+
// verdict still queued on the pipe.
|
|
535
|
+
await flushAndExit(emitVerdict ? 0 : 2);
|
|
428
536
|
}
|
|
429
537
|
}
|
|
430
|
-
const chain = await run(payloadText, scriptPaths, {
|
|
538
|
+
const chain = await run(payloadText, scriptPaths, {
|
|
539
|
+
chainMarker: !emitVerdict,
|
|
540
|
+
// B1: only the --emit-verdict contract short-circuits on a decision; the
|
|
541
|
+
// omp bridge parses the JSON envelope and needs every step's context stdout.
|
|
542
|
+
decisionShortCircuit: emitVerdict,
|
|
543
|
+
stdinStageMs,
|
|
544
|
+
});
|
|
431
545
|
if (!emitVerdict) {
|
|
432
546
|
process.stdout.write(JSON.stringify(chain));
|
|
433
547
|
} else {
|
|
@@ -435,10 +549,21 @@ try {
|
|
|
435
549
|
// verdict even when it exits 0 (the direct Claude contract: deny + exit 0
|
|
436
550
|
// still blocks). The first decision wins, matching per-script semantics
|
|
437
551
|
// where each hook's output is its own verdict and a deny short-circuits.
|
|
552
|
+
// The loose lookup decides which entry is REPORTED as the verdict owner (a
|
|
553
|
+
// non-zero decision emitter stays routed through the code-2/fail-closed
|
|
554
|
+
// paths below, per SPEC §FR-001); the strict predicate decides whether that
|
|
555
|
+
// entry may claim stdout outright.
|
|
438
556
|
const decisionResult = chain.results.find((r) =>
|
|
439
557
|
typeof r.stdout === 'string' && r.stdout.includes('"hookSpecificOutput"'));
|
|
558
|
+
const decisionOwnsVerdict = emitsPermissionDecision(decisionResult);
|
|
440
559
|
const last = decisionResult ?? chain.results[chain.results.length - 1];
|
|
441
560
|
|
|
561
|
+
// B1 (SPEC §FR-001): a clean decision owns the ENTIRE stdout — no context
|
|
562
|
+
// text before or after it. Claude Code parses stdout as one JSON document,
|
|
563
|
+
// so leading context turns a valid deny into "Hook JSON output validation
|
|
564
|
+
// failed"; the chain already stopped at the decision, so no later script's
|
|
565
|
+
// output can be concatenated in.
|
|
566
|
+
|
|
442
567
|
// TASK-234 review fix (critical): context stdout must be REPLAYED, not
|
|
443
568
|
// dropped. SessionStart/UserPromptSubmit hooks emit plain-text context
|
|
444
569
|
// (PROJECT_IMPORTANT mandate, skill-router guidance) — replaying only the
|
|
@@ -449,15 +574,18 @@ try {
|
|
|
449
574
|
.filter((r) => r !== decisionResult && r !== last && typeof r.stdout === 'string' && r.stdout.length > 0)
|
|
450
575
|
.map((r) => r.stdout)
|
|
451
576
|
.join('');
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
if (chain.skippedFailClosed) {
|
|
577
|
+
if (decisionOwnsVerdict) {
|
|
578
|
+
process.stdout.write(decisionResult.stdout);
|
|
579
|
+
if (decisionResult.stderr) process.stderr.write(decisionResult.stderr);
|
|
580
|
+
process.exitCode = 0;
|
|
581
|
+
} else if (chain.skippedFailClosed) {
|
|
582
|
+
// TASK-234 review fix (critical): a mid-chain break that skipped a
|
|
583
|
+
// fail-closed gate must fail CLOSED — the old per-script path ran every
|
|
584
|
+
// hook independently, so a killed advisory never skipped a gate.
|
|
457
585
|
if (contextStdout) process.stdout.write(contextStdout);
|
|
458
586
|
const skipped = scriptPaths
|
|
459
587
|
.slice(chain.results.length)
|
|
460
|
-
.map((p) =>
|
|
588
|
+
.map((p) => failClosedName(p))
|
|
461
589
|
.filter((name) => FAIL_CLOSED_SCRIPTS.has(name))
|
|
462
590
|
.join(', ');
|
|
463
591
|
process.stderr.write(`UKit hook chain broke before fail-closed gate(s) ran: ${skipped}\n`);
|
|
@@ -466,7 +594,7 @@ try {
|
|
|
466
594
|
// No script ran at all (empty chain or budget spent before the first
|
|
467
595
|
// child). With fail-closed scripts declared in the chain this must not
|
|
468
596
|
// fail open.
|
|
469
|
-
const hasFailClosed = scriptPaths.some((p) => FAIL_CLOSED_SCRIPTS.has(
|
|
597
|
+
const hasFailClosed = scriptPaths.some((p) => FAIL_CLOSED_SCRIPTS.has(failClosedName(p)));
|
|
470
598
|
if (hasFailClosed) {
|
|
471
599
|
process.stderr.write('UKit hook chain produced no verdict — fail-closed gate did not run\n');
|
|
472
600
|
process.exitCode = 2;
|
|
@@ -497,10 +625,14 @@ try {
|
|
|
497
625
|
}
|
|
498
626
|
}
|
|
499
627
|
}
|
|
628
|
+
// TASK-006 (F-RC-2): the verdict is on the wire — exit now. Waiting for a
|
|
629
|
+
// natural exit lets an abandoned in-proc step's ref'd handles pin the loop
|
|
630
|
+
// until the host's registered timeout kills us.
|
|
631
|
+
await flushAndExit(process.exitCode ?? 0);
|
|
500
632
|
} catch (error) {
|
|
501
633
|
process.stdout.write(JSON.stringify({
|
|
502
634
|
results: [],
|
|
503
635
|
wrapperError: error?.message || String(error),
|
|
504
636
|
}));
|
|
505
|
-
|
|
637
|
+
await flushAndExit(1);
|
|
506
638
|
}
|