@ngockhoale/ukit 2.3.8 → 2.3.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +90 -0
- package/package.json +1 -1
- package/src/core/compact/threshold.js +13 -0
- package/src/core/fileOps.js +61 -7
- package/src/index/taskRouting.js +35 -2
- package/templates/.claude/agents/bug-debugger.md +2 -2
- package/templates/.claude/agents/feature-implementer.md +5 -3
- package/templates/.claude/hooks/auto-allow-bash.sh +44 -7
- package/templates/.claude/hooks/auto-prune-bash.sh +44 -7
- package/templates/.claude/hooks/context-hardcap-gate.sh +33 -8
- package/templates/.claude/hooks/context-window-guard.sh +1 -1
- package/templates/.claude/hooks/reset-compact-pressure.sh +47 -7
- package/templates/.claude/hooks/skill-router.sh +80 -7
- package/templates/.claude/hooks/verification-guard.sh +55 -20
- package/templates/.claude/ukit/index/route-task.mjs +36 -3
- package/templates/.claude/ukit/runtime/compact-threshold.mjs +60 -5
- package/templates/.claude/ukit/runtime/execution-ledger.mjs +361 -62
- package/templates/.claude/ukit/runtime/token-utils.mjs +69 -6
- package/templates/.omp/agents/bug-debugger.md +2 -2
- package/templates/.omp/agents/feature-implementer.md +5 -3
- package/templates/.omp/hooks/pre/ukit-bridge.js +68 -16
|
@@ -11,7 +11,9 @@ Implement requested behavior with minimal scope drift.
|
|
|
11
11
|
- **Daily/ad-hoc mode** (DEFAULT): task didn't come from `docs/AI_HANDOFF/` → use the original lightweight workflow. Tests only when touched code already has coverage. No reviewer trigger.
|
|
12
12
|
- **Handoff mode**: task file is `docs/AI_HANDOFF/tasks/TASK-xxx.md` OR user explicitly invokes handoff (e.g. "execute task TASK-001") → activate full Quality Gate: test-first → green → reviewer.
|
|
13
13
|
|
|
14
|
-
If unsure which mode applies,
|
|
14
|
+
If unsure which mode applies, default to Daily/ad-hoc mode unless the task path or prompt
|
|
15
|
+
explicitly selects Handoff. Do not ask a user — you are a worker; report the ambiguity and
|
|
16
|
+
reasoning to the parent agent so it can decide whether to re-route.
|
|
15
17
|
|
|
16
18
|
**In Handoff mode you are running unattended — ask nothing.** You were spawned by an
|
|
17
19
|
orchestrator driving a pipeline; there is no human in your conversation to answer, and a
|
|
@@ -32,7 +34,7 @@ and even then, report it, don't ask about it.
|
|
|
32
34
|
- non-trivial: `docs/MEMORY.md` + `docs/PROJECT.md` + `docs/CODE_MAP.md`
|
|
33
35
|
- Identify target files and existing patterns.
|
|
34
36
|
- If task came from handoff, read `tasks/TASK-xxx.md` and locate its **Test Plan** + **Verification Commands**.
|
|
35
|
-
- Daily mode: if confidence is low or risk is high,
|
|
37
|
+
- Daily mode: if confidence is low or risk is high, inspect the next bounded source/context signal and hand back a concise `STATUS: BLOCKED` report with the exact missing decision or artifact if confidence remains low. Do not ask a user directly. Handoff mode: do not ask — decide and record the decision (see above).
|
|
36
38
|
|
|
37
39
|
### 2. Plan Approach (< 1 minute)
|
|
38
40
|
|
|
@@ -43,7 +45,7 @@ and even then, report it, don't ask about it.
|
|
|
43
45
|
|
|
44
46
|
- Write the test(s) from §2 / from task Test Plan.
|
|
45
47
|
- Run them: must FAIL for the expected reason. Capture output.
|
|
46
|
-
- If test passes immediately → test is wrong or behavior already exists. Fix the test
|
|
48
|
+
- If test passes immediately → test is wrong or behavior already exists. Fix the test; if the intended behavior cannot be inferred, hand back `STATUS: BLOCKED` with the observed behavior and the smallest decision the parent must resolve. Never silently stop.
|
|
47
49
|
- **Daily mode**: skip this step unless touched code already has tests (then follow original rule).
|
|
48
50
|
|
|
49
51
|
|
|
@@ -11,8 +11,10 @@ import { fileURLToPath } from 'node:url';
|
|
|
11
11
|
import { spawnSync } from 'node:child_process';
|
|
12
12
|
import {
|
|
13
13
|
evaluateCompletion,
|
|
14
|
+
evidencePromptKey,
|
|
14
15
|
incrementContinuation,
|
|
15
16
|
markNotified,
|
|
17
|
+
noteStopProgress,
|
|
16
18
|
readExecutionLedger,
|
|
17
19
|
readRouteState,
|
|
18
20
|
recordExecutionReceipt,
|
|
@@ -107,6 +109,15 @@ function classifyFailure(scriptName) {
|
|
|
107
109
|
return FAIL_CLOSED_SCRIPTS.has(scriptName) ? 'closed' : 'open';
|
|
108
110
|
}
|
|
109
111
|
|
|
112
|
+
// A child killed by its per-script budget produced NO verdict — that is an infrastructure
|
|
113
|
+
// event (slow disk, lock contention, process storm), not a safety decision. Routing it
|
|
114
|
+
// into the fail-closed branch below blocked every Edit/Write (and Read's sensitive-data
|
|
115
|
+
// guard) for as long as the machine was slow — the exact mid-run freeze class this bridge
|
|
116
|
+
// must not create. One deliberate exception: block-dangerous.sh is the gate the user
|
|
117
|
+
// explicitly required to never fail open (destructive-command protection), so a Bash chain
|
|
118
|
+
// whose dangerous-command check timed out stays blocked with the honest reason.
|
|
119
|
+
const TIMEOUT_STAYS_CLOSED = new Set(['block-dangerous.sh']);
|
|
120
|
+
|
|
110
121
|
function runtimeMetadata(event = {}, context = {}) {
|
|
111
122
|
const sessionManager = context?.sessionManager;
|
|
112
123
|
return {
|
|
@@ -160,12 +171,30 @@ function parseStructuredDecision(stdout) {
|
|
|
160
171
|
|
|
161
172
|
function translateExecResult(scriptName, execResult) {
|
|
162
173
|
const killed = Boolean(execResult?.killed);
|
|
163
|
-
const code = killed ? 1 : (execResult?.code ?? 0);
|
|
164
174
|
const stdout = execResult?.stdout ?? '';
|
|
165
175
|
const stderr = killed
|
|
166
176
|
? (execResult?.stderr || `${scriptName} was killed before it completed`)
|
|
167
177
|
: (execResult?.stderr ?? '');
|
|
168
178
|
|
|
179
|
+
if (killed) {
|
|
180
|
+
if (TIMEOUT_STAYS_CLOSED.has(scriptName)) {
|
|
181
|
+
return {
|
|
182
|
+
block: true,
|
|
183
|
+
reason: `${scriptName} exceeded its hook budget and was killed, so UKit could not verify the command is safe — the call stays blocked. Retry once; if this repeats, the machine is too slow for the guard to finish.`,
|
|
184
|
+
stdout,
|
|
185
|
+
stderr,
|
|
186
|
+
};
|
|
187
|
+
}
|
|
188
|
+
return {
|
|
189
|
+
block: false,
|
|
190
|
+
warning: `${scriptName} exceeded its hook budget and was killed — treated as "could not verify", not as a block (a timeout is an infrastructure event, not a verdict): ${stderr}`,
|
|
191
|
+
stdout,
|
|
192
|
+
stderr,
|
|
193
|
+
};
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
const code = execResult?.code ?? 0;
|
|
197
|
+
|
|
169
198
|
if (code === 0) {
|
|
170
199
|
return { block: false, stdout, stderr };
|
|
171
200
|
}
|
|
@@ -437,16 +466,18 @@ function sendContext(pi, context, deliverAs, { display = false } = {}) {
|
|
|
437
466
|
|
|
438
467
|
export async function runToolCall(pi, event, { projectRoot, context: extensionContext = {} }) {
|
|
439
468
|
const rawToolName = event.toolName ?? event.tool;
|
|
440
|
-
const
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
469
|
+
const mappedToolName = mapToolName(rawToolName);
|
|
470
|
+
// An unmapped tool that LOOKS like a file mutation must not sail past the guards — but
|
|
471
|
+
// hard-blocking it froze the run whenever omp renamed or added a tool, because the model
|
|
472
|
+
// cannot add a host adapter mapping mid-run and the block had no recovery path. Run the
|
|
473
|
+
// Edit guard chain on the normalized input instead: the guards key off
|
|
474
|
+
// tool_input.file_path and the mutation fields, not the host tool name. The mapping gap
|
|
475
|
+
// still surfaces as a visible warning so it gets fixed.
|
|
476
|
+
const unmappedMutation = !mappedToolName && looksMutationCapable(event.input);
|
|
477
|
+
const toolName = mappedToolName || (unmappedMutation ? 'Edit' : null);
|
|
478
|
+
const priorContext = unmappedMutation
|
|
479
|
+
? [`[UKit] unmapped mutation-capable tool "${rawToolName}" was normalized to Edit so the guard chain still ran; add a mapping for it in ukit-bridge.js.`]
|
|
480
|
+
: [];
|
|
450
481
|
|
|
451
482
|
const metadata = runtimeMetadata(event, extensionContext);
|
|
452
483
|
const payload = buildHookPayload('PreToolUse', {
|
|
@@ -462,8 +493,9 @@ export async function runToolCall(pi, event, { projectRoot, context: extensionCo
|
|
|
462
493
|
projectRoot,
|
|
463
494
|
failClosedOnTransportError,
|
|
464
495
|
});
|
|
465
|
-
|
|
466
|
-
|
|
496
|
+
const merged = { ...result, context: [...priorContext, ...(result.context || [])] };
|
|
497
|
+
if (!merged.block) sendContext(pi, merged.context, 'steer');
|
|
498
|
+
return { ...merged, toolName };
|
|
467
499
|
}
|
|
468
500
|
|
|
469
501
|
export async function runToolResult(pi, event, { projectRoot, context: extensionContext = {} }) {
|
|
@@ -581,9 +613,12 @@ export async function runSessionStop(
|
|
|
581
613
|
const evaluation = evaluateCompletion({ state, ledger });
|
|
582
614
|
if (!evaluation.continue) {
|
|
583
615
|
if (evaluation.capped) pi.logger?.warn?.(`[UKit] ${evaluation.reason}`);
|
|
584
|
-
|
|
616
|
+
const missingEvidence = Array.isArray(evaluation.missingEvidence) ? evaluation.missingEvidence : [];
|
|
617
|
+
// The visible-blocker exit (verification loop) carries no missingEvidence but must still
|
|
618
|
+
// reach the user — a silent release is indistinguishable from a stall.
|
|
619
|
+
if (missingEvidence.length > 0 || evaluation.notify) {
|
|
585
620
|
const notice = [
|
|
586
|
-
`[UKit] Stopping with unfinished work: ${
|
|
621
|
+
missingEvidence.length > 0 ? `[UKit] Stopping with unfinished work: ${missingEvidence.join(', ')}.` : '[UKit] Stopping automatic recovery.',
|
|
587
622
|
evaluation.reason,
|
|
588
623
|
].filter(Boolean).join(' ');
|
|
589
624
|
// VERIFIED (2026-08-29, source read of omp v18.0.10 pi-coding-agent/src) PLAN.md §3 D4:
|
|
@@ -598,9 +633,26 @@ export async function runSessionStop(
|
|
|
598
633
|
}
|
|
599
634
|
|
|
600
635
|
if (suppliedLedger === undefined) {
|
|
636
|
+
// Claude Code flags reentrant stops with stop_hook_active and the ledger CLI releases on
|
|
637
|
+
// them; omp's session_stop carries no such field. Detect the equivalent from the ledger:
|
|
638
|
+
// when the recovery turn produced no new receipts since the stop that blocked it,
|
|
639
|
+
// blocking again would only burn another turn in a self-sustaining loop — release with a
|
|
640
|
+
// visible reason instead. Vibecode autonomy keeps pushing (same as the CLI valve).
|
|
641
|
+
let reentrantStop = false;
|
|
642
|
+
try {
|
|
643
|
+
reentrantStop = (await noteStopProgress(projectRoot, payload)).reentrant === true;
|
|
644
|
+
} catch {
|
|
645
|
+
reentrantStop = false;
|
|
646
|
+
}
|
|
647
|
+
if (reentrantStop && state?.routeSummary?.autonomyLevel !== 'vibecode') {
|
|
648
|
+
const notice = `UKit stopped automatic recovery after one continuation: ${evaluation.reason}`;
|
|
649
|
+
pi.logger?.warn?.(`[UKit] ${notice}`);
|
|
650
|
+
sendContext(pi, [notice], 'nextTurn', { display: true });
|
|
651
|
+
return undefined;
|
|
652
|
+
}
|
|
601
653
|
try {
|
|
602
654
|
if (evaluation.finalNotice) await markNotified(projectRoot, payload, ledger);
|
|
603
|
-
else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null);
|
|
655
|
+
else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null, evidencePromptKey(state));
|
|
604
656
|
} catch (error) {
|
|
605
657
|
pi.logger?.warn?.(`[UKit] continuation bookkeeping failed open: ${error?.message || error}`);
|
|
606
658
|
}
|