@ngockhoale/ukit 2.3.0 → 2.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,58 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to UKit are documented here.
|
|
4
4
|
|
|
5
|
+
## 2.3.1 - 2026-09-09
|
|
6
|
+
|
|
7
|
+
"1 prompt mà đứng 10 lần" is fixed at the root. Four stall vectors in the completion
|
|
8
|
+
gate and router are gone, no stop is silent anymore, and routine backend model swaps
|
|
9
|
+
are now worked through instead of stalled on. All found by root-cause debugging with a
|
|
10
|
+
failing regression test per fix.
|
|
11
|
+
|
|
12
|
+
1) Stop-hook recursion: `execution-ledger.mjs --evaluate-stop` re-blocked the reentrant
|
|
13
|
+
Stop event (Claude Code re-invokes Stop with `stop_hook_active: true` after a block), so
|
|
14
|
+
one block became a self-sustaining block/continue chain ending in a cap-forced mid-task
|
|
15
|
+
stop. The gate now honors `stop_hook_active` and, when evidence is still missing, emits
|
|
16
|
+
a user-visible `systemMessage` instead of silently ending.
|
|
17
|
+
|
|
18
|
+
2) requestKey churn wiped evidence: `skill-router.sh` rebuilds requestKey per tool call
|
|
19
|
+
(commandText/targetFile are hashed in), so the first verification command re-keyed the
|
|
20
|
+
route mid-request and `--record` took the freshLedger path, erasing the write/verification
|
|
21
|
+
evidence the same request had produced — the gate then re-blocked until the cap forced a
|
|
22
|
+
silent stop. The ledger now keys evidence identity by the user prompt text
|
|
23
|
+
(`promptKey`); a re-key within one logical request carries evidence forward, a genuinely
|
|
24
|
+
new prompt starts clean.
|
|
25
|
+
|
|
26
|
+
3) Subagent stalls: subagent tool calls run the same hooks against the same session
|
|
27
|
+
ledger and route state (docs-confirmed; sidechain payloads carry `agent_id`/
|
|
28
|
+
`agent_type`). A subagent's Edit could re-route the request to a different
|
|
29
|
+
executionMode, which both broke the old promptKey identity (wiping main evidence) and
|
|
30
|
+
replaced the main route's completionEvidence. Fixes: promptKey is the prompt text alone
|
|
31
|
+
(null prompt never carries), and `skill-router.sh` early-exits on sidechain payloads so
|
|
32
|
+
subagent tool calls never touch the main route state. omp's bridge invokes the same
|
|
33
|
+
script, so every harness is covered.
|
|
34
|
+
|
|
35
|
+
4) No silent endings: the non-gated modes' `notify` and the post-cap `capped` paths now
|
|
36
|
+
emit a `systemMessage` alongside their stderr line — previously these ended turns with
|
|
37
|
+
the user seeing nothing.
|
|
38
|
+
|
|
39
|
+
Gateway model swaps (limit hit → a different vendor model behind the same alias) are
|
|
40
|
+
detected from the transcript's per-assistant `message.model` and surfaced in
|
|
41
|
+
`context-window-guard.sh` as ONE routine line per hour: keep working, do not restart or
|
|
42
|
+
re-plan; if a tool call errors after a swap, re-check the tool and call it again. The
|
|
43
|
+
swap itself cold-reads the prompt cache once — that latency is transport-level and
|
|
44
|
+
cannot be removed by hooks; everything after it now continues automatically.
|
|
45
|
+
|
|
46
|
+
Integrated on top of 2.3.0 (rebased; not published as 2.2.17): 2.3.0's stricter gates
|
|
47
|
+
made the re-key carry load-bearing for `targetedVerificationSucceeded`/`sourceFiles`
|
|
48
|
+
too — without carrying them, a mid-request re-key re-demanded targeted evidence and
|
|
49
|
+
re-introduced the stall this release fixes (regression-tested RED→GREEN). The 2.3.0
|
|
50
|
+
notify test that expected a silent stdout was updated to the no-silent-stops contract.
|
|
51
|
+
|
|
52
|
+
Verified: executionLedgerCli 29/29, skillRouterHook 33/33, contextWindowGuard 18/18;
|
|
53
|
+
full suite 1280/1280 (74 files) and `release:verify` green on the integrated tree;
|
|
54
|
+
template↔live mirrors byte-identical (live settings.json re-rendered through the install
|
|
55
|
+
pipeline, not copied raw).
|
|
56
|
+
|
|
5
57
|
## 2.3.0 - 2026-09-08
|
|
6
58
|
|
|
7
59
|
Model-agnostic hardening release. Every configured model name is treated as a gateway
|
package/package.json
CHANGED
|
@@ -131,24 +131,14 @@ const estimatedTokens = Math.round(contentChars / CHARS_PER_TOKEN);
|
|
|
131
131
|
const hardCap = loadHardCap();
|
|
132
132
|
const ratio = estimatedTokens / hardCap;
|
|
133
133
|
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
const pct = Math.round(ratio * 100);
|
|
137
|
-
const phase = ratio >= 1 ? 'hard' : 'soft';
|
|
138
|
-
|
|
139
|
-
// Debounce the directive block so it does not re-print in full every single prompt
|
|
140
|
-
// while the phase is unchanged — that would just be more tokens added to the same
|
|
141
|
-
// oversized context it is warning about. Re-arms on phase change (soft -> hard) or
|
|
142
|
-
// after the cooldown, and resets naturally once a real compaction/new session drops
|
|
143
|
-
// the live transcript back under 0.8, since this whole branch exits early above.
|
|
134
|
+
// Debounce/state file shared by the context warning and the gateway model-swap note.
|
|
144
135
|
const GUARD_STATE_PATH = path.join(projectRoot, '.ukit', 'storage', 'cache', 'context-guard-state.json');
|
|
145
|
-
const COOLDOWN_MS = phase === 'hard' ? 3 * 60 * 1000 : 8 * 60 * 1000;
|
|
146
136
|
|
|
147
137
|
function readGuardState() {
|
|
148
138
|
try {
|
|
149
139
|
return JSON.parse(fs.readFileSync(GUARD_STATE_PATH, 'utf8'));
|
|
150
140
|
} catch {
|
|
151
|
-
return { lastPhase: null, lastActionAt: 0 };
|
|
141
|
+
return { lastPhase: null, lastActionAt: 0, lastSwapModel: null, lastSwapNoteAt: 0 };
|
|
152
142
|
}
|
|
153
143
|
}
|
|
154
144
|
|
|
@@ -160,6 +150,68 @@ function writeGuardState(state) {
|
|
|
160
150
|
}
|
|
161
151
|
|
|
162
152
|
const guardState = readGuardState();
|
|
153
|
+
let persistedState = { ...guardState };
|
|
154
|
+
|
|
155
|
+
// ── Gateway model-swap note ──
|
|
156
|
+
// Assistant entries carry the REAL model that served them, so a swapping gateway
|
|
157
|
+
// (the backend alias resolves to a different vendor model, e.g. on limit) is directly
|
|
158
|
+
// visible here. Swapping is ROUTINE on such gateways, not an incident: a swap can
|
|
159
|
+
// look like a short stall (cold prompt cache re-reads the context once) and providers
|
|
160
|
+
// expose slightly different tools. So this is ONE line, cooldown-bounded — never a
|
|
161
|
+
// per-swap directive block — telling the session the only two things that matter:
|
|
162
|
+
// keep working through swaps, and on a tool error after a swap, re-check the tool and
|
|
163
|
+
// call it again instead of stopping. Runs before the ratio early-exit because a swap
|
|
164
|
+
// can happen at any context level.
|
|
165
|
+
const RECENT_MODEL_WINDOW = 40;
|
|
166
|
+
const SWAP_NOTE_COOLDOWN_MS = 60 * 60 * 1000;
|
|
167
|
+
const modelSequence = [];
|
|
168
|
+
for (let i = start; i < lines.length; i += 1) {
|
|
169
|
+
let entry;
|
|
170
|
+
try {
|
|
171
|
+
entry = JSON.parse(lines[i]);
|
|
172
|
+
} catch {
|
|
173
|
+
continue;
|
|
174
|
+
}
|
|
175
|
+
if (entry?.isSidechain || entry?.type !== 'assistant') continue;
|
|
176
|
+
const model = entry?.message?.model;
|
|
177
|
+
if (typeof model === 'string' && model.trim()) modelSequence.push(model.trim());
|
|
178
|
+
}
|
|
179
|
+
const recentModels = modelSequence.slice(-RECENT_MODEL_WINDOW);
|
|
180
|
+
const currentModel = recentModels[recentModels.length - 1] || null;
|
|
181
|
+
let previousModel = null;
|
|
182
|
+
for (let i = recentModels.length - 2; i >= 0; i -= 1) {
|
|
183
|
+
if (recentModels[i] !== currentModel) {
|
|
184
|
+
previousModel = recentModels[i];
|
|
185
|
+
break;
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
const swapDetected = Boolean(currentModel && previousModel);
|
|
189
|
+
if (swapDetected && persistedState.lastSwapModel !== currentModel) {
|
|
190
|
+
persistedState.lastSwapModel = currentModel;
|
|
191
|
+
const withinSwapCooldown = (Date.now() - Number(persistedState.lastSwapNoteAt || 0)) < SWAP_NOTE_COOLDOWN_MS;
|
|
192
|
+
if (!withinSwapCooldown) {
|
|
193
|
+
persistedState.lastSwapNoteAt = Date.now();
|
|
194
|
+
writeGuardState(persistedState);
|
|
195
|
+
process.stdout.write(
|
|
196
|
+
`UKIT GATEWAY — routine backend model swap ("${currentModel}" after "${previousModel}"): keep working, do not restart or re-plan; if a tool call errors after a swap, re-check the tool (providers differ) and simply call it again — never stop over it.\n`,
|
|
197
|
+
);
|
|
198
|
+
} else {
|
|
199
|
+
writeGuardState(persistedState);
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
if (ratio < 0.8) process.exit(0);
|
|
204
|
+
|
|
205
|
+
const pct = Math.round(ratio * 100);
|
|
206
|
+
const phase = ratio >= 1 ? 'hard' : 'soft';
|
|
207
|
+
|
|
208
|
+
// Debounce the directive block so it does not re-print in full every single prompt
|
|
209
|
+
// while the phase is unchanged — that would just be more tokens added to the same
|
|
210
|
+
// oversized context it is warning about. Re-arms on phase change (soft -> hard) or
|
|
211
|
+
// after the cooldown, and resets naturally once a real compaction/new session drops
|
|
212
|
+
// the live transcript back under 0.8, since this whole branch exits early above.
|
|
213
|
+
const COOLDOWN_MS = phase === 'hard' ? 3 * 60 * 1000 : 8 * 60 * 1000;
|
|
214
|
+
|
|
163
215
|
const now = Date.now();
|
|
164
216
|
const withinCooldown = guardState.lastPhase === phase
|
|
165
217
|
&& (now - Number(guardState.lastActionAt || 0)) < COOLDOWN_MS;
|
|
@@ -206,7 +258,7 @@ if (withinCooldown) {
|
|
|
206
258
|
if (sidechainEntries > 0) {
|
|
207
259
|
lines_out.push(`Note: ${sidechainEntries} subagent entries in this stretch. Keep concurrency at or below handoff.maxParallelAgents and keep returns short; do not widen the batch while this warning stands.`);
|
|
208
260
|
}
|
|
209
|
-
writeGuardState({ lastPhase: phase, lastActionAt: now });
|
|
261
|
+
writeGuardState({ ...persistedState, lastPhase: phase, lastActionAt: now });
|
|
210
262
|
} else {
|
|
211
263
|
lines_out.push('ACTION THIS TURN, before starting new investigation/subagents/pipeline phases:');
|
|
212
264
|
lines_out.push('1) Persist current progress now (update docs/STATUS.md, e.g. via the update-status skill) so nothing is lost.');
|
|
@@ -215,7 +267,7 @@ if (withinCooldown) {
|
|
|
215
267
|
if (sidechainEntries > 0) {
|
|
216
268
|
lines_out.push(`Note: ${sidechainEntries} subagent entries in this stretch — each teammate carries its own context window, and every finished report is injected back here, so running many at once is the fastest way to overflow this session. Avoid spawning more until context drops back under the cap.`);
|
|
217
269
|
}
|
|
218
|
-
writeGuardState({ lastPhase: phase, lastActionAt: now });
|
|
270
|
+
writeGuardState({ ...persistedState, lastPhase: phase, lastActionAt: now });
|
|
219
271
|
}
|
|
220
272
|
|
|
221
273
|
process.stdout.write(`${lines_out.join('\n')}\n`);
|
|
@@ -41,6 +41,33 @@ STATE_FILE="$PROJECT_ROOT/.claude/ukit/skill-router-state.json"
|
|
|
41
41
|
HOOK_DIR="$(cd "$(dirname "$0")" && pwd)"
|
|
42
42
|
THRESHOLD_SCRIPT="$HOOK_DIR/../ukit/runtime/compact-threshold.mjs"
|
|
43
43
|
|
|
44
|
+
# Sidechain early-exit: PreToolUse/PostToolUse payloads carry agent_id/agent_type only
|
|
45
|
+
# when the tool call runs INSIDE a subagent (documented marker). A subagent's Edit/Bash
|
|
46
|
+
# must not re-key or re-classify the MAIN request's route — that churn swapped
|
|
47
|
+
# completionEvidence mid-request and made the main Stop gate demand the wrong evidence.
|
|
48
|
+
# Fails OPEN: any parse problem, missing field, or node error falls through to normal
|
|
49
|
+
# routing. Pure-bash pre-check first: the node guard can only match when the literal
|
|
50
|
+
# "agent_id" key appears, so ordinary calls never pay the node spawn.
|
|
51
|
+
case "$INPUT" in
|
|
52
|
+
*'"agent_id"'*|*'"agent_type"'*)
|
|
53
|
+
if printf '%s' "$INPUT" | node -e '
|
|
54
|
+
const chunks = [];
|
|
55
|
+
process.stdin.on("data", (c) => chunks.push(c));
|
|
56
|
+
process.stdin.on("end", () => {
|
|
57
|
+
try {
|
|
58
|
+
const payload = JSON.parse(Buffer.concat(chunks).toString("utf8") || "{}");
|
|
59
|
+
process.exit(payload && typeof payload === "object"
|
|
60
|
+
&& (payload.agent_id || payload.agent_type) ? 0 : 1);
|
|
61
|
+
} catch {
|
|
62
|
+
process.exit(1);
|
|
63
|
+
}
|
|
64
|
+
});
|
|
65
|
+
' >/dev/null 2>&1; then
|
|
66
|
+
exit 0
|
|
67
|
+
fi
|
|
68
|
+
;;
|
|
69
|
+
esac
|
|
70
|
+
|
|
44
71
|
INPUT="$INPUT" PROJECT_ROOT="$PROJECT_ROOT" STATE_FILE="$STATE_FILE" HOOK_DIR="$HOOK_DIR" node <<'NODE'
|
|
45
72
|
const fs = require('fs');
|
|
46
73
|
const path = require('path');
|
|
@@ -175,6 +175,45 @@ function appendReceipt(receipts, receipt) {
|
|
|
175
175
|
return [...(receipts || []), compactReceipt(receipt)].slice(-MAX_RECEIPTS);
|
|
176
176
|
}
|
|
177
177
|
|
|
178
|
+
// A re-key within the same logical request must not drop the evidence the request
|
|
179
|
+
// already produced — that turned one finished task into a cap-exhausted forced stop.
|
|
180
|
+
// A genuinely different prompt keeps a clean slate so old work never satisfies a new
|
|
181
|
+
// request's completion gate.
|
|
182
|
+
function carriedEvidenceLedger(fresh, current) {
|
|
183
|
+
if (!current || !fresh.promptKey || !current.promptKey || current.promptKey !== fresh.promptKey) {
|
|
184
|
+
return fresh;
|
|
185
|
+
}
|
|
186
|
+
return {
|
|
187
|
+
...fresh,
|
|
188
|
+
sourceSucceeded: fresh.sourceSucceeded || current.sourceSucceeded === true,
|
|
189
|
+
// targeted-verification evidence (2.3.0) must survive the same re-key carry as the
|
|
190
|
+
// coarse flags, or the stricter gates re-demand evidence mid-request — the exact
|
|
191
|
+
// stall this carry exists to prevent.
|
|
192
|
+
targetedVerificationSucceeded: fresh.targetedVerificationSucceeded
|
|
193
|
+
|| current.targetedVerificationSucceeded === true,
|
|
194
|
+
sourceFiles: [...new Set([...(current.sourceFiles || []), ...(fresh.sourceFiles || [])])]
|
|
195
|
+
.slice(-MAX_SOURCE_FILES),
|
|
196
|
+
writeAttempted: fresh.writeAttempted || current.writeAttempted === true,
|
|
197
|
+
writeSucceeded: fresh.writeSucceeded || current.writeSucceeded === true,
|
|
198
|
+
verificationAttempted: fresh.verificationAttempted || current.verificationAttempted === true,
|
|
199
|
+
verificationSucceeded: fresh.verificationSucceeded || current.verificationSucceeded === true,
|
|
200
|
+
verificationFailed: fresh.verificationFailed || current.verificationFailed === true,
|
|
201
|
+
receipts: [...(current.receipts || [])].slice(-MAX_RECEIPTS),
|
|
202
|
+
};
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
// The router rebuilds requestKey per tool call (commandText/targetFile are part of the
|
|
206
|
+
// key), so requestKey changes mid-request on every verification command — and a subagent's
|
|
207
|
+
// Edit can additionally re-classify executionMode/taskType for the same user prompt. The
|
|
208
|
+
// logical request identity is therefore the prompt text alone: hash it so completion
|
|
209
|
+
// evidence survives any re-key/re-route within one request without leaking across requests.
|
|
210
|
+
// States with no recorded prompt text get no identity and never carry evidence.
|
|
211
|
+
function evidencePromptKey(routeState) {
|
|
212
|
+
const promptText = String(routeState?.routingContext?.lastExplicitUserPromptText || '').trim();
|
|
213
|
+
if (!promptText) return null;
|
|
214
|
+
return `prompt-${crypto.createHash('sha256').update(promptText).digest('hex').slice(0, 20)}`;
|
|
215
|
+
}
|
|
216
|
+
|
|
178
217
|
function freshLedger(payload, routeState, harness) {
|
|
179
218
|
return {
|
|
180
219
|
version: LEDGER_VERSION,
|
|
@@ -183,6 +222,7 @@ function freshLedger(payload, routeState, harness) {
|
|
|
183
222
|
transcriptPath: payload.transcript_path || payload.transcriptPath || null,
|
|
184
223
|
harness,
|
|
185
224
|
requestKey: routeState?.requestKey || null,
|
|
225
|
+
promptKey: evidencePromptKey(routeState),
|
|
186
226
|
routeFingerprint: routeState?.fingerprint || null,
|
|
187
227
|
sourceSucceeded: false,
|
|
188
228
|
sourceFiles: [],
|
|
@@ -209,7 +249,7 @@ export async function recordExecutionReceipt({
|
|
|
209
249
|
const routeState = await readRouteState(projectRoot);
|
|
210
250
|
const current = await readExecutionLedger(projectRoot, payload);
|
|
211
251
|
const ledger = !current || current.requestKey !== (routeState?.requestKey || null)
|
|
212
|
-
? freshLedger(payload, routeState, harness)
|
|
252
|
+
? carriedEvidenceLedger(freshLedger(payload, routeState, harness), current)
|
|
213
253
|
: { ...current, harness: current.harness || harness };
|
|
214
254
|
|
|
215
255
|
const failed = explicitError(payload);
|
|
@@ -453,12 +493,31 @@ async function main() {
|
|
|
453
493
|
const state = await readRouteState(projectRoot);
|
|
454
494
|
const ledger = await readExecutionLedger(projectRoot, payload) || {};
|
|
455
495
|
const result = evaluateCompletion({ state, ledger });
|
|
496
|
+
|
|
497
|
+
// Claude Code invokes Stop again after a Stop hook blocks the first stop. Re-blocking
|
|
498
|
+
// that recovery turn creates a self-sustaining loop, so let it end normally instead.
|
|
499
|
+
// If work still lacks evidence, surface the recovery reason to the user rather than
|
|
500
|
+
// silently ending after the automatic continuation.
|
|
501
|
+
if (payload.stop_hook_active === true) {
|
|
502
|
+
if (result.continue || result.capped || result.notify) {
|
|
503
|
+
const recoveryReason = result.reason
|
|
504
|
+
|| 'UKit completion gate reached its continuation limit; unfinished work was not retried again.';
|
|
505
|
+
process.stdout.write(`${JSON.stringify({
|
|
506
|
+
systemMessage: `UKit stopped automatic recovery after one continuation: ${recoveryReason}`,
|
|
507
|
+
})}\n`);
|
|
508
|
+
}
|
|
509
|
+
return;
|
|
510
|
+
}
|
|
511
|
+
|
|
456
512
|
if (result.continue) {
|
|
457
513
|
if (result.finalNotice) await markNotified(projectRoot, payload, ledger);
|
|
458
514
|
else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null);
|
|
459
515
|
process.stdout.write(`${JSON.stringify({ decision: 'block', reason: result.reason })}\n`);
|
|
460
516
|
} else if (result.capped || result.notify) {
|
|
517
|
+
// Non-blocking endings (non-gated modes, or cap reached after the final notice) must
|
|
518
|
+
// still tell the user what is unfinished — a silent end is indistinguishable from a stall.
|
|
461
519
|
process.stderr.write(`[ukit-completion] ${result.reason}\n`);
|
|
520
|
+
process.stdout.write(`${JSON.stringify({ systemMessage: result.reason })}\n`);
|
|
462
521
|
}
|
|
463
522
|
}
|
|
464
523
|
}
|