@ngockhoale/ukit 2.2.16 → 2.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +104 -0
- package/manifests/platform.full.yaml +11 -0
- package/package.json +1 -1
- package/src/core/executionContracts.js +130 -0
- package/src/core/runtimeConfig.js +13 -50
- package/src/index/taskRouting.js +10 -103
- package/templates/.claude/agents/ukit-vision-analyst.md +32 -21
- package/templates/.claude/hooks/context-window-guard.sh +66 -14
- package/templates/.claude/hooks/protect-files.sh +1 -0
- package/templates/.claude/hooks/sensitive-data-guard.sh +269 -0
- package/templates/.claude/hooks/skill-router.sh +33 -0
- package/templates/.claude/hooks/vision-router.sh +63 -37
- package/templates/.claude/settings.json +17 -1
- package/templates/.claude/ukit/index/extract-image.mjs +18 -8
- package/templates/.claude/ukit/index/route-task.mjs +53 -2
- package/templates/.claude/ukit/runtime/execution-ledger.mjs +180 -22
- package/templates/.claude/ukit/runtime/hook-chain-runner.mjs +8 -4
- package/templates/.omp/hooks/pre/ukit-bridge.js +15 -5
- package/templates/AGENTS.md +2 -2
- package/templates/CLAUDE.md +1 -1
- package/templates/ukit/storage/config.json +8 -7
- package/src/core/router/advisor.js +0 -42
- package/src/core/router/router.js +0 -180
- package/src/core/validation/confidence.js +0 -89
- package/src/core/validation/validator.js +0 -165
- package/templates/docs/INSTALL.md +0 -115
- package/templates/docs/STATUS.md +0 -81
- package/templates/docs/TASKS.md +0 -79
- package/templates/docs/UKIT_USAGE_GUIDE.md +0 -163
|
@@ -6,9 +6,10 @@
|
|
|
6
6
|
* blocks, so pasted images are invisible unless this script first decodes them
|
|
7
7
|
* to disk. This is the single home of the sha256 key: `pending-<sha>` markers
|
|
8
8
|
* (written by --mark-pending) and `<sha>.<ext>` image files (written by a normal
|
|
9
|
-
* extract) MUST agree exactly, or
|
|
10
|
-
* paths call the same hashPayload() function on the exact
|
|
11
|
-
* appears in the transcript — no trimming, no normalising,
|
|
9
|
+
* extract) MUST agree exactly, or analysis receipts can never be matched back to
|
|
10
|
+
* their markers. Both paths call the same hashPayload() function on the exact
|
|
11
|
+
* base64 string as it appears in the transcript — no trimming, no normalising,
|
|
12
|
+
* no decoding first.
|
|
12
13
|
*
|
|
13
14
|
* Usage:
|
|
14
15
|
* node extract-image.mjs [options]
|
|
@@ -323,14 +324,23 @@ function main() {
|
|
|
323
324
|
// A sha with an existing analyzed-<sha>.json receipt is a closed, content-addressed case:
|
|
324
325
|
// identical bytes already produced that analysis, so skipping it loses no information.
|
|
325
326
|
const isAlreadyAnalyzed = (sha) => fs.existsSync(path.join(markerDir, `analyzed-${sha}.json`));
|
|
326
|
-
const
|
|
327
|
-
const
|
|
327
|
+
const allSelected = selectImages(blocks, opts.limit);
|
|
328
|
+
const allRefs = resolveRefs(opts.refs);
|
|
329
|
+
const selected = allSelected.filter((img) => !isAlreadyAnalyzed(img.sha));
|
|
330
|
+
const refs = allRefs.filter((ref) => !isAlreadyAnalyzed(ref.sha));
|
|
331
|
+
// Closed cases reported back so callers (vision-router.sh) can stay silent on them:
|
|
332
|
+
// a sha with an analyzed-<sha>.json receipt needs no new dispatch, and without this
|
|
333
|
+
// list the caller cannot tell "already analyzed" from "never resolved".
|
|
334
|
+
const alreadyAnalyzed = [
|
|
335
|
+
...allSelected.filter((img) => isAlreadyAnalyzed(img.sha)).map((img) => ({ sha: img.sha, source: 'pasted' })),
|
|
336
|
+
...allRefs.filter((ref) => isAlreadyAnalyzed(ref.sha)).map((ref) => ({ sha: ref.sha, source: ref.source, ref: ref.value })),
|
|
337
|
+
];
|
|
328
338
|
|
|
329
339
|
// --detect is report-only, so it must not short-circuit --mark-pending when both are
|
|
330
340
|
// passed together. vision-router.sh combines them; short-circuiting here would arm
|
|
331
|
-
// nothing and leave
|
|
341
|
+
// nothing and leave every image unmarked.
|
|
332
342
|
if (opts.detect && !opts.markPending) {
|
|
333
|
-
printResult(opts, { imageCount: selected.length + refs.length, sessionId, images: [] });
|
|
343
|
+
printResult(opts, { imageCount: selected.length + refs.length, sessionId, images: [], alreadyAnalyzed });
|
|
334
344
|
return;
|
|
335
345
|
}
|
|
336
346
|
|
|
@@ -374,7 +384,7 @@ function main() {
|
|
|
374
384
|
}
|
|
375
385
|
}
|
|
376
386
|
pruneMarkerDirs(outDir);
|
|
377
|
-
printResult(opts, { imageCount: written.length, sessionId, images: written });
|
|
387
|
+
printResult(opts, { imageCount: written.length, sessionId, images: written, alreadyAnalyzed });
|
|
378
388
|
return;
|
|
379
389
|
}
|
|
380
390
|
|
|
@@ -579,6 +579,9 @@ function printRouteState(state) {
|
|
|
579
579
|
if (state.routeSummary?.executionContract?.modelTier) {
|
|
580
580
|
console.log(`modelTier: ${state.routeSummary.executionContract.modelTier}`);
|
|
581
581
|
}
|
|
582
|
+
if (state.routeSummary?.tierLane?.instruction) {
|
|
583
|
+
console.log(`tierLane: ${state.routeSummary.tierLane.instruction}`);
|
|
584
|
+
}
|
|
582
585
|
if (state.routeSummary?.escalatedTier) {
|
|
583
586
|
console.log(`escalatedTier: ${state.routeSummary.escalatedTier}`);
|
|
584
587
|
if (state.routeSummary.escalationReason) {
|
|
@@ -1475,6 +1478,15 @@ function buildRouteSummary({
|
|
|
1475
1478
|
const primaryTargets = summarizeCompactList(preview.primaryTargets ?? [], 2);
|
|
1476
1479
|
const relatedTests = summarizeCompactList(preview.relatedTests ?? [], 2);
|
|
1477
1480
|
const styleFiles = summarizeCompactList(preview.styleFiles ?? [], 1);
|
|
1481
|
+
// WS-C target-aware evidence: the completion ledger checks impact-evidence against these
|
|
1482
|
+
// files, so carry the full resolver list (target + primaryTargets + relatedTests), bounded
|
|
1483
|
+
// for compact route state. Absent when the route has no indexed context — the ledger then
|
|
1484
|
+
// falls back to its legacy any-read behaviour.
|
|
1485
|
+
const expectedSourceFiles = unique([
|
|
1486
|
+
...(routingContext.targetFile ? [routingContext.targetFile] : []),
|
|
1487
|
+
...(preview.primaryTargets ?? []),
|
|
1488
|
+
...(preview.relatedTests ?? []),
|
|
1489
|
+
].filter((filePath) => typeof filePath === 'string' && filePath.trim())).slice(0, 8);
|
|
1478
1490
|
const primaryCommands = unique(verificationRecommendation?.commands ?? []);
|
|
1479
1491
|
const fallbackCommands = unique(verificationRecommendation?.fallbackCommands ?? []);
|
|
1480
1492
|
const preferredOrder = unique(
|
|
@@ -1497,6 +1509,7 @@ function buildRouteSummary({
|
|
|
1497
1509
|
executionCandidates,
|
|
1498
1510
|
});
|
|
1499
1511
|
const executionContract = buildExecutionContract(executionMode);
|
|
1512
|
+
const tierLane = buildTierLane({ modelTier: executionContract?.modelTier ?? null });
|
|
1500
1513
|
const completionState = buildCompletionState({
|
|
1501
1514
|
executionMode,
|
|
1502
1515
|
verificationRecommendation,
|
|
@@ -1539,8 +1552,10 @@ function buildRouteSummary({
|
|
|
1539
1552
|
executionCandidates,
|
|
1540
1553
|
approachSelector,
|
|
1541
1554
|
executionContract,
|
|
1555
|
+
tierLane,
|
|
1542
1556
|
completionState,
|
|
1543
1557
|
continuationState,
|
|
1558
|
+
...(expectedSourceFiles.length > 0 ? { expectedSourceFiles } : {}),
|
|
1544
1559
|
intentMode: routingContext.intentMode ?? null,
|
|
1545
1560
|
handoffFile,
|
|
1546
1561
|
delegateHint: delegationRecommendation?.hint ?? null,
|
|
@@ -1894,9 +1909,10 @@ function buildExecutionContract(executionMode = null) {
|
|
|
1894
1909
|
maxReadPasses: 2,
|
|
1895
1910
|
maxContextPulls: 1,
|
|
1896
1911
|
verificationPolicy: 'targeted-if-covered',
|
|
1897
|
-
completionRule: 'require-write',
|
|
1912
|
+
completionRule: 'require-write-and-verification',
|
|
1898
1913
|
delegationPolicy: 'disallow-by-default',
|
|
1899
|
-
completionEvidence: ['write-evidence'],
|
|
1914
|
+
completionEvidence: ['write-evidence', 'verification-evidence'],
|
|
1915
|
+
postEditReviewPolicy: 'sidecar-non-blocking',
|
|
1900
1916
|
},
|
|
1901
1917
|
'find-cause': {
|
|
1902
1918
|
modelTier: 'code',
|
|
@@ -1915,6 +1931,7 @@ function buildExecutionContract(executionMode = null) {
|
|
|
1915
1931
|
delegationPolicy: 'allow-qualified-sidecar',
|
|
1916
1932
|
completionEvidence: ['write-evidence', 'verification-evidence'],
|
|
1917
1933
|
mirrorConsistencyRequired: true,
|
|
1934
|
+
postEditReviewPolicy: 'sidecar-non-blocking',
|
|
1918
1935
|
},
|
|
1919
1936
|
'map-impact': {
|
|
1920
1937
|
modelTier: 'code',
|
|
@@ -2182,10 +2199,36 @@ function applyEscalationToRouteSummary(routeSummary = null, { config = null, tar
|
|
|
2182
2199
|
delete routeSummary.escalatedTier;
|
|
2183
2200
|
delete routeSummary.escalationReason;
|
|
2184
2201
|
}
|
|
2202
|
+
// Recomputed here (not only in buildRouteSummary) so the lane always reflects the final
|
|
2203
|
+
// tier after escalation, on every path that emits a route summary.
|
|
2204
|
+
routeSummary.tierLane = buildTierLane({
|
|
2205
|
+
modelTier: routeSummary.executionContract?.modelTier ?? null,
|
|
2206
|
+
escalatedTier: escalation?.escalatedTier ?? null,
|
|
2207
|
+
});
|
|
2185
2208
|
|
|
2186
2209
|
return routeSummary;
|
|
2187
2210
|
}
|
|
2188
2211
|
|
|
2212
|
+
// WS-E tier-role wiring: a contract tier only takes effect when the executing session hands
|
|
2213
|
+
// the work to an agent whose definition binds that tier — otherwise every session behaves as
|
|
2214
|
+
// if it only had one model. `code` is the DEFAULT implementation lane, so a code-tier contract
|
|
2215
|
+
// produces no instruction (inline execution on a code session is already correct, and silence
|
|
2216
|
+
// keeps the hook flow smooth). Labels are stable tier roles (lite/code/smart), never provider
|
|
2217
|
+
// or model identity. Vision never appears here — it is a capability lane, not a cost tier.
|
|
2218
|
+
function buildTierLane({ modelTier = null, escalatedTier = null } = {}) {
|
|
2219
|
+
const tier = escalatedTier ?? modelTier;
|
|
2220
|
+
if (!tier || tier === 'vision') {
|
|
2221
|
+
return null;
|
|
2222
|
+
}
|
|
2223
|
+
return {
|
|
2224
|
+
tier,
|
|
2225
|
+
defaultLane: tier === 'code',
|
|
2226
|
+
instruction: tier === 'code'
|
|
2227
|
+
? null
|
|
2228
|
+
: `Hand this task to an agent bound to the ${tier} tier instead of doing it inline, unless this session already runs on the ${tier} tier.`,
|
|
2229
|
+
};
|
|
2230
|
+
}
|
|
2231
|
+
|
|
2189
2232
|
function advanceContinuationState(continuationState = null, previousContinuationState = null, thresholds = null) {
|
|
2190
2233
|
if (!continuationState || typeof continuationState !== 'object') {
|
|
2191
2234
|
return continuationState;
|
|
@@ -2875,6 +2918,14 @@ function compactRouteSummary(routeSummary = null) {
|
|
|
2875
2918
|
completionState: routeSummary.completionState ?? null,
|
|
2876
2919
|
continuationState: routeSummary.continuationState ?? null,
|
|
2877
2920
|
delegateHint: routeSummary.delegateHint ?? null,
|
|
2921
|
+
// WS-E tier-role binding: consumers need the hand-off instruction, not just the raw tier.
|
|
2922
|
+
...(routeSummary.tierLane ? { tierLane: routeSummary.tierLane } : {}),
|
|
2923
|
+
// WS-C target-aware evidence: the completion ledger reads these from the PERSISTED route
|
|
2924
|
+
// state — without this spread the CLI path never delivers them and the gate falls back
|
|
2925
|
+
// to legacy any-read behaviour.
|
|
2926
|
+
...(Array.isArray(routeSummary.expectedSourceFiles) && routeSummary.expectedSourceFiles.length > 0 ? {
|
|
2927
|
+
expectedSourceFiles: routeSummary.expectedSourceFiles,
|
|
2928
|
+
} : {}),
|
|
2878
2929
|
nextActionType: routeSummary.nextActionType ?? null,
|
|
2879
2930
|
nextActionCommand: routeSummary.nextActionCommand ?? null,
|
|
2880
2931
|
helperHint: routeSummary.helperHint ?? null,
|
|
@@ -8,6 +8,7 @@ import { fileURLToPath } from 'node:url';
|
|
|
8
8
|
|
|
9
9
|
const LEDGER_VERSION = 1;
|
|
10
10
|
const MAX_RECEIPTS = 24;
|
|
11
|
+
const MAX_SOURCE_FILES = 16;
|
|
11
12
|
const MAX_CONTINUATIONS = 6;
|
|
12
13
|
const IMPLEMENT_MODES = new Set([
|
|
13
14
|
'tiny-fix',
|
|
@@ -130,7 +131,7 @@ function compactReceipt(receipt) {
|
|
|
130
131
|
kind: receipt.kind,
|
|
131
132
|
success: receipt.success,
|
|
132
133
|
};
|
|
133
|
-
for (const key of ['toolName', 'toolUseId', 'file', 'command', 'exitCode', 'error']) {
|
|
134
|
+
for (const key of ['toolName', 'toolUseId', 'file', 'command', 'exitCode', 'scope', 'error']) {
|
|
134
135
|
if (receipt[key] !== undefined && receipt[key] !== null && receipt[key] !== '') {
|
|
135
136
|
compact[key] = receipt[key];
|
|
136
137
|
}
|
|
@@ -138,10 +139,81 @@ function compactReceipt(receipt) {
|
|
|
138
139
|
return compact;
|
|
139
140
|
}
|
|
140
141
|
|
|
142
|
+
/**
|
|
143
|
+
* Target-aware matching between a receipt's file and a routed expected file. Both sides
|
|
144
|
+
* may disagree about absolute vs relative spelling (hook payloads carry absolute paths,
|
|
145
|
+
* the router carries repo-relative ones), so exact equality, path-suffix, and bare-name
|
|
146
|
+
* matches all count.
|
|
147
|
+
*/
|
|
148
|
+
function fileMatchesExpected(receiptFile, expectedFile) {
|
|
149
|
+
if (!receiptFile || !expectedFile) return false;
|
|
150
|
+
const receipt = String(receiptFile).replace(/\\/g, '/').replace(/^\.\//, '');
|
|
151
|
+
const expected = String(expectedFile).replace(/\\/g, '/').replace(/^\.\//, '');
|
|
152
|
+
if (receipt === expected) return true;
|
|
153
|
+
if (receipt.endsWith(`/${expected}`) || expected.endsWith(`/${receipt}`)) return true;
|
|
154
|
+
if (!expected.includes('/')) {
|
|
155
|
+
const base = receipt.split('/').pop();
|
|
156
|
+
return base === expected;
|
|
157
|
+
}
|
|
158
|
+
return false;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
function matchesRoutedCommand(command, routedCommand) {
|
|
162
|
+
const receipt = String(command || '').trim();
|
|
163
|
+
const routed = String(routedCommand || '').trim();
|
|
164
|
+
if (!receipt || !routed) return false;
|
|
165
|
+
return receipt === routed || receipt.startsWith(routed) || routed.startsWith(receipt);
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
function routedVerificationCommands(routeSummary = {}) {
|
|
169
|
+
const preferred = Array.isArray(routeSummary?.preferredOrder) ? routeSummary.preferredOrder : [];
|
|
170
|
+
const primary = Array.isArray(routeSummary?.primaryCommands) ? routeSummary.primaryCommands : [];
|
|
171
|
+
return preferred.length > 0 ? preferred : primary;
|
|
172
|
+
}
|
|
173
|
+
|
|
141
174
|
function appendReceipt(receipts, receipt) {
|
|
142
175
|
return [...(receipts || []), compactReceipt(receipt)].slice(-MAX_RECEIPTS);
|
|
143
176
|
}
|
|
144
177
|
|
|
178
|
+
// A re-key within the same logical request must not drop the evidence the request
|
|
179
|
+
// already produced — that turned one finished task into a cap-exhausted forced stop.
|
|
180
|
+
// A genuinely different prompt keeps a clean slate so old work never satisfies a new
|
|
181
|
+
// request's completion gate.
|
|
182
|
+
function carriedEvidenceLedger(fresh, current) {
|
|
183
|
+
if (!current || !fresh.promptKey || !current.promptKey || current.promptKey !== fresh.promptKey) {
|
|
184
|
+
return fresh;
|
|
185
|
+
}
|
|
186
|
+
return {
|
|
187
|
+
...fresh,
|
|
188
|
+
sourceSucceeded: fresh.sourceSucceeded || current.sourceSucceeded === true,
|
|
189
|
+
// targeted-verification evidence (2.3.0) must survive the same re-key carry as the
|
|
190
|
+
// coarse flags, or the stricter gates re-demand evidence mid-request — the exact
|
|
191
|
+
// stall this carry exists to prevent.
|
|
192
|
+
targetedVerificationSucceeded: fresh.targetedVerificationSucceeded
|
|
193
|
+
|| current.targetedVerificationSucceeded === true,
|
|
194
|
+
sourceFiles: [...new Set([...(current.sourceFiles || []), ...(fresh.sourceFiles || [])])]
|
|
195
|
+
.slice(-MAX_SOURCE_FILES),
|
|
196
|
+
writeAttempted: fresh.writeAttempted || current.writeAttempted === true,
|
|
197
|
+
writeSucceeded: fresh.writeSucceeded || current.writeSucceeded === true,
|
|
198
|
+
verificationAttempted: fresh.verificationAttempted || current.verificationAttempted === true,
|
|
199
|
+
verificationSucceeded: fresh.verificationSucceeded || current.verificationSucceeded === true,
|
|
200
|
+
verificationFailed: fresh.verificationFailed || current.verificationFailed === true,
|
|
201
|
+
receipts: [...(current.receipts || [])].slice(-MAX_RECEIPTS),
|
|
202
|
+
};
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
// The router rebuilds requestKey per tool call (commandText/targetFile are part of the
|
|
206
|
+
// key), so requestKey changes mid-request on every verification command — and a subagent's
|
|
207
|
+
// Edit can additionally re-classify executionMode/taskType for the same user prompt. The
|
|
208
|
+
// logical request identity is therefore the prompt text alone: hash it so completion
|
|
209
|
+
// evidence survives any re-key/re-route within one request without leaking across requests.
|
|
210
|
+
// States with no recorded prompt text get no identity and never carry evidence.
|
|
211
|
+
function evidencePromptKey(routeState) {
|
|
212
|
+
const promptText = String(routeState?.routingContext?.lastExplicitUserPromptText || '').trim();
|
|
213
|
+
if (!promptText) return null;
|
|
214
|
+
return `prompt-${crypto.createHash('sha256').update(promptText).digest('hex').slice(0, 20)}`;
|
|
215
|
+
}
|
|
216
|
+
|
|
145
217
|
function freshLedger(payload, routeState, harness) {
|
|
146
218
|
return {
|
|
147
219
|
version: LEDGER_VERSION,
|
|
@@ -150,8 +222,10 @@ function freshLedger(payload, routeState, harness) {
|
|
|
150
222
|
transcriptPath: payload.transcript_path || payload.transcriptPath || null,
|
|
151
223
|
harness,
|
|
152
224
|
requestKey: routeState?.requestKey || null,
|
|
225
|
+
promptKey: evidencePromptKey(routeState),
|
|
153
226
|
routeFingerprint: routeState?.fingerprint || null,
|
|
154
227
|
sourceSucceeded: false,
|
|
228
|
+
sourceFiles: [],
|
|
155
229
|
writeAttempted: false,
|
|
156
230
|
writeSucceeded: false,
|
|
157
231
|
verificationAttempted: false,
|
|
@@ -175,7 +249,7 @@ export async function recordExecutionReceipt({
|
|
|
175
249
|
const routeState = await readRouteState(projectRoot);
|
|
176
250
|
const current = await readExecutionLedger(projectRoot, payload);
|
|
177
251
|
const ledger = !current || current.requestKey !== (routeState?.requestKey || null)
|
|
178
|
-
? freshLedger(payload, routeState, harness)
|
|
252
|
+
? carriedEvidenceLedger(freshLedger(payload, routeState, harness), current)
|
|
179
253
|
: { ...current, harness: current.harness || harness };
|
|
180
254
|
|
|
181
255
|
const failed = explicitError(payload);
|
|
@@ -194,6 +268,9 @@ export async function recordExecutionReceipt({
|
|
|
194
268
|
receipt.kind = 'source';
|
|
195
269
|
receipt.file = toolInput.file_path || toolInput.path || null;
|
|
196
270
|
ledger.sourceSucceeded ||= success;
|
|
271
|
+
if (success && receipt.file) {
|
|
272
|
+
ledger.sourceFiles = [...new Set([...(ledger.sourceFiles || []), receipt.file])].slice(-MAX_SOURCE_FILES);
|
|
273
|
+
}
|
|
197
274
|
} else if (toolName === 'Edit' || toolName === 'Write') {
|
|
198
275
|
receipt.kind = 'write';
|
|
199
276
|
receipt.file = toolInput.file_path || toolInput.path || toolInput.paths?.[0] || null;
|
|
@@ -202,6 +279,17 @@ export async function recordExecutionReceipt({
|
|
|
202
279
|
} else if (toolName === 'Bash' && isVerificationCommand(toolInput.command)) {
|
|
203
280
|
receipt.kind = 'verification';
|
|
204
281
|
receipt.command = String(toolInput.command || '').trim();
|
|
282
|
+
// WS-C routed-verification receipt: a command counts as "targeted" when it matches the
|
|
283
|
+
// routed plan (preferredOrder / primaryCommands). Off-plan verification still records
|
|
284
|
+
// as broad — counted only when the route carried no commands at all.
|
|
285
|
+
const routedCommands = routedVerificationCommands(routeState?.routeSummary);
|
|
286
|
+
if (routedCommands.length > 0) {
|
|
287
|
+
const matched = routedCommands.find((cmd) => matchesRoutedCommand(receipt.command, cmd));
|
|
288
|
+
receipt.scope = matched ? 'targeted' : 'broad';
|
|
289
|
+
if (receipt.scope === 'targeted') {
|
|
290
|
+
ledger.targetedVerificationSucceeded ||= success;
|
|
291
|
+
}
|
|
292
|
+
}
|
|
205
293
|
ledger.verificationAttempted = true;
|
|
206
294
|
ledger.verificationSucceeded ||= success;
|
|
207
295
|
ledger.verificationFailed ||= !success;
|
|
@@ -224,30 +312,66 @@ function requiredEvidence(state = {}) {
|
|
|
224
312
|
return [...new Set(routeSummary.completionState?.missingEvidence || [])];
|
|
225
313
|
}
|
|
226
314
|
|
|
227
|
-
function evidenceSatisfied(evidence, ledger = {}) {
|
|
315
|
+
function evidenceSatisfied(evidence, ledger = {}, routeSummary = {}) {
|
|
228
316
|
if (evidence === 'write-evidence') return ledger.writeSucceeded === true;
|
|
229
|
-
if (evidence === 'verification-evidence')
|
|
230
|
-
|
|
317
|
+
if (evidence === 'verification-evidence') {
|
|
318
|
+
// WS-C: when the route names concrete verification commands, only a receipt that ran
|
|
319
|
+
// one of them counts — an unrelated `yarn test` no longer satisfies the gate.
|
|
320
|
+
const routedCommands = routedVerificationCommands(routeSummary);
|
|
321
|
+
if (routedCommands.length > 0) {
|
|
322
|
+
return ledger.targetedVerificationSucceeded === true;
|
|
323
|
+
}
|
|
324
|
+
return ledger.verificationSucceeded === true;
|
|
325
|
+
}
|
|
326
|
+
if (evidence === 'impact-evidence') {
|
|
327
|
+
// WS-C: when the route names expected source files, a read must cover one of them —
|
|
328
|
+
// ANY read (e.g. docs) no longer satisfies the impact-evidence gate.
|
|
329
|
+
const expectedFiles = Array.isArray(routeSummary?.expectedSourceFiles)
|
|
330
|
+
? routeSummary.expectedSourceFiles
|
|
331
|
+
: [];
|
|
332
|
+
if (expectedFiles.length > 0) {
|
|
333
|
+
const readFiles = Array.isArray(ledger.sourceFiles) ? ledger.sourceFiles : [];
|
|
334
|
+
return readFiles.some((file) => expectedFiles.some((expected) => fileMatchesExpected(file, expected)));
|
|
335
|
+
}
|
|
336
|
+
return ledger.sourceSucceeded === true;
|
|
337
|
+
}
|
|
231
338
|
return false;
|
|
232
339
|
}
|
|
233
340
|
|
|
234
|
-
function recoveryInstruction(missingEvidence, ledger = {}) {
|
|
341
|
+
function recoveryInstruction(missingEvidence, ledger = {}, routeSummary = {}) {
|
|
342
|
+
let instruction = null;
|
|
235
343
|
if (missingEvidence.includes('write-evidence')) {
|
|
236
344
|
if (!ledger.sourceSucceeded) {
|
|
237
|
-
|
|
345
|
+
instruction = 'Pull one bounded indexed source slice, then make the requested Edit/Write in this continuation.';
|
|
346
|
+
} else if (ledger.writeAttempted && !ledger.writeSucceeded) {
|
|
347
|
+
instruction = 'Recover from the failed mutation using the current error, then retry the smallest correct Edit/Write.';
|
|
348
|
+
} else {
|
|
349
|
+
instruction = 'Make the smallest correct Edit/Write now; do not end after more read-only analysis.';
|
|
238
350
|
}
|
|
239
|
-
|
|
240
|
-
|
|
351
|
+
} else if (missingEvidence.includes('verification-evidence')) {
|
|
352
|
+
if (ledger.verificationFailed) {
|
|
353
|
+
instruction = 'Use the latest failed verification output, fix the failure, and rerun targeted verification.';
|
|
354
|
+
} else {
|
|
355
|
+
const routedCommands = routedVerificationCommands(routeSummary);
|
|
356
|
+
if (routedCommands.length > 0 && ledger.verificationSucceeded && !ledger.targetedVerificationSucceeded) {
|
|
357
|
+
instruction = `Verification ran but did not match the routed plan (${routedCommands.slice(0, 3).join('; ')}). Run one of those and inspect its result before stopping.`;
|
|
358
|
+
} else {
|
|
359
|
+
instruction = 'Run the routed targeted verification now and inspect its result before stopping.';
|
|
360
|
+
}
|
|
241
361
|
}
|
|
242
|
-
return 'Make the smallest correct Edit/Write now; do not end after more read-only analysis.';
|
|
243
362
|
}
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
363
|
+
// WS-C: the target hint travels with the reason whenever impact evidence is missing and
|
|
364
|
+
// the route names expected files — it must survive branch precedence above.
|
|
365
|
+
if (missingEvidence.includes('impact-evidence')) {
|
|
366
|
+
const expectedFiles = Array.isArray(routeSummary?.expectedSourceFiles) ? routeSummary.expectedSourceFiles : [];
|
|
367
|
+
if (expectedFiles.length > 0 && !expectedFiles.some(
|
|
368
|
+
(expected) => (ledger.sourceFiles || []).some((file) => fileMatchesExpected(file, expected))
|
|
369
|
+
)) {
|
|
370
|
+
const hint = `Read the routed target files (${expectedFiles.slice(0, 3).join(', ')}) — reads of unrelated files do not count as impact evidence.`;
|
|
371
|
+
instruction = instruction ? `${instruction} ${hint}` : hint;
|
|
247
372
|
}
|
|
248
|
-
return 'Run the routed targeted verification now and inspect its result before stopping.';
|
|
249
373
|
}
|
|
250
|
-
return 'Complete the current routed milestone before stopping.';
|
|
374
|
+
return instruction ?? 'Complete the current routed milestone before stopping.';
|
|
251
375
|
}
|
|
252
376
|
|
|
253
377
|
export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
@@ -263,7 +387,7 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
|
263
387
|
|
|
264
388
|
const sameRequest = !ledger?.requestKey || !state?.requestKey || ledger.requestKey === state.requestKey;
|
|
265
389
|
const effectiveLedger = sameRequest ? ledger : {};
|
|
266
|
-
const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger));
|
|
390
|
+
const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger, routeSummary));
|
|
267
391
|
if (missingEvidence.length === 0) {
|
|
268
392
|
return { continue: false, notify: false, missingEvidence: [] };
|
|
269
393
|
}
|
|
@@ -278,7 +402,15 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
|
278
402
|
};
|
|
279
403
|
}
|
|
280
404
|
|
|
281
|
-
|
|
405
|
+
// Continuation attempts only make sense within the request that minted them: a new
|
|
406
|
+
// routed request in the same session must start with a fresh budget, otherwise a cap
|
|
407
|
+
// exhausted on task A suppresses recovery for task B.
|
|
408
|
+
const staleContinuations = effectiveLedger?.continuationRequestKey
|
|
409
|
+
&& state?.requestKey
|
|
410
|
+
&& effectiveLedger.continuationRequestKey !== state.requestKey;
|
|
411
|
+
const continuationCount = staleContinuations
|
|
412
|
+
? 0
|
|
413
|
+
: Number(effectiveLedger?.continuationCount || 0);
|
|
282
414
|
if (continuationCount >= MAX_CONTINUATIONS) {
|
|
283
415
|
if (effectiveLedger?.notified === true) {
|
|
284
416
|
return {
|
|
@@ -300,7 +432,7 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
|
300
432
|
}
|
|
301
433
|
|
|
302
434
|
const finalAttempt = continuationCount === MAX_CONTINUATIONS - 1;
|
|
303
|
-
const instruction = recoveryInstruction(missingEvidence, effectiveLedger);
|
|
435
|
+
const instruction = recoveryInstruction(missingEvidence, effectiveLedger, routeSummary);
|
|
304
436
|
return {
|
|
305
437
|
continue: true,
|
|
306
438
|
missingEvidence,
|
|
@@ -312,11 +444,18 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
|
312
444
|
};
|
|
313
445
|
}
|
|
314
446
|
|
|
315
|
-
export async function incrementContinuation(projectRoot, payload = {}, ledger = null) {
|
|
447
|
+
export async function incrementContinuation(projectRoot, payload = {}, ledger = null, requestKey = null) {
|
|
316
448
|
const current = ledger || await readExecutionLedger(projectRoot, payload) || freshLedger(payload, null, 'unknown');
|
|
449
|
+
// Mirror evaluateCompletion's staleness rule: a count minted by an earlier request must
|
|
450
|
+
// not be carried into the new request's budget, or the cap fires early (evaluate says 0,
|
|
451
|
+
// persist says 7) and the next request inherits a nearly exhausted budget.
|
|
452
|
+
const stale = requestKey
|
|
453
|
+
&& current?.continuationRequestKey
|
|
454
|
+
&& current.continuationRequestKey !== requestKey;
|
|
317
455
|
const next = {
|
|
318
456
|
...current,
|
|
319
|
-
continuationCount: Number(current.continuationCount || 0) + 1,
|
|
457
|
+
continuationCount: (stale ? 0 : Number(current.continuationCount || 0)) + 1,
|
|
458
|
+
...(requestKey ? { continuationRequestKey: requestKey } : {}),
|
|
320
459
|
lastContinuationAt: Date.now(),
|
|
321
460
|
updatedAt: Date.now(),
|
|
322
461
|
};
|
|
@@ -354,12 +493,31 @@ async function main() {
|
|
|
354
493
|
const state = await readRouteState(projectRoot);
|
|
355
494
|
const ledger = await readExecutionLedger(projectRoot, payload) || {};
|
|
356
495
|
const result = evaluateCompletion({ state, ledger });
|
|
496
|
+
|
|
497
|
+
// Claude Code invokes Stop again after a Stop hook blocks the first stop. Re-blocking
|
|
498
|
+
// that recovery turn creates a self-sustaining loop, so let it end normally instead.
|
|
499
|
+
// If work still lacks evidence, surface the recovery reason to the user rather than
|
|
500
|
+
// silently ending after the automatic continuation.
|
|
501
|
+
if (payload.stop_hook_active === true) {
|
|
502
|
+
if (result.continue || result.capped || result.notify) {
|
|
503
|
+
const recoveryReason = result.reason
|
|
504
|
+
|| 'UKit completion gate reached its continuation limit; unfinished work was not retried again.';
|
|
505
|
+
process.stdout.write(`${JSON.stringify({
|
|
506
|
+
systemMessage: `UKit stopped automatic recovery after one continuation: ${recoveryReason}`,
|
|
507
|
+
})}\n`);
|
|
508
|
+
}
|
|
509
|
+
return;
|
|
510
|
+
}
|
|
511
|
+
|
|
357
512
|
if (result.continue) {
|
|
358
513
|
if (result.finalNotice) await markNotified(projectRoot, payload, ledger);
|
|
359
|
-
else await incrementContinuation(projectRoot, payload, ledger);
|
|
514
|
+
else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null);
|
|
360
515
|
process.stdout.write(`${JSON.stringify({ decision: 'block', reason: result.reason })}\n`);
|
|
361
|
-
} else if (result.capped) {
|
|
516
|
+
} else if (result.capped || result.notify) {
|
|
517
|
+
// Non-blocking endings (non-gated modes, or cap reached after the final notice) must
|
|
518
|
+
// still tell the user what is unfinished — a silent end is indistinguishable from a stall.
|
|
362
519
|
process.stderr.write(`[ukit-completion] ${result.reason}\n`);
|
|
520
|
+
process.stdout.write(`${JSON.stringify({ systemMessage: result.reason })}\n`);
|
|
363
521
|
}
|
|
364
522
|
}
|
|
365
523
|
}
|
|
@@ -38,7 +38,11 @@ function run(payloadText, scriptPaths) {
|
|
|
38
38
|
? path.resolve(path.dirname(firstScript), '../..')
|
|
39
39
|
: (payload.cwd || process.cwd());
|
|
40
40
|
const startedAt = Date.now();
|
|
41
|
-
|
|
41
|
+
// The chain budget must always be able to run EVERY child at its full per-child
|
|
42
|
+
// budget — a fixed total silently starves later scripts once a chain grows
|
|
43
|
+
// (UserPromptSubmit now carries 4 hooks). The floor keeps short chains at 10s.
|
|
44
|
+
const totalBudgetMs = Math.max(TOTAL_BUDGET_MS, scriptPaths.length * CHILD_BUDGET_MS);
|
|
45
|
+
const deadline = startedAt + totalBudgetMs;
|
|
42
46
|
const results = [];
|
|
43
47
|
|
|
44
48
|
for (const scriptPath of scriptPaths) {
|
|
@@ -49,7 +53,7 @@ function run(payloadText, scriptPaths) {
|
|
|
49
53
|
scriptName,
|
|
50
54
|
code: 1,
|
|
51
55
|
stdout: '',
|
|
52
|
-
stderr: `hook chain exceeded its ${
|
|
56
|
+
stderr: `hook chain exceeded its ${totalBudgetMs}ms total budget`,
|
|
53
57
|
killed: true,
|
|
54
58
|
elapsedMs: 0,
|
|
55
59
|
});
|
|
@@ -89,7 +93,7 @@ function run(payloadText, scriptPaths) {
|
|
|
89
93
|
toolName: payload?.tool_name || null,
|
|
90
94
|
toolUseId: payload?.tool_use_id || null,
|
|
91
95
|
elapsedMs,
|
|
92
|
-
budgetMs:
|
|
96
|
+
budgetMs: totalBudgetMs,
|
|
93
97
|
scripts: results.map(({ scriptName, code, killed, elapsedMs: scriptElapsedMs }) => ({
|
|
94
98
|
scriptName,
|
|
95
99
|
code,
|
|
@@ -98,7 +102,7 @@ function run(payloadText, scriptPaths) {
|
|
|
98
102
|
})),
|
|
99
103
|
});
|
|
100
104
|
|
|
101
|
-
return { results, elapsedMs, budgetMs:
|
|
105
|
+
return { results, elapsedMs, budgetMs: totalBudgetMs };
|
|
102
106
|
}
|
|
103
107
|
|
|
104
108
|
try {
|
|
@@ -20,7 +20,7 @@ import {
|
|
|
20
20
|
|
|
21
21
|
export const HOOK_EVENT_MAP = {
|
|
22
22
|
tool_call: {
|
|
23
|
-
'Read|Grep|Glob': [],
|
|
23
|
+
'Read|Grep|Glob': ['sensitive-data-guard.sh'],
|
|
24
24
|
'Edit|Write': [
|
|
25
25
|
'protect-files.sh',
|
|
26
26
|
'stale-spec-guard.sh',
|
|
@@ -32,6 +32,7 @@ export const HOOK_EVENT_MAP = {
|
|
|
32
32
|
Bash: [
|
|
33
33
|
'auto-allow-bash.sh',
|
|
34
34
|
'block-dangerous.sh',
|
|
35
|
+
'sensitive-data-guard.sh',
|
|
35
36
|
'handoff-model-guard.sh',
|
|
36
37
|
'context-hardcap-gate.sh',
|
|
37
38
|
],
|
|
@@ -41,7 +42,7 @@ export const HOOK_EVENT_MAP = {
|
|
|
41
42
|
'Edit|Write': ['post-edit-verify.sh', 'record-execution.sh'],
|
|
42
43
|
Bash: ['compress-output.sh', 'record-execution.sh'],
|
|
43
44
|
},
|
|
44
|
-
before_agent_start: ['skill-router.sh', 'vision-router.sh', 'context-window-guard.sh'],
|
|
45
|
+
before_agent_start: ['sensitive-data-guard.sh', 'skill-router.sh', 'vision-router.sh', 'context-window-guard.sh'],
|
|
45
46
|
'session.compacting': ['reinject-context.sh'],
|
|
46
47
|
session_start: ['auto-prune-bash.sh', 'reset-compact-pressure.sh', 'handoff-resume.sh'],
|
|
47
48
|
};
|
|
@@ -84,6 +85,7 @@ export const FAIL_CLOSED_SCRIPTS = new Set([
|
|
|
84
85
|
'handoff-model-guard.sh',
|
|
85
86
|
'context-hardcap-gate.sh',
|
|
86
87
|
'block-dangerous.sh',
|
|
88
|
+
'sensitive-data-guard.sh',
|
|
87
89
|
]);
|
|
88
90
|
|
|
89
91
|
export const ADVISORY_SCRIPTS = new Set([
|
|
@@ -168,7 +170,15 @@ function translateExecResult(scriptName, execResult) {
|
|
|
168
170
|
|
|
169
171
|
export { translateExecResult };
|
|
170
172
|
|
|
171
|
-
|
|
173
|
+
// Mirrors hook-chain-runner.mjs's TOTAL/CHILD budget constants: the exec timeout must
|
|
174
|
+
// exceed the runner's own chain budget (max(10s, scripts×4s)) plus parse margin, or the
|
|
175
|
+
// bridge orphans the runner mid-chain once a chain grows past 2 scripts.
|
|
176
|
+
const HOOK_CHAIN_BASE_BUDGET_MS = 10000;
|
|
177
|
+
const HOOK_CHAIN_CHILD_BUDGET_MS = 4000;
|
|
178
|
+
function chainExecTimeoutMs(scriptCount) {
|
|
179
|
+
const budget = Math.max(HOOK_CHAIN_BASE_BUDGET_MS, scriptCount * HOOK_CHAIN_CHILD_BUDGET_MS);
|
|
180
|
+
return Math.min(30000, budget + 2000);
|
|
181
|
+
}
|
|
172
182
|
|
|
173
183
|
function recordHookErrorDiagnostic(projectRoot, sessionId, diagnostic) {
|
|
174
184
|
try {
|
|
@@ -226,7 +236,7 @@ export async function runScriptChain(
|
|
|
226
236
|
execResult = await pi.exec(
|
|
227
237
|
nodeExecutable,
|
|
228
238
|
[runnerPath, JSON.stringify(payload), ...scriptPaths],
|
|
229
|
-
{ cwd: projectRoot, timeout:
|
|
239
|
+
{ cwd: projectRoot, timeout: chainExecTimeoutMs(scripts.length) },
|
|
230
240
|
);
|
|
231
241
|
} catch (error) {
|
|
232
242
|
execResult = { code: 1, stdout: '', stderr: error?.message ?? String(error), killed: false };
|
|
@@ -515,7 +525,7 @@ export async function runSessionStop(
|
|
|
515
525
|
if (suppliedLedger === undefined) {
|
|
516
526
|
try {
|
|
517
527
|
if (evaluation.finalNotice) await markNotified(projectRoot, payload, ledger);
|
|
518
|
-
else await incrementContinuation(projectRoot, payload, ledger);
|
|
528
|
+
else await incrementContinuation(projectRoot, payload, ledger, state?.requestKey || null);
|
|
519
529
|
} catch (error) {
|
|
520
530
|
pi.logger?.warn?.(`[UKit] continuation bookkeeping failed open: ${error?.message || error}`);
|
|
521
531
|
}
|
package/templates/AGENTS.md
CHANGED
|
@@ -206,10 +206,10 @@ This is internal orchestration — end users do not need to know about tiers, th
|
|
|
206
206
|
|
|
207
207
|
### Vision lane (capability, not a cost tier)
|
|
208
208
|
|
|
209
|
-
`unic-vision` is a **capability lane**, not a fourth cost tier — it is orthogonal to lite/code/smart above and never appears as a row in the tier table.
|
|
209
|
+
`unic-vision` is a **capability lane**, not a fourth cost tier — it is orthogonal to lite/code/smart above and never appears as a row in the tier table. Whether a mapping can read images is a capability fact, not a provider fact: a mapping that has not **verified** native vision must never guess at image contents — choose native-first when verified, otherwise route to the specialist.
|
|
210
210
|
|
|
211
211
|
- **Gateway detection**: UNIC routing for a Claude Code session is decided ONLY by what changes Claude Code's own outbound endpoint — `ANTHROPIC_BASE_URL` (env var, or the `env.ANTHROPIC_BASE_URL` key in project/home `.claude/settings.json`) containing `unicjsc.com`. Other tools' configs — Codex `config.toml`, Kilo `secrets.json`, or an `OPENAI_BASE_URL` env var — describe a different tool's endpoint entirely and never decide this session's routing.
|
|
212
|
-
- **
|
|
212
|
+
- **Advisory routing (no hard block)**: when an image reaches the prompt, the vision router reminds the session to have `ukit-vision-analyst` analyse it before relying on its contents. Edits are **never blocked** — correctness relies on the model routing images to the analyst instead of guessing.
|
|
213
213
|
- This is internal orchestration — end users never invoke a vision command directly; `ukit install` plus natural language remains the whole surface. No new commands.
|
|
214
214
|
|
|
215
215
|
## Session Start — OpenCode
|
package/templates/CLAUDE.md
CHANGED
|
@@ -206,7 +206,7 @@ This is internal orchestration — end users do not need to know about tiers, th
|
|
|
206
206
|
|
|
207
207
|
### Vision lane (capability, not a cost tier)
|
|
208
208
|
|
|
209
|
-
`unic-vision` is a **capability lane**, not a fourth cost tier — it is orthogonal to lite/code/smart above and never appears as a row in the tier table.
|
|
209
|
+
`unic-vision` is a **capability lane**, not a fourth cost tier — it is orthogonal to lite/code/smart above and never appears as a row in the tier table. Whether a mapping can read images is a capability fact, not a provider fact: a mapping that has not **verified** native vision must never guess at image contents — choose native-first when verified, otherwise route to the specialist.
|
|
210
210
|
|
|
211
211
|
- **Gateway detection**: UNIC routing for a Claude Code session is decided ONLY by what changes Claude Code's own outbound endpoint — `ANTHROPIC_BASE_URL` (env var, or the `env.ANTHROPIC_BASE_URL` key in project/home `.claude/settings.json`) containing `unicjsc.com`. Other tools' configs — Codex `config.toml`, Kilo `secrets.json`, or an `OPENAI_BASE_URL` env var — describe a different tool's endpoint entirely and never decide this session's routing.
|
|
212
212
|
- **Advisory routing (no hard block)**: when an image reaches the prompt, the vision router reminds the session to have `ukit-vision-analyst` analyse it before relying on its contents. Edits are **never blocked** — correctness relies on the model routing images to the analyst instead of guessing.
|